visual-ai-assertions 0.16.0 → 0.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -6
- package/dist/index.cjs +145 -51
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +100 -1
- package/dist/index.d.ts +100 -1
- package/dist/index.js +144 -51
- package/dist/index.js.map +1 -1
- package/package.json +4 -2
package/dist/index.js
CHANGED
|
@@ -5,6 +5,13 @@ var ReasoningEffort = {
|
|
|
5
5
|
HIGH: "high",
|
|
6
6
|
XHIGH: "xhigh"
|
|
7
7
|
};
|
|
8
|
+
var ImageDetail = {
|
|
9
|
+
AUTO: "auto",
|
|
10
|
+
LOW: "low",
|
|
11
|
+
HIGH: "high"
|
|
12
|
+
};
|
|
13
|
+
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
14
|
+
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
8
15
|
var Provider = {
|
|
9
16
|
ANTHROPIC: "anthropic",
|
|
10
17
|
OPENAI: "openai",
|
|
@@ -14,6 +21,7 @@ var Provider = {
|
|
|
14
21
|
var Model = {
|
|
15
22
|
Anthropic: {
|
|
16
23
|
FABLE_5: "claude-fable-5",
|
|
24
|
+
OPUS_5: "claude-opus-5",
|
|
17
25
|
OPUS_4_8: "claude-opus-4-8",
|
|
18
26
|
OPUS_4_7: "claude-opus-4-7",
|
|
19
27
|
OPUS_4_6: "claude-opus-4-6",
|
|
@@ -34,6 +42,8 @@ var Model = {
|
|
|
34
42
|
GPT_5_MINI: "gpt-5-mini"
|
|
35
43
|
},
|
|
36
44
|
Google: {
|
|
45
|
+
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
46
|
+
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
37
47
|
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
38
48
|
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
39
49
|
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
@@ -47,9 +57,11 @@ var Model = {
|
|
|
47
57
|
* recognizes them. All listed models accept image input.
|
|
48
58
|
*/
|
|
49
59
|
OpenRouter: {
|
|
60
|
+
GROK_4_6: "x-ai/grok-4.6",
|
|
50
61
|
GROK_4_5: "x-ai/grok-4.5",
|
|
51
62
|
KIMI_K3: "moonshotai/kimi-k3",
|
|
52
63
|
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
64
|
+
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
53
65
|
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
54
66
|
QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
|
|
55
67
|
}
|
|
@@ -532,6 +544,7 @@ function parseRetryAfter(value) {
|
|
|
532
544
|
// src/providers/anthropic.ts
|
|
533
545
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
534
546
|
Model.Anthropic.FABLE_5,
|
|
547
|
+
Model.Anthropic.OPUS_5,
|
|
535
548
|
Model.Anthropic.OPUS_4_8,
|
|
536
549
|
Model.Anthropic.OPUS_4_7,
|
|
537
550
|
Model.Anthropic.SONNET_5
|
|
@@ -627,7 +640,10 @@ var AnthropicDriver = class {
|
|
|
627
640
|
text,
|
|
628
641
|
usage: {
|
|
629
642
|
inputTokens: message.usage.input_tokens,
|
|
630
|
-
outputTokens: message.usage.output_tokens
|
|
643
|
+
outputTokens: message.usage.output_tokens,
|
|
644
|
+
...message.usage.cache_read_input_tokens !== void 0 && {
|
|
645
|
+
cachedInputTokens: message.usage.cache_read_input_tokens
|
|
646
|
+
}
|
|
631
647
|
}
|
|
632
648
|
};
|
|
633
649
|
} catch (err) {
|
|
@@ -644,23 +660,41 @@ function needsCodeExecution(model) {
|
|
|
644
660
|
return match !== null && match[1] !== void 0 && parseInt(match[1], 10) >= 3;
|
|
645
661
|
}
|
|
646
662
|
var GOOGLE_THINKING_LEVEL = {
|
|
647
|
-
low: "
|
|
648
|
-
medium: "
|
|
649
|
-
high: "
|
|
663
|
+
low: "low",
|
|
664
|
+
medium: "medium",
|
|
665
|
+
high: "high",
|
|
650
666
|
xhigh: "high"
|
|
651
667
|
};
|
|
668
|
+
var GOOGLE_MEDIA_RESOLUTION = {
|
|
669
|
+
low: "MEDIA_RESOLUTION_LOW",
|
|
670
|
+
high: "MEDIA_RESOLUTION_HIGH"
|
|
671
|
+
};
|
|
672
|
+
function toGeminiUsage(um) {
|
|
673
|
+
if (!um) return void 0;
|
|
674
|
+
const thoughts = um.thoughtsTokenCount ?? 0;
|
|
675
|
+
return {
|
|
676
|
+
inputTokens: um.promptTokenCount ?? 0,
|
|
677
|
+
outputTokens: (um.candidatesTokenCount ?? 0) + thoughts,
|
|
678
|
+
...um.thoughtsTokenCount !== void 0 && { reasoningTokens: um.thoughtsTokenCount },
|
|
679
|
+
...um.cachedContentTokenCount !== void 0 && {
|
|
680
|
+
cachedInputTokens: um.cachedContentTokenCount
|
|
681
|
+
}
|
|
682
|
+
};
|
|
683
|
+
}
|
|
652
684
|
var GoogleDriver = class {
|
|
653
685
|
client;
|
|
654
686
|
model;
|
|
655
687
|
maxTokens;
|
|
656
688
|
apiKeyOrEnv;
|
|
657
689
|
reasoningEffort;
|
|
690
|
+
imageDetail;
|
|
658
691
|
constructor(config) {
|
|
659
692
|
this.model = config.model;
|
|
660
693
|
this.maxTokens = config.maxTokens;
|
|
661
694
|
this.client = null;
|
|
662
695
|
this.apiKeyOrEnv = config.apiKey;
|
|
663
696
|
this.reasoningEffort = config.reasoningEffort;
|
|
697
|
+
this.imageDetail = config.imageDetail;
|
|
664
698
|
}
|
|
665
699
|
toGeminiParts(images) {
|
|
666
700
|
return images.map((img) => ({
|
|
@@ -700,6 +734,9 @@ var GoogleDriver = class {
|
|
|
700
734
|
thinkingConfig: {
|
|
701
735
|
thinkingLevel: GOOGLE_THINKING_LEVEL[this.reasoningEffort]
|
|
702
736
|
}
|
|
737
|
+
},
|
|
738
|
+
...this.imageDetail && GOOGLE_MEDIA_RESOLUTION[this.imageDetail] && {
|
|
739
|
+
mediaResolution: GOOGLE_MEDIA_RESOLUTION[this.imageDetail]
|
|
703
740
|
}
|
|
704
741
|
}
|
|
705
742
|
});
|
|
@@ -717,14 +754,9 @@ var GoogleDriver = class {
|
|
|
717
754
|
);
|
|
718
755
|
}
|
|
719
756
|
const text = response.text ?? "";
|
|
720
|
-
const thoughtsTokenCount = response.usageMetadata?.thoughtsTokenCount;
|
|
721
757
|
return {
|
|
722
758
|
text,
|
|
723
|
-
usage: response.usageMetadata
|
|
724
|
-
inputTokens: response.usageMetadata.promptTokenCount ?? 0,
|
|
725
|
-
outputTokens: response.usageMetadata.candidatesTokenCount ?? 0,
|
|
726
|
-
...thoughtsTokenCount !== void 0 && { reasoningTokens: thoughtsTokenCount }
|
|
727
|
-
} : void 0
|
|
759
|
+
usage: toGeminiUsage(response.usageMetadata)
|
|
728
760
|
};
|
|
729
761
|
} catch (err) {
|
|
730
762
|
if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) throw err;
|
|
@@ -755,10 +787,7 @@ var GoogleDriver = class {
|
|
|
755
787
|
return {
|
|
756
788
|
imageData: Buffer.from(imagePart.inlineData.data, "base64"),
|
|
757
789
|
mimeType: imagePart.inlineData.mimeType,
|
|
758
|
-
usage: response.usageMetadata
|
|
759
|
-
inputTokens: response.usageMetadata.promptTokenCount ?? 0,
|
|
760
|
-
outputTokens: response.usageMetadata.candidatesTokenCount ?? 0
|
|
761
|
-
} : void 0
|
|
790
|
+
usage: toGeminiUsage(response.usageMetadata)
|
|
762
791
|
};
|
|
763
792
|
} catch (err) {
|
|
764
793
|
if (err instanceof VisualAIProviderError) throw err;
|
|
@@ -774,12 +803,14 @@ var OpenAIDriver = class {
|
|
|
774
803
|
maxTokens;
|
|
775
804
|
apiKeyOrEnv;
|
|
776
805
|
reasoningEffort;
|
|
806
|
+
imageDetail;
|
|
777
807
|
constructor(config) {
|
|
778
808
|
this.model = config.model;
|
|
779
809
|
this.maxTokens = config.maxTokens;
|
|
780
810
|
this.client = null;
|
|
781
811
|
this.apiKeyOrEnv = config.apiKey;
|
|
782
812
|
this.reasoningEffort = config.reasoningEffort;
|
|
813
|
+
this.imageDetail = config.imageDetail;
|
|
783
814
|
}
|
|
784
815
|
async getClient() {
|
|
785
816
|
if (this.client) return this.client;
|
|
@@ -801,9 +832,11 @@ var OpenAIDriver = class {
|
|
|
801
832
|
}
|
|
802
833
|
async sendMessage(images, prompt, options) {
|
|
803
834
|
const client = await this.getClient();
|
|
835
|
+
const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
|
|
804
836
|
const imageBlocks = images.map((img) => ({
|
|
805
837
|
type: "input_image",
|
|
806
|
-
image_url: `data:${img.mimeType};base64,${img.base64}
|
|
838
|
+
image_url: `data:${img.mimeType};base64,${img.base64}`,
|
|
839
|
+
...detail ? { detail } : {}
|
|
807
840
|
}));
|
|
808
841
|
try {
|
|
809
842
|
const format = options?.responseSchema ? {
|
|
@@ -828,21 +861,23 @@ var OpenAIDriver = class {
|
|
|
828
861
|
}
|
|
829
862
|
const response = await client.responses.create(requestParams);
|
|
830
863
|
if (response.status && response.status !== "completed") {
|
|
831
|
-
const
|
|
864
|
+
const detail2 = response.incomplete_details?.reason ? ` (${response.incomplete_details.reason})` : "";
|
|
832
865
|
throw new VisualAITruncationError(
|
|
833
|
-
`Response truncated: OpenAI returned status "${response.status}"${
|
|
866
|
+
`Response truncated: OpenAI returned status "${response.status}"${detail2}. The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
|
|
834
867
|
response.output_text ?? "",
|
|
835
868
|
this.maxTokens
|
|
836
869
|
);
|
|
837
870
|
}
|
|
838
871
|
const text = response.output_text ?? "";
|
|
839
872
|
const reasoningTokens = response.usage?.output_tokens_details?.reasoning_tokens;
|
|
873
|
+
const cachedInputTokens = response.usage?.input_tokens_details?.cached_tokens;
|
|
840
874
|
return {
|
|
841
875
|
text,
|
|
842
876
|
usage: response.usage ? {
|
|
843
877
|
inputTokens: response.usage.input_tokens,
|
|
844
878
|
outputTokens: response.usage.output_tokens,
|
|
845
|
-
...reasoningTokens !== void 0 && { reasoningTokens }
|
|
879
|
+
...reasoningTokens !== void 0 && { reasoningTokens },
|
|
880
|
+
...cachedInputTokens !== void 0 && { cachedInputTokens }
|
|
846
881
|
} : void 0
|
|
847
882
|
};
|
|
848
883
|
} catch (err) {
|
|
@@ -866,12 +901,14 @@ var OpenRouterDriver = class {
|
|
|
866
901
|
maxTokens;
|
|
867
902
|
apiKeyOrEnv;
|
|
868
903
|
reasoningEffort;
|
|
904
|
+
imageDetail;
|
|
869
905
|
constructor(config) {
|
|
870
906
|
this.model = config.model;
|
|
871
907
|
this.maxTokens = config.maxTokens;
|
|
872
908
|
this.client = null;
|
|
873
909
|
this.apiKeyOrEnv = config.apiKey;
|
|
874
910
|
this.reasoningEffort = config.reasoningEffort;
|
|
911
|
+
this.imageDetail = config.imageDetail;
|
|
875
912
|
}
|
|
876
913
|
async getClient() {
|
|
877
914
|
if (this.client) return this.client;
|
|
@@ -895,9 +932,13 @@ var OpenRouterDriver = class {
|
|
|
895
932
|
}
|
|
896
933
|
async sendMessage(images, prompt, options) {
|
|
897
934
|
const client = await this.getClient();
|
|
935
|
+
const detail = this.imageDetail && this.imageDetail !== "auto" ? this.imageDetail : void 0;
|
|
898
936
|
const imageParts = images.map((img) => ({
|
|
899
937
|
type: "image_url",
|
|
900
|
-
image_url: {
|
|
938
|
+
image_url: {
|
|
939
|
+
url: `data:${img.mimeType};base64,${img.base64}`,
|
|
940
|
+
...detail ? { detail } : {}
|
|
941
|
+
}
|
|
901
942
|
}));
|
|
902
943
|
try {
|
|
903
944
|
const responseFormat = options?.responseSchema ? {
|
|
@@ -938,12 +979,16 @@ var OpenRouterDriver = class {
|
|
|
938
979
|
);
|
|
939
980
|
}
|
|
940
981
|
const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
|
|
982
|
+
const cachedInputTokens = response.usage?.prompt_tokens_details?.cached_tokens;
|
|
983
|
+
const cost = response.usage?.cost;
|
|
941
984
|
return {
|
|
942
985
|
text,
|
|
943
986
|
usage: response.usage ? {
|
|
944
987
|
inputTokens: response.usage.prompt_tokens,
|
|
945
988
|
outputTokens: response.usage.completion_tokens,
|
|
946
|
-
...reasoningTokens !== void 0 && { reasoningTokens }
|
|
989
|
+
...reasoningTokens !== void 0 && { reasoningTokens },
|
|
990
|
+
...cachedInputTokens !== void 0 && { cachedInputTokens },
|
|
991
|
+
...cost !== void 0 && { cost }
|
|
947
992
|
} : void 0
|
|
948
993
|
};
|
|
949
994
|
} catch (err) {
|
|
@@ -1029,6 +1074,8 @@ function resolveConfig(config) {
|
|
|
1029
1074
|
model,
|
|
1030
1075
|
maxTokens,
|
|
1031
1076
|
reasoningEffort: config.reasoningEffort,
|
|
1077
|
+
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1078
|
+
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
1032
1079
|
debug,
|
|
1033
1080
|
debugPrompt,
|
|
1034
1081
|
debugResponse,
|
|
@@ -1043,6 +1090,10 @@ var PRICING_TABLE = {
|
|
|
1043
1090
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1044
1091
|
outputPricePerToken: 50 / PER_MILLION
|
|
1045
1092
|
},
|
|
1093
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
|
|
1094
|
+
inputPricePerToken: 5 / PER_MILLION,
|
|
1095
|
+
outputPricePerToken: 25 / PER_MILLION
|
|
1096
|
+
},
|
|
1046
1097
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_4_8}`]: {
|
|
1047
1098
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1048
1099
|
outputPricePerToken: 25 / PER_MILLION
|
|
@@ -1072,12 +1123,12 @@ var PRICING_TABLE = {
|
|
|
1072
1123
|
outputPricePerToken: 30 / PER_MILLION
|
|
1073
1124
|
},
|
|
1074
1125
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_TERRA}`]: {
|
|
1075
|
-
inputPricePerToken: 2
|
|
1076
|
-
outputPricePerToken:
|
|
1126
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1127
|
+
outputPricePerToken: 12 / PER_MILLION
|
|
1077
1128
|
},
|
|
1078
1129
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_LUNA}`]: {
|
|
1079
|
-
inputPricePerToken:
|
|
1080
|
-
outputPricePerToken:
|
|
1130
|
+
inputPricePerToken: 0.2 / PER_MILLION,
|
|
1131
|
+
outputPricePerToken: 1.2 / PER_MILLION
|
|
1081
1132
|
},
|
|
1082
1133
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_5}`]: {
|
|
1083
1134
|
inputPricePerToken: 5 / PER_MILLION,
|
|
@@ -1107,6 +1158,18 @@ var PRICING_TABLE = {
|
|
|
1107
1158
|
inputPricePerToken: 0.25 / PER_MILLION,
|
|
1108
1159
|
outputPricePerToken: 2 / PER_MILLION
|
|
1109
1160
|
},
|
|
1161
|
+
// Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
|
|
1162
|
+
// on 2027-01-01 (https://blog.google/.../3-8-flash-and-3-8-flash-cyber/).
|
|
1163
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_8_FLASH}`]: {
|
|
1164
|
+
inputPricePerToken: 0.75 / PER_MILLION,
|
|
1165
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1166
|
+
},
|
|
1167
|
+
// Introductory pricing through 2026-12-31; reverts to $1.50/$7.50 per MTok
|
|
1168
|
+
// on 2027-01-01 (https://blog.google/.../introducing-gemini-3-7-flash/).
|
|
1169
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_7_FLASH}`]: {
|
|
1170
|
+
inputPricePerToken: 0.75 / PER_MILLION,
|
|
1171
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1172
|
+
},
|
|
1110
1173
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
|
|
1111
1174
|
inputPricePerToken: 1.5 / PER_MILLION,
|
|
1112
1175
|
outputPricePerToken: 7.5 / PER_MILLION
|
|
@@ -1133,6 +1196,10 @@ var PRICING_TABLE = {
|
|
|
1133
1196
|
},
|
|
1134
1197
|
// OpenRouter passes through upstream per-model pricing (verified 2026-07-22
|
|
1135
1198
|
// against https://openrouter.ai/api/v1/models).
|
|
1199
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
|
|
1200
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1201
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1202
|
+
},
|
|
1136
1203
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
|
|
1137
1204
|
inputPricePerToken: 2 / PER_MILLION,
|
|
1138
1205
|
outputPricePerToken: 6 / PER_MILLION
|
|
@@ -1145,6 +1212,10 @@ var PRICING_TABLE = {
|
|
|
1145
1212
|
inputPricePerToken: 0.82 / PER_MILLION,
|
|
1146
1213
|
outputPricePerToken: 3.75 / PER_MILLION
|
|
1147
1214
|
},
|
|
1215
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_8_MAX}`]: {
|
|
1216
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1217
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1218
|
+
},
|
|
1148
1219
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
|
|
1149
1220
|
inputPricePerToken: 0.32 / PER_MILLION,
|
|
1150
1221
|
outputPricePerToken: 1.28 / PER_MILLION
|
|
@@ -1174,8 +1245,9 @@ function usageLog(config, method, usage) {
|
|
|
1174
1245
|
const costStr = usage.estimatedCost !== void 0 ? `$${usage.estimatedCost.toFixed(6)}` : "unknown";
|
|
1175
1246
|
const reasoningStr = config.reasoningEffort ? `reasoning: ${config.reasoningEffort}` : `reasoning: ${PROVIDER_DEFAULT_REASONING[config.provider]} (provider default)`;
|
|
1176
1247
|
const reasoningTokenStr = usage.reasoningTokens !== void 0 ? ` (${usage.reasoningTokens} reasoning)` : "";
|
|
1248
|
+
const cachedTokenStr = usage.cachedInputTokens !== void 0 ? ` (${usage.cachedInputTokens} cached)` : "";
|
|
1177
1249
|
process.stderr.write(
|
|
1178
|
-
`[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
|
|
1250
|
+
`[visual-ai-assertions] ${method} usage: ${usage.inputTokens} input${cachedTokenStr} + ${usage.outputTokens} output${reasoningTokenStr} tokens (${costStr}) in ${usage.durationSeconds?.toFixed(3) ?? "0.000"}s [${config.model}, ${reasoningStr}]
|
|
1179
1251
|
`
|
|
1180
1252
|
);
|
|
1181
1253
|
}
|
|
@@ -1186,7 +1258,11 @@ function processUsage(method, rawUsage, durationSeconds, config) {
|
|
|
1186
1258
|
inputTokens,
|
|
1187
1259
|
outputTokens,
|
|
1188
1260
|
...rawUsage?.reasoningTokens !== void 0 && { reasoningTokens: rawUsage.reasoningTokens },
|
|
1261
|
+
...rawUsage?.cachedInputTokens !== void 0 && {
|
|
1262
|
+
cachedInputTokens: rawUsage.cachedInputTokens
|
|
1263
|
+
},
|
|
1189
1264
|
estimatedCost: calculateCost(config.provider, config.model, inputTokens, outputTokens),
|
|
1265
|
+
...rawUsage?.cost !== void 0 && { reportedCost: rawUsage.cost },
|
|
1190
1266
|
durationSeconds
|
|
1191
1267
|
};
|
|
1192
1268
|
usageLog(config, method, usage);
|
|
@@ -1230,7 +1306,9 @@ import sharp from "sharp";
|
|
|
1230
1306
|
var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
|
|
1231
1307
|
Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
1232
1308
|
Model.Google.GEMINI_3_5_FLASH,
|
|
1233
|
-
Model.Google.GEMINI_3_6_FLASH
|
|
1309
|
+
Model.Google.GEMINI_3_6_FLASH,
|
|
1310
|
+
Model.Google.GEMINI_3_7_FLASH,
|
|
1311
|
+
Model.Google.GEMINI_3_8_FLASH
|
|
1234
1312
|
]);
|
|
1235
1313
|
async function generateAiDiff(imgA, imgB, model, driver) {
|
|
1236
1314
|
if (!driver.generateImage) {
|
|
@@ -1320,7 +1398,6 @@ var EXTENSION_TO_MIME = {
|
|
|
1320
1398
|
".webp": "image/webp",
|
|
1321
1399
|
".gif": "image/gif"
|
|
1322
1400
|
};
|
|
1323
|
-
var MAX_DIMENSION = 1568;
|
|
1324
1401
|
var URL_FETCH_TIMEOUT_MS = 1e4;
|
|
1325
1402
|
function isSupportedMimeType(value) {
|
|
1326
1403
|
return SUPPORTED_FORMATS.has(value);
|
|
@@ -1344,14 +1421,14 @@ function detectMimeType(data) {
|
|
|
1344
1421
|
}
|
|
1345
1422
|
throw new VisualAIImageError("Unable to detect image format from file content");
|
|
1346
1423
|
}
|
|
1347
|
-
async function resizeIfNeeded(data, mimeType) {
|
|
1424
|
+
async function resizeIfNeeded(data, mimeType, maxDimension) {
|
|
1348
1425
|
if (mimeType === "image/gif") {
|
|
1349
1426
|
return data;
|
|
1350
1427
|
}
|
|
1351
1428
|
if (mimeType === "image/png" && data.length >= 24) {
|
|
1352
1429
|
const width2 = data.readUInt32BE(16);
|
|
1353
1430
|
const height2 = data.readUInt32BE(20);
|
|
1354
|
-
if (width2 <=
|
|
1431
|
+
if (width2 <= maxDimension && height2 <= maxDimension) {
|
|
1355
1432
|
return data;
|
|
1356
1433
|
}
|
|
1357
1434
|
}
|
|
@@ -1359,12 +1436,12 @@ async function resizeIfNeeded(data, mimeType) {
|
|
|
1359
1436
|
const metadata = await pipeline.metadata();
|
|
1360
1437
|
const width = metadata.width ?? 0;
|
|
1361
1438
|
const height = metadata.height ?? 0;
|
|
1362
|
-
if (width <=
|
|
1439
|
+
if (width <= maxDimension && height <= maxDimension) {
|
|
1363
1440
|
return data;
|
|
1364
1441
|
}
|
|
1365
1442
|
return pipeline.resize({
|
|
1366
|
-
width:
|
|
1367
|
-
height:
|
|
1443
|
+
width: maxDimension,
|
|
1444
|
+
height: maxDimension,
|
|
1368
1445
|
fit: "inside",
|
|
1369
1446
|
withoutEnlargement: true
|
|
1370
1447
|
}).toBuffer();
|
|
@@ -1419,7 +1496,7 @@ function loadFromBase64(input) {
|
|
|
1419
1496
|
}
|
|
1420
1497
|
return { data, mimeType: mimeType ?? detectMimeType(data) };
|
|
1421
1498
|
}
|
|
1422
|
-
async function normalizeImage(input) {
|
|
1499
|
+
async function normalizeImage(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1423
1500
|
let data;
|
|
1424
1501
|
let mimeType;
|
|
1425
1502
|
if (Buffer.isBuffer(input)) {
|
|
@@ -1448,7 +1525,7 @@ async function normalizeImage(input) {
|
|
|
1448
1525
|
"Invalid image input: expected Buffer, Uint8Array, file path, URL, or base64 string"
|
|
1449
1526
|
);
|
|
1450
1527
|
}
|
|
1451
|
-
data = await resizeIfNeeded(data, mimeType);
|
|
1528
|
+
data = await resizeIfNeeded(data, mimeType, maxDimension);
|
|
1452
1529
|
let cachedBase64;
|
|
1453
1530
|
return {
|
|
1454
1531
|
data,
|
|
@@ -1752,7 +1829,7 @@ async function probeDurationSeconds(videoPath) {
|
|
|
1752
1829
|
});
|
|
1753
1830
|
});
|
|
1754
1831
|
}
|
|
1755
|
-
async function extractFrames(videoPath, options = {}) {
|
|
1832
|
+
async function extractFrames(videoPath, options = {}, maxDimension = FRAME_MAX_DIMENSION) {
|
|
1756
1833
|
const fps = options.fps ?? DEFAULT_FPS;
|
|
1757
1834
|
const maxFrames = options.maxFrames ?? DEFAULT_MAX_FRAMES;
|
|
1758
1835
|
const maxDurationSeconds = options.maxDurationSeconds ?? DEFAULT_MAX_DURATION_SECONDS;
|
|
@@ -1781,7 +1858,7 @@ async function extractFrames(videoPath, options = {}) {
|
|
|
1781
1858
|
}
|
|
1782
1859
|
const outputDir = await mkdtemp(join2(tmpdir(), "visual-ai-frames-"));
|
|
1783
1860
|
try {
|
|
1784
|
-
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${
|
|
1861
|
+
const filter = `fps=${fps},scale='if(gt(iw,ih),min(${maxDimension},iw),-2)':'if(gt(iw,ih),-2,min(${maxDimension},ih))':flags=area`;
|
|
1785
1862
|
await new Promise((resolve2, reject) => {
|
|
1786
1863
|
let settled = false;
|
|
1787
1864
|
const cmd = ffmpeg(videoPath);
|
|
@@ -1879,7 +1956,7 @@ function isFramesInput(input) {
|
|
|
1879
1956
|
function isTimestampedFrameInput(frame) {
|
|
1880
1957
|
return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
|
|
1881
1958
|
}
|
|
1882
|
-
async function normalizeFrames(input) {
|
|
1959
|
+
async function normalizeFrames(input, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1883
1960
|
const rawFrames = input.frames;
|
|
1884
1961
|
const fps = input.fps ?? DEFAULT_FPS;
|
|
1885
1962
|
if (rawFrames.length === 0) {
|
|
@@ -1904,7 +1981,7 @@ async function normalizeFrames(input) {
|
|
|
1904
1981
|
`Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
|
|
1905
1982
|
);
|
|
1906
1983
|
}
|
|
1907
|
-
const image = await normalizeImage(imageInput);
|
|
1984
|
+
const image = await normalizeImage(imageInput, maxDimension);
|
|
1908
1985
|
return {
|
|
1909
1986
|
data: image.data,
|
|
1910
1987
|
mimeType: image.mimeType,
|
|
@@ -1920,14 +1997,14 @@ async function normalizeFrames(input) {
|
|
|
1920
1997
|
await saveDebugFrames(frames);
|
|
1921
1998
|
return { kind: "video", frames, durationSeconds };
|
|
1922
1999
|
}
|
|
1923
|
-
async function normalizeMedia(input, videoOptions) {
|
|
2000
|
+
async function normalizeMedia(input, videoOptions, maxDimension = DEFAULT_MAX_IMAGE_DIMENSION) {
|
|
1924
2001
|
if (isFramesInput(input)) {
|
|
1925
|
-
return normalizeFrames(input);
|
|
2002
|
+
return normalizeFrames(input, maxDimension);
|
|
1926
2003
|
}
|
|
1927
2004
|
if (isVideoInput(input)) {
|
|
1928
2005
|
const { path, cleanup } = await resolveVideoToPath(input);
|
|
1929
2006
|
try {
|
|
1930
|
-
const { frames, durationSeconds } = await extractFrames(path, videoOptions);
|
|
2007
|
+
const { frames, durationSeconds } = await extractFrames(path, videoOptions, maxDimension);
|
|
1931
2008
|
await saveDebugFrames(frames);
|
|
1932
2009
|
return { kind: "video", frames, durationSeconds };
|
|
1933
2010
|
} finally {
|
|
@@ -1937,7 +2014,7 @@ async function normalizeMedia(input, videoOptions) {
|
|
|
1937
2014
|
}
|
|
1938
2015
|
}
|
|
1939
2016
|
}
|
|
1940
|
-
const image = await normalizeImage(input);
|
|
2017
|
+
const image = await normalizeImage(input, maxDimension);
|
|
1941
2018
|
return { kind: "image", image };
|
|
1942
2019
|
}
|
|
1943
2020
|
|
|
@@ -1978,7 +2055,17 @@ var UsageInfoSchema = z.object({
|
|
|
1978
2055
|
outputTokens: z.number(),
|
|
1979
2056
|
/** Reasoning/thinking tokens consumed by the model (informational, typically included within outputTokens). */
|
|
1980
2057
|
reasoningTokens: z.number().optional(),
|
|
2058
|
+
/**
|
|
2059
|
+
* Prompt tokens served from the provider's cache, when reported. Informational
|
|
2060
|
+
* only — `estimatedCost` does not apply a cache discount, because providers
|
|
2061
|
+
* differ on whether these are counted inside `inputTokens` (OpenAI, OpenRouter,
|
|
2062
|
+
* Google) or billed as a separate bucket alongside it (Anthropic).
|
|
2063
|
+
*/
|
|
2064
|
+
cachedInputTokens: z.number().optional(),
|
|
2065
|
+
/** Cost in USD from the library's local pricing table (inputTokens/outputTokens × per-model rates). */
|
|
1981
2066
|
estimatedCost: z.number().optional(),
|
|
2067
|
+
/** Actual cost in USD reported by the provider itself, when available (OpenRouter). Authoritative over `estimatedCost`. */
|
|
2068
|
+
reportedCost: z.number().optional(),
|
|
1982
2069
|
durationSeconds: z.number().nonnegative().optional()
|
|
1983
2070
|
});
|
|
1984
2071
|
var BaseResultSchema = z.object({
|
|
@@ -2118,16 +2205,18 @@ function visualAI(config = {}) {
|
|
|
2118
2205
|
apiKey: resolvedConfig.apiKey,
|
|
2119
2206
|
model: resolvedConfig.model,
|
|
2120
2207
|
maxTokens: resolvedConfig.maxTokens,
|
|
2121
|
-
reasoningEffort: resolvedConfig.reasoningEffort
|
|
2208
|
+
reasoningEffort: resolvedConfig.reasoningEffort,
|
|
2209
|
+
imageDetail: resolvedConfig.imageDetail
|
|
2122
2210
|
};
|
|
2123
2211
|
const driver = createDriver(resolvedConfig.provider, driverConfig);
|
|
2212
|
+
const maxImageDimension = resolvedConfig.maxImageDimension;
|
|
2124
2213
|
async function checkElementsVisibility(image, elements, visible, options) {
|
|
2125
2214
|
const methodName = visible ? "elementsVisible" : "elementsHidden";
|
|
2126
2215
|
if (elements.length === 0) {
|
|
2127
2216
|
throw new VisualAIConfigError(`At least one element is required for ${methodName}()`);
|
|
2128
2217
|
}
|
|
2129
2218
|
return withErrorDebug(resolvedConfig, methodName, async () => {
|
|
2130
|
-
const img = await normalizeImage(image);
|
|
2219
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2131
2220
|
const prompt = buildElementsVisibilityPrompt(elements, visible, options);
|
|
2132
2221
|
debugLog(resolvedConfig, `${methodName} prompt`, prompt, "prompt");
|
|
2133
2222
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2146,7 +2235,7 @@ function visualAI(config = {}) {
|
|
|
2146
2235
|
throw new VisualAIConfigError("At least one statement is required for check()");
|
|
2147
2236
|
}
|
|
2148
2237
|
return withErrorDebug(resolvedConfig, "check", async () => {
|
|
2149
|
-
const media = await normalizeMedia(input, options?.video);
|
|
2238
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension);
|
|
2150
2239
|
const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
|
|
2151
2240
|
const prompt = buildCheckPrompt(stmts, {
|
|
2152
2241
|
instructions: options?.instructions,
|
|
@@ -2165,7 +2254,7 @@ function visualAI(config = {}) {
|
|
|
2165
2254
|
},
|
|
2166
2255
|
async ask(input, userPrompt, options) {
|
|
2167
2256
|
return withErrorDebug(resolvedConfig, "ask", async () => {
|
|
2168
|
-
const media = await normalizeMedia(input, options?.video);
|
|
2257
|
+
const media = await normalizeMedia(input, options?.video, maxImageDimension);
|
|
2169
2258
|
const { images, mediaContext, framesMetadata } = mediaToProviderInputs(media);
|
|
2170
2259
|
const prompt = buildAskPrompt(userPrompt, {
|
|
2171
2260
|
instructions: options?.instructions,
|
|
@@ -2184,7 +2273,10 @@ function visualAI(config = {}) {
|
|
|
2184
2273
|
},
|
|
2185
2274
|
async compare(imageA, imageB, options) {
|
|
2186
2275
|
return withErrorDebug(resolvedConfig, "compare", async () => {
|
|
2187
|
-
const [imgA, imgB] = await Promise.all([
|
|
2276
|
+
const [imgA, imgB] = await Promise.all([
|
|
2277
|
+
normalizeImage(imageA, maxImageDimension),
|
|
2278
|
+
normalizeImage(imageB, maxImageDimension)
|
|
2279
|
+
]);
|
|
2188
2280
|
const prompt = buildComparePrompt({
|
|
2189
2281
|
userPrompt: options?.prompt,
|
|
2190
2282
|
instructions: options?.instructions
|
|
@@ -2222,7 +2314,7 @@ function visualAI(config = {}) {
|
|
|
2222
2314
|
},
|
|
2223
2315
|
async accessibility(image, options) {
|
|
2224
2316
|
return withErrorDebug(resolvedConfig, "accessibility", async () => {
|
|
2225
|
-
const img = await normalizeImage(image);
|
|
2317
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2226
2318
|
const prompt = buildAccessibilityPrompt(options);
|
|
2227
2319
|
debugLog(resolvedConfig, "accessibility prompt", prompt, "prompt");
|
|
2228
2320
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2241,7 +2333,7 @@ function visualAI(config = {}) {
|
|
|
2241
2333
|
},
|
|
2242
2334
|
async layout(image, options) {
|
|
2243
2335
|
return withErrorDebug(resolvedConfig, "layout", async () => {
|
|
2244
|
-
const img = await normalizeImage(image);
|
|
2336
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2245
2337
|
const prompt = buildLayoutPrompt(options);
|
|
2246
2338
|
debugLog(resolvedConfig, "layout prompt", prompt, "prompt");
|
|
2247
2339
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2255,7 +2347,7 @@ function visualAI(config = {}) {
|
|
|
2255
2347
|
},
|
|
2256
2348
|
async pageLoad(image, options) {
|
|
2257
2349
|
return withErrorDebug(resolvedConfig, "pageLoad", async () => {
|
|
2258
|
-
const img = await normalizeImage(image);
|
|
2350
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2259
2351
|
const prompt = buildPageLoadPrompt(options);
|
|
2260
2352
|
debugLog(resolvedConfig, "pageLoad prompt", prompt, "prompt");
|
|
2261
2353
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2269,7 +2361,7 @@ function visualAI(config = {}) {
|
|
|
2269
2361
|
},
|
|
2270
2362
|
async content(image, options) {
|
|
2271
2363
|
return withErrorDebug(resolvedConfig, "content", async () => {
|
|
2272
|
-
const img = await normalizeImage(image);
|
|
2364
|
+
const img = await normalizeImage(image, maxImageDimension);
|
|
2273
2365
|
const prompt = buildContentPrompt(options);
|
|
2274
2366
|
debugLog(resolvedConfig, "content prompt", prompt, "prompt");
|
|
2275
2367
|
const response = await timedSendMessage(driver, [img], prompt, checkSchemaOptions);
|
|
@@ -2347,6 +2439,7 @@ export {
|
|
|
2347
2439
|
ConfidenceSchema,
|
|
2348
2440
|
Content,
|
|
2349
2441
|
DEFAULT_MODELS,
|
|
2442
|
+
ImageDetail,
|
|
2350
2443
|
IssueCategorySchema,
|
|
2351
2444
|
IssuePrioritySchema,
|
|
2352
2445
|
IssueSchema,
|