visual-ai-assertions 0.25.0 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +146 -76
- package/dist/index.cjs +225 -137
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +47 -13
- package/dist/index.d.ts +47 -13
- package/dist/index.js +225 -137
- package/dist/index.js.map +1 -1
- package/package.json +2 -5
package/dist/index.cjs
CHANGED
|
@@ -67,135 +67,6 @@ __export(index_exports, {
|
|
|
67
67
|
});
|
|
68
68
|
module.exports = __toCommonJS(index_exports);
|
|
69
69
|
|
|
70
|
-
// src/constants.ts
|
|
71
|
-
var ReasoningEffort = {
|
|
72
|
-
LOW: "low",
|
|
73
|
-
MEDIUM: "medium",
|
|
74
|
-
HIGH: "high",
|
|
75
|
-
XHIGH: "xhigh"
|
|
76
|
-
};
|
|
77
|
-
var ImageDetail = {
|
|
78
|
-
AUTO: "auto",
|
|
79
|
-
LOW: "low",
|
|
80
|
-
HIGH: "high"
|
|
81
|
-
};
|
|
82
|
-
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
83
|
-
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
84
|
-
var Provider = {
|
|
85
|
-
ANTHROPIC: "anthropic",
|
|
86
|
-
OPENAI: "openai",
|
|
87
|
-
GOOGLE: "google",
|
|
88
|
-
OPENROUTER: "openrouter"
|
|
89
|
-
};
|
|
90
|
-
var Model = {
|
|
91
|
-
Anthropic: {
|
|
92
|
-
FABLE_5_1: "claude-fable-5-1",
|
|
93
|
-
FABLE_5: "claude-fable-5",
|
|
94
|
-
OPUS_5: "claude-opus-5",
|
|
95
|
-
OPUS_4_8: "claude-opus-4-8",
|
|
96
|
-
OPUS_4_7: "claude-opus-4-7",
|
|
97
|
-
OPUS_4_6: "claude-opus-4-6",
|
|
98
|
-
SONNET_5: "claude-sonnet-5",
|
|
99
|
-
SONNET_4_6: "claude-sonnet-4-6",
|
|
100
|
-
HAIKU_4_5: "claude-haiku-4-5"
|
|
101
|
-
},
|
|
102
|
-
OpenAI: {
|
|
103
|
-
GPT_6_ASTRA: "gpt-6-astra",
|
|
104
|
-
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
105
|
-
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
106
|
-
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
107
|
-
GPT_5_5: "gpt-5.5",
|
|
108
|
-
GPT_5_4: "gpt-5.4",
|
|
109
|
-
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
110
|
-
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
111
|
-
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
112
|
-
GPT_5_2: "gpt-5.2",
|
|
113
|
-
GPT_5_MINI: "gpt-5-mini"
|
|
114
|
-
},
|
|
115
|
-
Google: {
|
|
116
|
-
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
117
|
-
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
118
|
-
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
119
|
-
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
120
|
-
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
121
|
-
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
122
|
-
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
123
|
-
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
124
|
-
},
|
|
125
|
-
/**
|
|
126
|
-
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
127
|
-
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
128
|
-
* recognizes them. All listed models accept image input.
|
|
129
|
-
*/
|
|
130
|
-
OpenRouter: {
|
|
131
|
-
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
132
|
-
GROK_4_6: "x-ai/grok-4.6",
|
|
133
|
-
GROK_4_5: "x-ai/grok-4.5",
|
|
134
|
-
KIMI_K3: "moonshotai/kimi-k3",
|
|
135
|
-
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
136
|
-
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
137
|
-
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
138
|
-
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
139
|
-
GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
|
|
140
|
-
}
|
|
141
|
-
};
|
|
142
|
-
var DEFAULT_MODELS = {
|
|
143
|
-
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
|
|
144
|
-
[Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
|
|
145
|
-
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
146
|
-
[Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
|
|
147
|
-
};
|
|
148
|
-
var DEFAULT_MAX_TOKENS = 4096;
|
|
149
|
-
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
150
|
-
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
151
|
-
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
152
|
-
Model.OpenAI.GPT_6_ASTRA
|
|
153
|
-
]);
|
|
154
|
-
var MODEL_TO_PROVIDER = new Map([
|
|
155
|
-
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
156
|
-
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
157
|
-
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
158
|
-
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
159
|
-
]);
|
|
160
|
-
var VALID_PROVIDERS = Object.values(Provider);
|
|
161
|
-
var PROVIDER_DEFAULT_REASONING = {
|
|
162
|
-
openai: "medium",
|
|
163
|
-
anthropic: "off",
|
|
164
|
-
google: "off",
|
|
165
|
-
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
166
|
-
openrouter: "off"
|
|
167
|
-
};
|
|
168
|
-
var Content = {
|
|
169
|
-
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
170
|
-
PLACEHOLDER_TEXT: "placeholder-text",
|
|
171
|
-
/** Detects error messages, banners, stack traces, or error codes */
|
|
172
|
-
ERROR_MESSAGES: "error-messages",
|
|
173
|
-
/** Detects broken image icons or failed-to-load image indicators */
|
|
174
|
-
BROKEN_IMAGES: "broken-images",
|
|
175
|
-
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
176
|
-
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
177
|
-
};
|
|
178
|
-
var Layout = {
|
|
179
|
-
/** Detects elements that unintentionally overlap each other */
|
|
180
|
-
OVERLAP: "overlap",
|
|
181
|
-
/** Detects content cut off or extending beyond container boundaries */
|
|
182
|
-
OVERFLOW: "overflow",
|
|
183
|
-
/** Detects inconsistent alignment of text, images, and UI components */
|
|
184
|
-
ALIGNMENT: "alignment"
|
|
185
|
-
};
|
|
186
|
-
var Accessibility = {
|
|
187
|
-
/** Detects insufficient color contrast between text and backgrounds */
|
|
188
|
-
CONTRAST: "contrast",
|
|
189
|
-
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
190
|
-
READABILITY: "readability",
|
|
191
|
-
/** Detects interactive elements that are not visually distinct */
|
|
192
|
-
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
193
|
-
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
194
|
-
COLOR_BLINDNESS: "color-blindness",
|
|
195
|
-
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
196
|
-
COLOR_ALONE: "color-alone"
|
|
197
|
-
};
|
|
198
|
-
|
|
199
70
|
// src/errors.ts
|
|
200
71
|
var VisualAIError = class extends Error {
|
|
201
72
|
code;
|
|
@@ -562,7 +433,7 @@ function visibleRole(finalState, requireCorrectRendering) {
|
|
|
562
433
|
var ELEMENTS_VISIBLE_CLIPPING_RULES = [
|
|
563
434
|
"When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
|
|
564
435
|
"An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
|
|
565
|
-
"
|
|
436
|
+
"The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
|
|
566
437
|
];
|
|
567
438
|
var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
|
|
568
439
|
var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
|
|
@@ -593,6 +464,145 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
|
|
|
593
464
|
return buildCheckPrompt(statements, { role, instructions });
|
|
594
465
|
}
|
|
595
466
|
|
|
467
|
+
// src/constants.ts
|
|
468
|
+
var ReasoningEffort = {
|
|
469
|
+
MINIMAL: "minimal",
|
|
470
|
+
LOW: "low",
|
|
471
|
+
MEDIUM: "medium",
|
|
472
|
+
HIGH: "high",
|
|
473
|
+
XHIGH: "xhigh"
|
|
474
|
+
};
|
|
475
|
+
var ImageDetail = {
|
|
476
|
+
AUTO: "auto",
|
|
477
|
+
LOW: "low",
|
|
478
|
+
HIGH: "high"
|
|
479
|
+
};
|
|
480
|
+
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
481
|
+
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
482
|
+
var Provider = {
|
|
483
|
+
ANTHROPIC: "anthropic",
|
|
484
|
+
OPENAI: "openai",
|
|
485
|
+
GOOGLE: "google",
|
|
486
|
+
OPENROUTER: "openrouter"
|
|
487
|
+
};
|
|
488
|
+
var Model = {
|
|
489
|
+
Anthropic: {
|
|
490
|
+
FABLE_5_1: "claude-fable-5-1",
|
|
491
|
+
FABLE_5: "claude-fable-5",
|
|
492
|
+
OPUS_5_5: "claude-opus-5-5",
|
|
493
|
+
OPUS_5: "claude-opus-5",
|
|
494
|
+
OPUS_4_8: "claude-opus-4-8",
|
|
495
|
+
OPUS_4_7: "claude-opus-4-7",
|
|
496
|
+
OPUS_4_6: "claude-opus-4-6",
|
|
497
|
+
SONNET_5_5: "claude-sonnet-5-5",
|
|
498
|
+
SONNET_5: "claude-sonnet-5",
|
|
499
|
+
SONNET_4_6: "claude-sonnet-4-6",
|
|
500
|
+
HAIKU_5_5: "claude-haiku-5-5",
|
|
501
|
+
HAIKU_4_5: "claude-haiku-4-5"
|
|
502
|
+
},
|
|
503
|
+
OpenAI: {
|
|
504
|
+
GPT_6_ASTRA: "gpt-6-astra",
|
|
505
|
+
GPT_6_1_SOL: "gpt-6.1-sol",
|
|
506
|
+
GPT_6_SOL: "gpt-6-sol",
|
|
507
|
+
GPT_6_LUNA: "gpt-6-luna",
|
|
508
|
+
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
509
|
+
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
510
|
+
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
511
|
+
GPT_5_5: "gpt-5.5",
|
|
512
|
+
GPT_5_4: "gpt-5.4",
|
|
513
|
+
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
514
|
+
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
515
|
+
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
516
|
+
GPT_5_2: "gpt-5.2",
|
|
517
|
+
GPT_5_MINI: "gpt-5-mini"
|
|
518
|
+
},
|
|
519
|
+
Google: {
|
|
520
|
+
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
521
|
+
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
522
|
+
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
523
|
+
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
524
|
+
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
525
|
+
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
526
|
+
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
527
|
+
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
528
|
+
},
|
|
529
|
+
/**
|
|
530
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
531
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
532
|
+
* recognizes them. All listed models accept image input.
|
|
533
|
+
*/
|
|
534
|
+
OpenRouter: {
|
|
535
|
+
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
536
|
+
GROK_4_7: "x-ai/grok-4.7",
|
|
537
|
+
GROK_4_6: "x-ai/grok-4.6",
|
|
538
|
+
GROK_4_5: "x-ai/grok-4.5",
|
|
539
|
+
KIMI_K3: "moonshotai/kimi-k3",
|
|
540
|
+
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
541
|
+
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
542
|
+
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
543
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
544
|
+
GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
|
|
545
|
+
MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
|
|
546
|
+
}
|
|
547
|
+
};
|
|
548
|
+
var DEFAULT_MODELS = {
|
|
549
|
+
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
|
|
550
|
+
[Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
|
|
551
|
+
[Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
|
|
552
|
+
[Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
|
|
553
|
+
};
|
|
554
|
+
var DEFAULT_MAX_TOKENS = 4096;
|
|
555
|
+
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
556
|
+
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
557
|
+
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
558
|
+
Model.OpenRouter.QWEN_3_8_MAX,
|
|
559
|
+
Model.OpenRouter.QWEN_3_7_PLUS
|
|
560
|
+
]);
|
|
561
|
+
var MODEL_TO_PROVIDER = new Map([
|
|
562
|
+
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
563
|
+
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
564
|
+
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
565
|
+
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
566
|
+
]);
|
|
567
|
+
var VALID_PROVIDERS = Object.values(Provider);
|
|
568
|
+
var PROVIDER_DEFAULT_REASONING = {
|
|
569
|
+
openai: "medium",
|
|
570
|
+
anthropic: "off",
|
|
571
|
+
google: "off",
|
|
572
|
+
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
573
|
+
openrouter: "off"
|
|
574
|
+
};
|
|
575
|
+
var Content = {
|
|
576
|
+
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
577
|
+
PLACEHOLDER_TEXT: "placeholder-text",
|
|
578
|
+
/** Detects error messages, banners, stack traces, or error codes */
|
|
579
|
+
ERROR_MESSAGES: "error-messages",
|
|
580
|
+
/** Detects broken image icons or failed-to-load image indicators */
|
|
581
|
+
BROKEN_IMAGES: "broken-images",
|
|
582
|
+
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
583
|
+
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
584
|
+
};
|
|
585
|
+
var Layout = {
|
|
586
|
+
/** Detects elements that unintentionally overlap each other */
|
|
587
|
+
OVERLAP: "overlap",
|
|
588
|
+
/** Detects content cut off or extending beyond container boundaries */
|
|
589
|
+
OVERFLOW: "overflow",
|
|
590
|
+
/** Detects inconsistent alignment of text, images, and UI components */
|
|
591
|
+
ALIGNMENT: "alignment"
|
|
592
|
+
};
|
|
593
|
+
var Accessibility = {
|
|
594
|
+
/** Detects insufficient color contrast between text and backgrounds */
|
|
595
|
+
CONTRAST: "contrast",
|
|
596
|
+
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
597
|
+
READABILITY: "readability",
|
|
598
|
+
/** Detects interactive elements that are not visually distinct */
|
|
599
|
+
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
600
|
+
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
601
|
+
COLOR_BLINDNESS: "color-blindness",
|
|
602
|
+
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
603
|
+
COLOR_ALONE: "color-alone"
|
|
604
|
+
};
|
|
605
|
+
|
|
596
606
|
// src/templates/accessibility.ts
|
|
597
607
|
var ALL_CHECKS = Object.values(Accessibility);
|
|
598
608
|
var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
|
|
@@ -702,17 +712,23 @@ function parseRetryAfter(value) {
|
|
|
702
712
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
703
713
|
Model.Anthropic.FABLE_5_1,
|
|
704
714
|
Model.Anthropic.FABLE_5,
|
|
715
|
+
Model.Anthropic.OPUS_5_5,
|
|
705
716
|
Model.Anthropic.OPUS_5,
|
|
706
717
|
Model.Anthropic.OPUS_4_8,
|
|
707
718
|
Model.Anthropic.OPUS_4_7,
|
|
708
|
-
Model.Anthropic.
|
|
719
|
+
Model.Anthropic.SONNET_5_5,
|
|
720
|
+
Model.Anthropic.SONNET_5,
|
|
721
|
+
Model.Anthropic.HAIKU_5_5
|
|
709
722
|
]);
|
|
710
723
|
function mapEffort(level, model) {
|
|
724
|
+
if (level === "minimal") return "low";
|
|
711
725
|
if (level !== "xhigh") return level;
|
|
712
726
|
return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
|
|
713
727
|
}
|
|
714
728
|
var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
|
|
715
729
|
var EFFORT_TO_BUDGET_TOKENS = {
|
|
730
|
+
// 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
|
|
731
|
+
minimal: 1024,
|
|
716
732
|
low: 1024,
|
|
717
733
|
medium: 4096,
|
|
718
734
|
high: 8192,
|
|
@@ -829,6 +845,9 @@ function sleep(ms) {
|
|
|
829
845
|
return new Promise((resolve2) => setTimeout(resolve2, ms));
|
|
830
846
|
}
|
|
831
847
|
var GOOGLE_THINKING_LEVEL = {
|
|
848
|
+
// Gemini does define a "minimal" thinking level, but some models reject it
|
|
849
|
+
// (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
|
|
850
|
+
minimal: "low",
|
|
832
851
|
low: "low",
|
|
833
852
|
medium: "medium",
|
|
834
853
|
high: "high",
|
|
@@ -1146,6 +1165,9 @@ var OpenAIDriver = class {
|
|
|
1146
1165
|
// src/providers/openrouter.ts
|
|
1147
1166
|
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
1148
1167
|
var OPENROUTER_REASONING_EFFORT = {
|
|
1168
|
+
// OpenRouter normalizes upstream vendors to low/medium/high only, so
|
|
1169
|
+
// "minimal" has no native equivalent and clamps to the floor.
|
|
1170
|
+
minimal: "low",
|
|
1149
1171
|
low: "low",
|
|
1150
1172
|
medium: "medium",
|
|
1151
1173
|
high: "high",
|
|
@@ -1305,10 +1327,20 @@ function parseBooleanEnv(envName, value) {
|
|
|
1305
1327
|
`Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
|
|
1306
1328
|
);
|
|
1307
1329
|
}
|
|
1330
|
+
function parseReasoningEffortEnv(envName, value) {
|
|
1331
|
+
if (value === void 0 || value === "") return void 0;
|
|
1332
|
+
const levels = Object.values(ReasoningEffort);
|
|
1333
|
+
const lower = value.toLowerCase();
|
|
1334
|
+
if (levels.includes(lower)) return lower;
|
|
1335
|
+
throw new VisualAIConfigError(
|
|
1336
|
+
`Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
|
|
1337
|
+
);
|
|
1338
|
+
}
|
|
1308
1339
|
var debugDeprecationWarned = false;
|
|
1309
1340
|
function resolveConfig(config) {
|
|
1310
1341
|
const provider = resolveProvider(config);
|
|
1311
1342
|
const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
|
|
1343
|
+
const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
|
|
1312
1344
|
const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
|
|
1313
1345
|
const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
|
|
1314
1346
|
const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
|
|
@@ -1326,12 +1358,12 @@ function resolveConfig(config) {
|
|
|
1326
1358
|
}
|
|
1327
1359
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
1328
1360
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
1329
|
-
const effortNeedsLargeBudget =
|
|
1361
|
+
const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
|
|
1330
1362
|
const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
|
|
1331
1363
|
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
|
|
1332
1364
|
maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
|
|
1333
1365
|
if (debug) {
|
|
1334
|
-
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${
|
|
1366
|
+
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
|
|
1335
1367
|
process.stderr.write(
|
|
1336
1368
|
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
|
|
1337
1369
|
`
|
|
@@ -1343,7 +1375,7 @@ function resolveConfig(config) {
|
|
|
1343
1375
|
apiKey: config.apiKey,
|
|
1344
1376
|
model,
|
|
1345
1377
|
maxTokens,
|
|
1346
|
-
reasoningEffort
|
|
1378
|
+
reasoningEffort,
|
|
1347
1379
|
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1348
1380
|
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
1349
1381
|
timeout: config.timeout,
|
|
@@ -1366,6 +1398,10 @@ var PRICING_TABLE = {
|
|
|
1366
1398
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1367
1399
|
outputPricePerToken: 50 / PER_MILLION
|
|
1368
1400
|
},
|
|
1401
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
|
|
1402
|
+
inputPricePerToken: 4 / PER_MILLION,
|
|
1403
|
+
outputPricePerToken: 20 / PER_MILLION
|
|
1404
|
+
},
|
|
1369
1405
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
|
|
1370
1406
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1371
1407
|
outputPricePerToken: 25 / PER_MILLION
|
|
@@ -1374,6 +1410,10 @@ var PRICING_TABLE = {
|
|
|
1374
1410
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1375
1411
|
outputPricePerToken: 25 / PER_MILLION
|
|
1376
1412
|
},
|
|
1413
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
|
|
1414
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1415
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1416
|
+
},
|
|
1377
1417
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
|
|
1378
1418
|
inputPricePerToken: 3 / PER_MILLION,
|
|
1379
1419
|
outputPricePerToken: 15 / PER_MILLION
|
|
@@ -1390,6 +1430,12 @@ var PRICING_TABLE = {
|
|
|
1390
1430
|
inputPricePerToken: 3 / PER_MILLION,
|
|
1391
1431
|
outputPricePerToken: 15 / PER_MILLION
|
|
1392
1432
|
},
|
|
1433
|
+
// Prompts above 100K tokens bill at $0.50/$2.50, far beyond screenshot-sized
|
|
1434
|
+
// calls. Cached input is $0.01/MTok and cache writes $0.125/MTok (not modelled).
|
|
1435
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.HAIKU_5_5}`]: {
|
|
1436
|
+
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1437
|
+
outputPricePerToken: 0.5 / PER_MILLION
|
|
1438
|
+
},
|
|
1393
1439
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.HAIKU_4_5}`]: {
|
|
1394
1440
|
inputPricePerToken: 1 / PER_MILLION,
|
|
1395
1441
|
outputPricePerToken: 5 / PER_MILLION
|
|
@@ -1400,6 +1446,22 @@ var PRICING_TABLE = {
|
|
|
1400
1446
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1401
1447
|
outputPricePerToken: 50 / PER_MILLION
|
|
1402
1448
|
},
|
|
1449
|
+
// Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
|
|
1450
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
|
|
1451
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1452
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1453
|
+
},
|
|
1454
|
+
// Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
|
|
1455
|
+
// bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
|
|
1456
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
|
|
1457
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1458
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1459
|
+
},
|
|
1460
|
+
// Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
|
|
1461
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
|
|
1462
|
+
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1463
|
+
outputPricePerToken: 0.5 / PER_MILLION
|
|
1464
|
+
},
|
|
1403
1465
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
|
|
1404
1466
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1405
1467
|
outputPricePerToken: 30 / PER_MILLION
|
|
@@ -1496,6 +1558,10 @@ var PRICING_TABLE = {
|
|
|
1496
1558
|
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1497
1559
|
outputPricePerToken: 0.2 / PER_MILLION
|
|
1498
1560
|
},
|
|
1561
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
|
|
1562
|
+
inputPricePerToken: 1.6 / PER_MILLION,
|
|
1563
|
+
outputPricePerToken: 4.8 / PER_MILLION
|
|
1564
|
+
},
|
|
1499
1565
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
|
|
1500
1566
|
inputPricePerToken: 2 / PER_MILLION,
|
|
1501
1567
|
outputPricePerToken: 6 / PER_MILLION
|
|
@@ -1529,6 +1595,13 @@ var PRICING_TABLE = {
|
|
|
1529
1595
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
|
|
1530
1596
|
inputPricePerToken: 0.15 / PER_MILLION,
|
|
1531
1597
|
outputPricePerToken: 0.5 / PER_MILLION
|
|
1598
|
+
},
|
|
1599
|
+
// Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
|
|
1600
|
+
// upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
|
|
1601
|
+
// is $0.0036/MTok, not modelled (no provider gets a cache discount here).
|
|
1602
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
|
|
1603
|
+
inputPricePerToken: 0.435 / PER_MILLION,
|
|
1604
|
+
outputPricePerToken: 0.87 / PER_MILLION
|
|
1532
1605
|
}
|
|
1533
1606
|
};
|
|
1534
1607
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -2550,6 +2623,12 @@ function stripCodeFences(text) {
|
|
|
2550
2623
|
}
|
|
2551
2624
|
var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
|
|
2552
2625
|
var AskResponseSchema = AskResultSchema.omit({ usage: true });
|
|
2626
|
+
var AskImageResponseSchema = AskResponseSchema.omit({
|
|
2627
|
+
frameReferences: true,
|
|
2628
|
+
timestampReferences: true
|
|
2629
|
+
});
|
|
2630
|
+
var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
|
|
2631
|
+
var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
|
|
2553
2632
|
var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
|
|
2554
2633
|
var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
|
|
2555
2634
|
function parseJson(text) {
|
|
@@ -2642,7 +2721,11 @@ function createDriver(provider, config) {
|
|
|
2642
2721
|
return PROVIDER_REGISTRY[provider](config);
|
|
2643
2722
|
}
|
|
2644
2723
|
var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
|
|
2645
|
-
var
|
|
2724
|
+
var askSchemaOptionsByMedia = {
|
|
2725
|
+
image: toSchemaOptions(AskImageResponseSchema),
|
|
2726
|
+
video: toSchemaOptions(AskFramesResponseSchema),
|
|
2727
|
+
"native-video": toSchemaOptions(AskNativeVideoResponseSchema)
|
|
2728
|
+
};
|
|
2646
2729
|
var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
|
|
2647
2730
|
function mediaToProviderInputs(media) {
|
|
2648
2731
|
if (media.kind === "image") {
|
|
@@ -2787,7 +2870,12 @@ function visualAI(config = {}) {
|
|
|
2787
2870
|
media: dispatch.mediaContext
|
|
2788
2871
|
});
|
|
2789
2872
|
debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
|
|
2790
|
-
const { response, metadata } = await sendMedia(
|
|
2873
|
+
const { response, metadata } = await sendMedia(
|
|
2874
|
+
driver,
|
|
2875
|
+
dispatch,
|
|
2876
|
+
prompt,
|
|
2877
|
+
askSchemaOptionsByMedia[dispatch.mediaContext.kind]
|
|
2878
|
+
);
|
|
2791
2879
|
debugLog(resolvedConfig, "ask response", response.text, "response");
|
|
2792
2880
|
const result = parseAskResponse(response.text);
|
|
2793
2881
|
return {
|
|
@@ -2810,7 +2898,7 @@ function visualAI(config = {}) {
|
|
|
2810
2898
|
debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
|
|
2811
2899
|
const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
|
|
2812
2900
|
debugLog(resolvedConfig, "compare response", response.text, "response");
|
|
2813
|
-
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model
|
|
2901
|
+
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
|
|
2814
2902
|
const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
|
|
2815
2903
|
let diffImage;
|
|
2816
2904
|
if (effectiveDiffImage) {
|