visual-ai-assertions 0.25.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +96 -76
- package/dist/index.cjs +216 -136
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +46 -13
- package/dist/index.d.ts +46 -13
- package/dist/index.js +216 -136
- package/dist/index.js.map +1 -1
- package/package.json +2 -3
package/dist/index.cjs
CHANGED
|
@@ -67,135 +67,6 @@ __export(index_exports, {
|
|
|
67
67
|
});
|
|
68
68
|
module.exports = __toCommonJS(index_exports);
|
|
69
69
|
|
|
70
|
-
// src/constants.ts
|
|
71
|
-
var ReasoningEffort = {
|
|
72
|
-
LOW: "low",
|
|
73
|
-
MEDIUM: "medium",
|
|
74
|
-
HIGH: "high",
|
|
75
|
-
XHIGH: "xhigh"
|
|
76
|
-
};
|
|
77
|
-
var ImageDetail = {
|
|
78
|
-
AUTO: "auto",
|
|
79
|
-
LOW: "low",
|
|
80
|
-
HIGH: "high"
|
|
81
|
-
};
|
|
82
|
-
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
83
|
-
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
84
|
-
var Provider = {
|
|
85
|
-
ANTHROPIC: "anthropic",
|
|
86
|
-
OPENAI: "openai",
|
|
87
|
-
GOOGLE: "google",
|
|
88
|
-
OPENROUTER: "openrouter"
|
|
89
|
-
};
|
|
90
|
-
var Model = {
|
|
91
|
-
Anthropic: {
|
|
92
|
-
FABLE_5_1: "claude-fable-5-1",
|
|
93
|
-
FABLE_5: "claude-fable-5",
|
|
94
|
-
OPUS_5: "claude-opus-5",
|
|
95
|
-
OPUS_4_8: "claude-opus-4-8",
|
|
96
|
-
OPUS_4_7: "claude-opus-4-7",
|
|
97
|
-
OPUS_4_6: "claude-opus-4-6",
|
|
98
|
-
SONNET_5: "claude-sonnet-5",
|
|
99
|
-
SONNET_4_6: "claude-sonnet-4-6",
|
|
100
|
-
HAIKU_4_5: "claude-haiku-4-5"
|
|
101
|
-
},
|
|
102
|
-
OpenAI: {
|
|
103
|
-
GPT_6_ASTRA: "gpt-6-astra",
|
|
104
|
-
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
105
|
-
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
106
|
-
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
107
|
-
GPT_5_5: "gpt-5.5",
|
|
108
|
-
GPT_5_4: "gpt-5.4",
|
|
109
|
-
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
110
|
-
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
111
|
-
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
112
|
-
GPT_5_2: "gpt-5.2",
|
|
113
|
-
GPT_5_MINI: "gpt-5-mini"
|
|
114
|
-
},
|
|
115
|
-
Google: {
|
|
116
|
-
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
117
|
-
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
118
|
-
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
119
|
-
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
120
|
-
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
121
|
-
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
122
|
-
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
123
|
-
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
124
|
-
},
|
|
125
|
-
/**
|
|
126
|
-
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
127
|
-
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
128
|
-
* recognizes them. All listed models accept image input.
|
|
129
|
-
*/
|
|
130
|
-
OpenRouter: {
|
|
131
|
-
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
132
|
-
GROK_4_6: "x-ai/grok-4.6",
|
|
133
|
-
GROK_4_5: "x-ai/grok-4.5",
|
|
134
|
-
KIMI_K3: "moonshotai/kimi-k3",
|
|
135
|
-
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
136
|
-
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
137
|
-
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
138
|
-
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
139
|
-
GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
|
|
140
|
-
}
|
|
141
|
-
};
|
|
142
|
-
var DEFAULT_MODELS = {
|
|
143
|
-
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
|
|
144
|
-
[Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
|
|
145
|
-
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
146
|
-
[Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
|
|
147
|
-
};
|
|
148
|
-
var DEFAULT_MAX_TOKENS = 4096;
|
|
149
|
-
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
150
|
-
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
151
|
-
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
152
|
-
Model.OpenAI.GPT_6_ASTRA
|
|
153
|
-
]);
|
|
154
|
-
var MODEL_TO_PROVIDER = new Map([
|
|
155
|
-
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
156
|
-
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
157
|
-
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
158
|
-
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
159
|
-
]);
|
|
160
|
-
var VALID_PROVIDERS = Object.values(Provider);
|
|
161
|
-
var PROVIDER_DEFAULT_REASONING = {
|
|
162
|
-
openai: "medium",
|
|
163
|
-
anthropic: "off",
|
|
164
|
-
google: "off",
|
|
165
|
-
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
166
|
-
openrouter: "off"
|
|
167
|
-
};
|
|
168
|
-
var Content = {
|
|
169
|
-
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
170
|
-
PLACEHOLDER_TEXT: "placeholder-text",
|
|
171
|
-
/** Detects error messages, banners, stack traces, or error codes */
|
|
172
|
-
ERROR_MESSAGES: "error-messages",
|
|
173
|
-
/** Detects broken image icons or failed-to-load image indicators */
|
|
174
|
-
BROKEN_IMAGES: "broken-images",
|
|
175
|
-
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
176
|
-
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
177
|
-
};
|
|
178
|
-
var Layout = {
|
|
179
|
-
/** Detects elements that unintentionally overlap each other */
|
|
180
|
-
OVERLAP: "overlap",
|
|
181
|
-
/** Detects content cut off or extending beyond container boundaries */
|
|
182
|
-
OVERFLOW: "overflow",
|
|
183
|
-
/** Detects inconsistent alignment of text, images, and UI components */
|
|
184
|
-
ALIGNMENT: "alignment"
|
|
185
|
-
};
|
|
186
|
-
var Accessibility = {
|
|
187
|
-
/** Detects insufficient color contrast between text and backgrounds */
|
|
188
|
-
CONTRAST: "contrast",
|
|
189
|
-
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
190
|
-
READABILITY: "readability",
|
|
191
|
-
/** Detects interactive elements that are not visually distinct */
|
|
192
|
-
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
193
|
-
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
194
|
-
COLOR_BLINDNESS: "color-blindness",
|
|
195
|
-
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
196
|
-
COLOR_ALONE: "color-alone"
|
|
197
|
-
};
|
|
198
|
-
|
|
199
70
|
// src/errors.ts
|
|
200
71
|
var VisualAIError = class extends Error {
|
|
201
72
|
code;
|
|
@@ -562,7 +433,7 @@ function visibleRole(finalState, requireCorrectRendering) {
|
|
|
562
433
|
var ELEMENTS_VISIBLE_CLIPPING_RULES = [
|
|
563
434
|
"When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
|
|
564
435
|
"An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
|
|
565
|
-
"
|
|
436
|
+
"The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
|
|
566
437
|
];
|
|
567
438
|
var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
|
|
568
439
|
var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
|
|
@@ -593,6 +464,144 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
|
|
|
593
464
|
return buildCheckPrompt(statements, { role, instructions });
|
|
594
465
|
}
|
|
595
466
|
|
|
467
|
+
// src/constants.ts
|
|
468
|
+
var ReasoningEffort = {
|
|
469
|
+
MINIMAL: "minimal",
|
|
470
|
+
LOW: "low",
|
|
471
|
+
MEDIUM: "medium",
|
|
472
|
+
HIGH: "high",
|
|
473
|
+
XHIGH: "xhigh"
|
|
474
|
+
};
|
|
475
|
+
var ImageDetail = {
|
|
476
|
+
AUTO: "auto",
|
|
477
|
+
LOW: "low",
|
|
478
|
+
HIGH: "high"
|
|
479
|
+
};
|
|
480
|
+
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
481
|
+
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
482
|
+
var Provider = {
|
|
483
|
+
ANTHROPIC: "anthropic",
|
|
484
|
+
OPENAI: "openai",
|
|
485
|
+
GOOGLE: "google",
|
|
486
|
+
OPENROUTER: "openrouter"
|
|
487
|
+
};
|
|
488
|
+
var Model = {
|
|
489
|
+
Anthropic: {
|
|
490
|
+
FABLE_5_1: "claude-fable-5-1",
|
|
491
|
+
FABLE_5: "claude-fable-5",
|
|
492
|
+
OPUS_5_5: "claude-opus-5-5",
|
|
493
|
+
OPUS_5: "claude-opus-5",
|
|
494
|
+
OPUS_4_8: "claude-opus-4-8",
|
|
495
|
+
OPUS_4_7: "claude-opus-4-7",
|
|
496
|
+
OPUS_4_6: "claude-opus-4-6",
|
|
497
|
+
SONNET_5_5: "claude-sonnet-5-5",
|
|
498
|
+
SONNET_5: "claude-sonnet-5",
|
|
499
|
+
SONNET_4_6: "claude-sonnet-4-6",
|
|
500
|
+
HAIKU_4_5: "claude-haiku-4-5"
|
|
501
|
+
},
|
|
502
|
+
OpenAI: {
|
|
503
|
+
GPT_6_ASTRA: "gpt-6-astra",
|
|
504
|
+
GPT_6_1_SOL: "gpt-6.1-sol",
|
|
505
|
+
GPT_6_SOL: "gpt-6-sol",
|
|
506
|
+
GPT_6_LUNA: "gpt-6-luna",
|
|
507
|
+
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
508
|
+
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
509
|
+
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
510
|
+
GPT_5_5: "gpt-5.5",
|
|
511
|
+
GPT_5_4: "gpt-5.4",
|
|
512
|
+
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
513
|
+
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
514
|
+
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
515
|
+
GPT_5_2: "gpt-5.2",
|
|
516
|
+
GPT_5_MINI: "gpt-5-mini"
|
|
517
|
+
},
|
|
518
|
+
Google: {
|
|
519
|
+
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
520
|
+
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
521
|
+
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
522
|
+
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
523
|
+
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
524
|
+
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
525
|
+
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
526
|
+
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
527
|
+
},
|
|
528
|
+
/**
|
|
529
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
530
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
531
|
+
* recognizes them. All listed models accept image input.
|
|
532
|
+
*/
|
|
533
|
+
OpenRouter: {
|
|
534
|
+
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
535
|
+
GROK_4_7: "x-ai/grok-4.7",
|
|
536
|
+
GROK_4_6: "x-ai/grok-4.6",
|
|
537
|
+
GROK_4_5: "x-ai/grok-4.5",
|
|
538
|
+
KIMI_K3: "moonshotai/kimi-k3",
|
|
539
|
+
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
540
|
+
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
541
|
+
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
542
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
543
|
+
GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
|
|
544
|
+
MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
|
|
545
|
+
}
|
|
546
|
+
};
|
|
547
|
+
var DEFAULT_MODELS = {
|
|
548
|
+
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
|
|
549
|
+
[Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
|
|
550
|
+
[Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
|
|
551
|
+
[Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
|
|
552
|
+
};
|
|
553
|
+
var DEFAULT_MAX_TOKENS = 4096;
|
|
554
|
+
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
555
|
+
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
556
|
+
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
557
|
+
Model.OpenRouter.QWEN_3_8_MAX,
|
|
558
|
+
Model.OpenRouter.QWEN_3_7_PLUS
|
|
559
|
+
]);
|
|
560
|
+
var MODEL_TO_PROVIDER = new Map([
|
|
561
|
+
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
562
|
+
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
563
|
+
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
564
|
+
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
565
|
+
]);
|
|
566
|
+
var VALID_PROVIDERS = Object.values(Provider);
|
|
567
|
+
var PROVIDER_DEFAULT_REASONING = {
|
|
568
|
+
openai: "medium",
|
|
569
|
+
anthropic: "off",
|
|
570
|
+
google: "off",
|
|
571
|
+
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
572
|
+
openrouter: "off"
|
|
573
|
+
};
|
|
574
|
+
var Content = {
|
|
575
|
+
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
576
|
+
PLACEHOLDER_TEXT: "placeholder-text",
|
|
577
|
+
/** Detects error messages, banners, stack traces, or error codes */
|
|
578
|
+
ERROR_MESSAGES: "error-messages",
|
|
579
|
+
/** Detects broken image icons or failed-to-load image indicators */
|
|
580
|
+
BROKEN_IMAGES: "broken-images",
|
|
581
|
+
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
582
|
+
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
583
|
+
};
|
|
584
|
+
var Layout = {
|
|
585
|
+
/** Detects elements that unintentionally overlap each other */
|
|
586
|
+
OVERLAP: "overlap",
|
|
587
|
+
/** Detects content cut off or extending beyond container boundaries */
|
|
588
|
+
OVERFLOW: "overflow",
|
|
589
|
+
/** Detects inconsistent alignment of text, images, and UI components */
|
|
590
|
+
ALIGNMENT: "alignment"
|
|
591
|
+
};
|
|
592
|
+
var Accessibility = {
|
|
593
|
+
/** Detects insufficient color contrast between text and backgrounds */
|
|
594
|
+
CONTRAST: "contrast",
|
|
595
|
+
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
596
|
+
READABILITY: "readability",
|
|
597
|
+
/** Detects interactive elements that are not visually distinct */
|
|
598
|
+
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
599
|
+
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
600
|
+
COLOR_BLINDNESS: "color-blindness",
|
|
601
|
+
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
602
|
+
COLOR_ALONE: "color-alone"
|
|
603
|
+
};
|
|
604
|
+
|
|
596
605
|
// src/templates/accessibility.ts
|
|
597
606
|
var ALL_CHECKS = Object.values(Accessibility);
|
|
598
607
|
var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
|
|
@@ -702,17 +711,22 @@ function parseRetryAfter(value) {
|
|
|
702
711
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
703
712
|
Model.Anthropic.FABLE_5_1,
|
|
704
713
|
Model.Anthropic.FABLE_5,
|
|
714
|
+
Model.Anthropic.OPUS_5_5,
|
|
705
715
|
Model.Anthropic.OPUS_5,
|
|
706
716
|
Model.Anthropic.OPUS_4_8,
|
|
707
717
|
Model.Anthropic.OPUS_4_7,
|
|
718
|
+
Model.Anthropic.SONNET_5_5,
|
|
708
719
|
Model.Anthropic.SONNET_5
|
|
709
720
|
]);
|
|
710
721
|
function mapEffort(level, model) {
|
|
722
|
+
if (level === "minimal") return "low";
|
|
711
723
|
if (level !== "xhigh") return level;
|
|
712
724
|
return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
|
|
713
725
|
}
|
|
714
726
|
var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
|
|
715
727
|
var EFFORT_TO_BUDGET_TOKENS = {
|
|
728
|
+
// 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
|
|
729
|
+
minimal: 1024,
|
|
716
730
|
low: 1024,
|
|
717
731
|
medium: 4096,
|
|
718
732
|
high: 8192,
|
|
@@ -829,6 +843,9 @@ function sleep(ms) {
|
|
|
829
843
|
return new Promise((resolve2) => setTimeout(resolve2, ms));
|
|
830
844
|
}
|
|
831
845
|
var GOOGLE_THINKING_LEVEL = {
|
|
846
|
+
// Gemini does define a "minimal" thinking level, but some models reject it
|
|
847
|
+
// (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
|
|
848
|
+
minimal: "low",
|
|
832
849
|
low: "low",
|
|
833
850
|
medium: "medium",
|
|
834
851
|
high: "high",
|
|
@@ -1146,6 +1163,9 @@ var OpenAIDriver = class {
|
|
|
1146
1163
|
// src/providers/openrouter.ts
|
|
1147
1164
|
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
1148
1165
|
var OPENROUTER_REASONING_EFFORT = {
|
|
1166
|
+
// OpenRouter normalizes upstream vendors to low/medium/high only, so
|
|
1167
|
+
// "minimal" has no native equivalent and clamps to the floor.
|
|
1168
|
+
minimal: "low",
|
|
1149
1169
|
low: "low",
|
|
1150
1170
|
medium: "medium",
|
|
1151
1171
|
high: "high",
|
|
@@ -1305,10 +1325,20 @@ function parseBooleanEnv(envName, value) {
|
|
|
1305
1325
|
`Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
|
|
1306
1326
|
);
|
|
1307
1327
|
}
|
|
1328
|
+
function parseReasoningEffortEnv(envName, value) {
|
|
1329
|
+
if (value === void 0 || value === "") return void 0;
|
|
1330
|
+
const levels = Object.values(ReasoningEffort);
|
|
1331
|
+
const lower = value.toLowerCase();
|
|
1332
|
+
if (levels.includes(lower)) return lower;
|
|
1333
|
+
throw new VisualAIConfigError(
|
|
1334
|
+
`Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
|
|
1335
|
+
);
|
|
1336
|
+
}
|
|
1308
1337
|
var debugDeprecationWarned = false;
|
|
1309
1338
|
function resolveConfig(config) {
|
|
1310
1339
|
const provider = resolveProvider(config);
|
|
1311
1340
|
const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
|
|
1341
|
+
const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
|
|
1312
1342
|
const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
|
|
1313
1343
|
const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
|
|
1314
1344
|
const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
|
|
@@ -1326,12 +1356,12 @@ function resolveConfig(config) {
|
|
|
1326
1356
|
}
|
|
1327
1357
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
1328
1358
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
1329
|
-
const effortNeedsLargeBudget =
|
|
1359
|
+
const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
|
|
1330
1360
|
const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
|
|
1331
1361
|
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
|
|
1332
1362
|
maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
|
|
1333
1363
|
if (debug) {
|
|
1334
|
-
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${
|
|
1364
|
+
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
|
|
1335
1365
|
process.stderr.write(
|
|
1336
1366
|
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
|
|
1337
1367
|
`
|
|
@@ -1343,7 +1373,7 @@ function resolveConfig(config) {
|
|
|
1343
1373
|
apiKey: config.apiKey,
|
|
1344
1374
|
model,
|
|
1345
1375
|
maxTokens,
|
|
1346
|
-
reasoningEffort
|
|
1376
|
+
reasoningEffort,
|
|
1347
1377
|
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1348
1378
|
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
1349
1379
|
timeout: config.timeout,
|
|
@@ -1366,6 +1396,10 @@ var PRICING_TABLE = {
|
|
|
1366
1396
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1367
1397
|
outputPricePerToken: 50 / PER_MILLION
|
|
1368
1398
|
},
|
|
1399
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
|
|
1400
|
+
inputPricePerToken: 4 / PER_MILLION,
|
|
1401
|
+
outputPricePerToken: 20 / PER_MILLION
|
|
1402
|
+
},
|
|
1369
1403
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
|
|
1370
1404
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1371
1405
|
outputPricePerToken: 25 / PER_MILLION
|
|
@@ -1374,6 +1408,10 @@ var PRICING_TABLE = {
|
|
|
1374
1408
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1375
1409
|
outputPricePerToken: 25 / PER_MILLION
|
|
1376
1410
|
},
|
|
1411
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
|
|
1412
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1413
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1414
|
+
},
|
|
1377
1415
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
|
|
1378
1416
|
inputPricePerToken: 3 / PER_MILLION,
|
|
1379
1417
|
outputPricePerToken: 15 / PER_MILLION
|
|
@@ -1400,6 +1438,22 @@ var PRICING_TABLE = {
|
|
|
1400
1438
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1401
1439
|
outputPricePerToken: 50 / PER_MILLION
|
|
1402
1440
|
},
|
|
1441
|
+
// Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
|
|
1442
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
|
|
1443
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1444
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1445
|
+
},
|
|
1446
|
+
// Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
|
|
1447
|
+
// bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
|
|
1448
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
|
|
1449
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1450
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1451
|
+
},
|
|
1452
|
+
// Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
|
|
1453
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
|
|
1454
|
+
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1455
|
+
outputPricePerToken: 0.5 / PER_MILLION
|
|
1456
|
+
},
|
|
1403
1457
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
|
|
1404
1458
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1405
1459
|
outputPricePerToken: 30 / PER_MILLION
|
|
@@ -1496,6 +1550,10 @@ var PRICING_TABLE = {
|
|
|
1496
1550
|
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1497
1551
|
outputPricePerToken: 0.2 / PER_MILLION
|
|
1498
1552
|
},
|
|
1553
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
|
|
1554
|
+
inputPricePerToken: 1.6 / PER_MILLION,
|
|
1555
|
+
outputPricePerToken: 4.8 / PER_MILLION
|
|
1556
|
+
},
|
|
1499
1557
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
|
|
1500
1558
|
inputPricePerToken: 2 / PER_MILLION,
|
|
1501
1559
|
outputPricePerToken: 6 / PER_MILLION
|
|
@@ -1529,6 +1587,13 @@ var PRICING_TABLE = {
|
|
|
1529
1587
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
|
|
1530
1588
|
inputPricePerToken: 0.15 / PER_MILLION,
|
|
1531
1589
|
outputPricePerToken: 0.5 / PER_MILLION
|
|
1590
|
+
},
|
|
1591
|
+
// Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
|
|
1592
|
+
// upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
|
|
1593
|
+
// is $0.0036/MTok, not modelled (no provider gets a cache discount here).
|
|
1594
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
|
|
1595
|
+
inputPricePerToken: 0.435 / PER_MILLION,
|
|
1596
|
+
outputPricePerToken: 0.87 / PER_MILLION
|
|
1532
1597
|
}
|
|
1533
1598
|
};
|
|
1534
1599
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -2550,6 +2615,12 @@ function stripCodeFences(text) {
|
|
|
2550
2615
|
}
|
|
2551
2616
|
var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
|
|
2552
2617
|
var AskResponseSchema = AskResultSchema.omit({ usage: true });
|
|
2618
|
+
var AskImageResponseSchema = AskResponseSchema.omit({
|
|
2619
|
+
frameReferences: true,
|
|
2620
|
+
timestampReferences: true
|
|
2621
|
+
});
|
|
2622
|
+
var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
|
|
2623
|
+
var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
|
|
2553
2624
|
var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
|
|
2554
2625
|
var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
|
|
2555
2626
|
function parseJson(text) {
|
|
@@ -2642,7 +2713,11 @@ function createDriver(provider, config) {
|
|
|
2642
2713
|
return PROVIDER_REGISTRY[provider](config);
|
|
2643
2714
|
}
|
|
2644
2715
|
var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
|
|
2645
|
-
var
|
|
2716
|
+
var askSchemaOptionsByMedia = {
|
|
2717
|
+
image: toSchemaOptions(AskImageResponseSchema),
|
|
2718
|
+
video: toSchemaOptions(AskFramesResponseSchema),
|
|
2719
|
+
"native-video": toSchemaOptions(AskNativeVideoResponseSchema)
|
|
2720
|
+
};
|
|
2646
2721
|
var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
|
|
2647
2722
|
function mediaToProviderInputs(media) {
|
|
2648
2723
|
if (media.kind === "image") {
|
|
@@ -2787,7 +2862,12 @@ function visualAI(config = {}) {
|
|
|
2787
2862
|
media: dispatch.mediaContext
|
|
2788
2863
|
});
|
|
2789
2864
|
debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
|
|
2790
|
-
const { response, metadata } = await sendMedia(
|
|
2865
|
+
const { response, metadata } = await sendMedia(
|
|
2866
|
+
driver,
|
|
2867
|
+
dispatch,
|
|
2868
|
+
prompt,
|
|
2869
|
+
askSchemaOptionsByMedia[dispatch.mediaContext.kind]
|
|
2870
|
+
);
|
|
2791
2871
|
debugLog(resolvedConfig, "ask response", response.text, "response");
|
|
2792
2872
|
const result = parseAskResponse(response.text);
|
|
2793
2873
|
return {
|
|
@@ -2810,7 +2890,7 @@ function visualAI(config = {}) {
|
|
|
2810
2890
|
debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
|
|
2811
2891
|
const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
|
|
2812
2892
|
debugLog(resolvedConfig, "compare response", response.text, "response");
|
|
2813
|
-
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model
|
|
2893
|
+
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
|
|
2814
2894
|
const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
|
|
2815
2895
|
let diffImage;
|
|
2816
2896
|
if (effectiveDiffImage) {
|