visual-ai-assertions 0.25.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +96 -76
- package/dist/index.cjs +216 -136
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +46 -13
- package/dist/index.d.ts +46 -13
- package/dist/index.js +216 -136
- package/dist/index.js.map +1 -1
- package/package.json +2 -3
package/dist/index.js
CHANGED
|
@@ -1,132 +1,3 @@
|
|
|
1
|
-
// src/constants.ts
|
|
2
|
-
var ReasoningEffort = {
|
|
3
|
-
LOW: "low",
|
|
4
|
-
MEDIUM: "medium",
|
|
5
|
-
HIGH: "high",
|
|
6
|
-
XHIGH: "xhigh"
|
|
7
|
-
};
|
|
8
|
-
var ImageDetail = {
|
|
9
|
-
AUTO: "auto",
|
|
10
|
-
LOW: "low",
|
|
11
|
-
HIGH: "high"
|
|
12
|
-
};
|
|
13
|
-
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
14
|
-
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
15
|
-
var Provider = {
|
|
16
|
-
ANTHROPIC: "anthropic",
|
|
17
|
-
OPENAI: "openai",
|
|
18
|
-
GOOGLE: "google",
|
|
19
|
-
OPENROUTER: "openrouter"
|
|
20
|
-
};
|
|
21
|
-
var Model = {
|
|
22
|
-
Anthropic: {
|
|
23
|
-
FABLE_5_1: "claude-fable-5-1",
|
|
24
|
-
FABLE_5: "claude-fable-5",
|
|
25
|
-
OPUS_5: "claude-opus-5",
|
|
26
|
-
OPUS_4_8: "claude-opus-4-8",
|
|
27
|
-
OPUS_4_7: "claude-opus-4-7",
|
|
28
|
-
OPUS_4_6: "claude-opus-4-6",
|
|
29
|
-
SONNET_5: "claude-sonnet-5",
|
|
30
|
-
SONNET_4_6: "claude-sonnet-4-6",
|
|
31
|
-
HAIKU_4_5: "claude-haiku-4-5"
|
|
32
|
-
},
|
|
33
|
-
OpenAI: {
|
|
34
|
-
GPT_6_ASTRA: "gpt-6-astra",
|
|
35
|
-
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
36
|
-
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
37
|
-
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
38
|
-
GPT_5_5: "gpt-5.5",
|
|
39
|
-
GPT_5_4: "gpt-5.4",
|
|
40
|
-
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
41
|
-
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
42
|
-
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
43
|
-
GPT_5_2: "gpt-5.2",
|
|
44
|
-
GPT_5_MINI: "gpt-5-mini"
|
|
45
|
-
},
|
|
46
|
-
Google: {
|
|
47
|
-
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
48
|
-
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
49
|
-
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
50
|
-
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
51
|
-
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
52
|
-
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
53
|
-
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
54
|
-
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
55
|
-
},
|
|
56
|
-
/**
|
|
57
|
-
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
58
|
-
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
59
|
-
* recognizes them. All listed models accept image input.
|
|
60
|
-
*/
|
|
61
|
-
OpenRouter: {
|
|
62
|
-
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
63
|
-
GROK_4_6: "x-ai/grok-4.6",
|
|
64
|
-
GROK_4_5: "x-ai/grok-4.5",
|
|
65
|
-
KIMI_K3: "moonshotai/kimi-k3",
|
|
66
|
-
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
67
|
-
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
68
|
-
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
69
|
-
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
70
|
-
GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
|
|
71
|
-
}
|
|
72
|
-
};
|
|
73
|
-
var DEFAULT_MODELS = {
|
|
74
|
-
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
|
|
75
|
-
[Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
|
|
76
|
-
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
77
|
-
[Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
|
|
78
|
-
};
|
|
79
|
-
var DEFAULT_MAX_TOKENS = 4096;
|
|
80
|
-
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
81
|
-
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
82
|
-
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
83
|
-
Model.OpenAI.GPT_6_ASTRA
|
|
84
|
-
]);
|
|
85
|
-
var MODEL_TO_PROVIDER = new Map([
|
|
86
|
-
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
87
|
-
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
88
|
-
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
89
|
-
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
90
|
-
]);
|
|
91
|
-
var VALID_PROVIDERS = Object.values(Provider);
|
|
92
|
-
var PROVIDER_DEFAULT_REASONING = {
|
|
93
|
-
openai: "medium",
|
|
94
|
-
anthropic: "off",
|
|
95
|
-
google: "off",
|
|
96
|
-
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
97
|
-
openrouter: "off"
|
|
98
|
-
};
|
|
99
|
-
var Content = {
|
|
100
|
-
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
101
|
-
PLACEHOLDER_TEXT: "placeholder-text",
|
|
102
|
-
/** Detects error messages, banners, stack traces, or error codes */
|
|
103
|
-
ERROR_MESSAGES: "error-messages",
|
|
104
|
-
/** Detects broken image icons or failed-to-load image indicators */
|
|
105
|
-
BROKEN_IMAGES: "broken-images",
|
|
106
|
-
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
107
|
-
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
108
|
-
};
|
|
109
|
-
var Layout = {
|
|
110
|
-
/** Detects elements that unintentionally overlap each other */
|
|
111
|
-
OVERLAP: "overlap",
|
|
112
|
-
/** Detects content cut off or extending beyond container boundaries */
|
|
113
|
-
OVERFLOW: "overflow",
|
|
114
|
-
/** Detects inconsistent alignment of text, images, and UI components */
|
|
115
|
-
ALIGNMENT: "alignment"
|
|
116
|
-
};
|
|
117
|
-
var Accessibility = {
|
|
118
|
-
/** Detects insufficient color contrast between text and backgrounds */
|
|
119
|
-
CONTRAST: "contrast",
|
|
120
|
-
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
121
|
-
READABILITY: "readability",
|
|
122
|
-
/** Detects interactive elements that are not visually distinct */
|
|
123
|
-
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
124
|
-
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
125
|
-
COLOR_BLINDNESS: "color-blindness",
|
|
126
|
-
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
127
|
-
COLOR_ALONE: "color-alone"
|
|
128
|
-
};
|
|
129
|
-
|
|
130
1
|
// src/errors.ts
|
|
131
2
|
var VisualAIError = class extends Error {
|
|
132
3
|
code;
|
|
@@ -493,7 +364,7 @@ function visibleRole(finalState, requireCorrectRendering) {
|
|
|
493
364
|
var ELEMENTS_VISIBLE_CLIPPING_RULES = [
|
|
494
365
|
"When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
|
|
495
366
|
"An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
|
|
496
|
-
"
|
|
367
|
+
"The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
|
|
497
368
|
];
|
|
498
369
|
var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
|
|
499
370
|
var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
|
|
@@ -524,6 +395,144 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
|
|
|
524
395
|
return buildCheckPrompt(statements, { role, instructions });
|
|
525
396
|
}
|
|
526
397
|
|
|
398
|
+
// src/constants.ts
|
|
399
|
+
var ReasoningEffort = {
|
|
400
|
+
MINIMAL: "minimal",
|
|
401
|
+
LOW: "low",
|
|
402
|
+
MEDIUM: "medium",
|
|
403
|
+
HIGH: "high",
|
|
404
|
+
XHIGH: "xhigh"
|
|
405
|
+
};
|
|
406
|
+
var ImageDetail = {
|
|
407
|
+
AUTO: "auto",
|
|
408
|
+
LOW: "low",
|
|
409
|
+
HIGH: "high"
|
|
410
|
+
};
|
|
411
|
+
var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
|
|
412
|
+
var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
|
|
413
|
+
var Provider = {
|
|
414
|
+
ANTHROPIC: "anthropic",
|
|
415
|
+
OPENAI: "openai",
|
|
416
|
+
GOOGLE: "google",
|
|
417
|
+
OPENROUTER: "openrouter"
|
|
418
|
+
};
|
|
419
|
+
var Model = {
|
|
420
|
+
Anthropic: {
|
|
421
|
+
FABLE_5_1: "claude-fable-5-1",
|
|
422
|
+
FABLE_5: "claude-fable-5",
|
|
423
|
+
OPUS_5_5: "claude-opus-5-5",
|
|
424
|
+
OPUS_5: "claude-opus-5",
|
|
425
|
+
OPUS_4_8: "claude-opus-4-8",
|
|
426
|
+
OPUS_4_7: "claude-opus-4-7",
|
|
427
|
+
OPUS_4_6: "claude-opus-4-6",
|
|
428
|
+
SONNET_5_5: "claude-sonnet-5-5",
|
|
429
|
+
SONNET_5: "claude-sonnet-5",
|
|
430
|
+
SONNET_4_6: "claude-sonnet-4-6",
|
|
431
|
+
HAIKU_4_5: "claude-haiku-4-5"
|
|
432
|
+
},
|
|
433
|
+
OpenAI: {
|
|
434
|
+
GPT_6_ASTRA: "gpt-6-astra",
|
|
435
|
+
GPT_6_1_SOL: "gpt-6.1-sol",
|
|
436
|
+
GPT_6_SOL: "gpt-6-sol",
|
|
437
|
+
GPT_6_LUNA: "gpt-6-luna",
|
|
438
|
+
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
439
|
+
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
440
|
+
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
441
|
+
GPT_5_5: "gpt-5.5",
|
|
442
|
+
GPT_5_4: "gpt-5.4",
|
|
443
|
+
GPT_5_4_PRO: "gpt-5.4-pro",
|
|
444
|
+
GPT_5_4_MINI: "gpt-5.4-mini",
|
|
445
|
+
GPT_5_4_NANO: "gpt-5.4-nano",
|
|
446
|
+
GPT_5_2: "gpt-5.2",
|
|
447
|
+
GPT_5_MINI: "gpt-5-mini"
|
|
448
|
+
},
|
|
449
|
+
Google: {
|
|
450
|
+
GEMINI_3_8_FLASH: "gemini-3.8-flash",
|
|
451
|
+
GEMINI_3_7_FLASH: "gemini-3.7-flash",
|
|
452
|
+
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
453
|
+
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
454
|
+
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
455
|
+
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
456
|
+
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
457
|
+
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
458
|
+
},
|
|
459
|
+
/**
|
|
460
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
461
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
462
|
+
* recognizes them. All listed models accept image input.
|
|
463
|
+
*/
|
|
464
|
+
OpenRouter: {
|
|
465
|
+
MUSE_SPARK_1_3: "meta/muse-spark-1.3",
|
|
466
|
+
GROK_4_7: "x-ai/grok-4.7",
|
|
467
|
+
GROK_4_6: "x-ai/grok-4.6",
|
|
468
|
+
GROK_4_5: "x-ai/grok-4.5",
|
|
469
|
+
KIMI_K3: "moonshotai/kimi-k3",
|
|
470
|
+
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
471
|
+
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
472
|
+
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
473
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
474
|
+
GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
|
|
475
|
+
MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
|
|
476
|
+
}
|
|
477
|
+
};
|
|
478
|
+
var DEFAULT_MODELS = {
|
|
479
|
+
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
|
|
480
|
+
[Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
|
|
481
|
+
[Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
|
|
482
|
+
[Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
|
|
483
|
+
};
|
|
484
|
+
var DEFAULT_MAX_TOKENS = 4096;
|
|
485
|
+
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
486
|
+
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
487
|
+
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
488
|
+
Model.OpenRouter.QWEN_3_8_MAX,
|
|
489
|
+
Model.OpenRouter.QWEN_3_7_PLUS
|
|
490
|
+
]);
|
|
491
|
+
var MODEL_TO_PROVIDER = new Map([
|
|
492
|
+
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
493
|
+
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
494
|
+
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
495
|
+
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
496
|
+
]);
|
|
497
|
+
var VALID_PROVIDERS = Object.values(Provider);
|
|
498
|
+
var PROVIDER_DEFAULT_REASONING = {
|
|
499
|
+
openai: "medium",
|
|
500
|
+
anthropic: "off",
|
|
501
|
+
google: "off",
|
|
502
|
+
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
503
|
+
openrouter: "off"
|
|
504
|
+
};
|
|
505
|
+
var Content = {
|
|
506
|
+
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
507
|
+
PLACEHOLDER_TEXT: "placeholder-text",
|
|
508
|
+
/** Detects error messages, banners, stack traces, or error codes */
|
|
509
|
+
ERROR_MESSAGES: "error-messages",
|
|
510
|
+
/** Detects broken image icons or failed-to-load image indicators */
|
|
511
|
+
BROKEN_IMAGES: "broken-images",
|
|
512
|
+
/** Detects UI elements that unintentionally overlap and obscure content */
|
|
513
|
+
OVERLAPPING_ELEMENTS: "overlapping-elements"
|
|
514
|
+
};
|
|
515
|
+
var Layout = {
|
|
516
|
+
/** Detects elements that unintentionally overlap each other */
|
|
517
|
+
OVERLAP: "overlap",
|
|
518
|
+
/** Detects content cut off or extending beyond container boundaries */
|
|
519
|
+
OVERFLOW: "overflow",
|
|
520
|
+
/** Detects inconsistent alignment of text, images, and UI components */
|
|
521
|
+
ALIGNMENT: "alignment"
|
|
522
|
+
};
|
|
523
|
+
var Accessibility = {
|
|
524
|
+
/** Detects insufficient color contrast between text and backgrounds */
|
|
525
|
+
CONTRAST: "contrast",
|
|
526
|
+
/** Detects text that is cut off, overlapping, too small, or obscured */
|
|
527
|
+
READABILITY: "readability",
|
|
528
|
+
/** Detects interactive elements that are not visually distinct */
|
|
529
|
+
INTERACTIVE_VISIBILITY: "interactive-visibility",
|
|
530
|
+
/** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
|
|
531
|
+
COLOR_BLINDNESS: "color-blindness",
|
|
532
|
+
/** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
|
|
533
|
+
COLOR_ALONE: "color-alone"
|
|
534
|
+
};
|
|
535
|
+
|
|
527
536
|
// src/templates/accessibility.ts
|
|
528
537
|
var ALL_CHECKS = Object.values(Accessibility);
|
|
529
538
|
var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
|
|
@@ -633,17 +642,22 @@ function parseRetryAfter(value) {
|
|
|
633
642
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
634
643
|
Model.Anthropic.FABLE_5_1,
|
|
635
644
|
Model.Anthropic.FABLE_5,
|
|
645
|
+
Model.Anthropic.OPUS_5_5,
|
|
636
646
|
Model.Anthropic.OPUS_5,
|
|
637
647
|
Model.Anthropic.OPUS_4_8,
|
|
638
648
|
Model.Anthropic.OPUS_4_7,
|
|
649
|
+
Model.Anthropic.SONNET_5_5,
|
|
639
650
|
Model.Anthropic.SONNET_5
|
|
640
651
|
]);
|
|
641
652
|
function mapEffort(level, model) {
|
|
653
|
+
if (level === "minimal") return "low";
|
|
642
654
|
if (level !== "xhigh") return level;
|
|
643
655
|
return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
|
|
644
656
|
}
|
|
645
657
|
var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
|
|
646
658
|
var EFFORT_TO_BUDGET_TOKENS = {
|
|
659
|
+
// 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
|
|
660
|
+
minimal: 1024,
|
|
647
661
|
low: 1024,
|
|
648
662
|
medium: 4096,
|
|
649
663
|
high: 8192,
|
|
@@ -760,6 +774,9 @@ function sleep(ms) {
|
|
|
760
774
|
return new Promise((resolve2) => setTimeout(resolve2, ms));
|
|
761
775
|
}
|
|
762
776
|
var GOOGLE_THINKING_LEVEL = {
|
|
777
|
+
// Gemini does define a "minimal" thinking level, but some models reject it
|
|
778
|
+
// (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
|
|
779
|
+
minimal: "low",
|
|
763
780
|
low: "low",
|
|
764
781
|
medium: "medium",
|
|
765
782
|
high: "high",
|
|
@@ -1077,6 +1094,9 @@ var OpenAIDriver = class {
|
|
|
1077
1094
|
// src/providers/openrouter.ts
|
|
1078
1095
|
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
1079
1096
|
var OPENROUTER_REASONING_EFFORT = {
|
|
1097
|
+
// OpenRouter normalizes upstream vendors to low/medium/high only, so
|
|
1098
|
+
// "minimal" has no native equivalent and clamps to the floor.
|
|
1099
|
+
minimal: "low",
|
|
1080
1100
|
low: "low",
|
|
1081
1101
|
medium: "medium",
|
|
1082
1102
|
high: "high",
|
|
@@ -1236,10 +1256,20 @@ function parseBooleanEnv(envName, value) {
|
|
|
1236
1256
|
`Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
|
|
1237
1257
|
);
|
|
1238
1258
|
}
|
|
1259
|
+
function parseReasoningEffortEnv(envName, value) {
|
|
1260
|
+
if (value === void 0 || value === "") return void 0;
|
|
1261
|
+
const levels = Object.values(ReasoningEffort);
|
|
1262
|
+
const lower = value.toLowerCase();
|
|
1263
|
+
if (levels.includes(lower)) return lower;
|
|
1264
|
+
throw new VisualAIConfigError(
|
|
1265
|
+
`Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
|
|
1266
|
+
);
|
|
1267
|
+
}
|
|
1239
1268
|
var debugDeprecationWarned = false;
|
|
1240
1269
|
function resolveConfig(config) {
|
|
1241
1270
|
const provider = resolveProvider(config);
|
|
1242
1271
|
const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
|
|
1272
|
+
const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
|
|
1243
1273
|
const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
|
|
1244
1274
|
const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
|
|
1245
1275
|
const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
|
|
@@ -1257,12 +1287,12 @@ function resolveConfig(config) {
|
|
|
1257
1287
|
}
|
|
1258
1288
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
1259
1289
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
1260
|
-
const effortNeedsLargeBudget =
|
|
1290
|
+
const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
|
|
1261
1291
|
const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
|
|
1262
1292
|
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
|
|
1263
1293
|
maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
|
|
1264
1294
|
if (debug) {
|
|
1265
|
-
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${
|
|
1295
|
+
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
|
|
1266
1296
|
process.stderr.write(
|
|
1267
1297
|
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
|
|
1268
1298
|
`
|
|
@@ -1274,7 +1304,7 @@ function resolveConfig(config) {
|
|
|
1274
1304
|
apiKey: config.apiKey,
|
|
1275
1305
|
model,
|
|
1276
1306
|
maxTokens,
|
|
1277
|
-
reasoningEffort
|
|
1307
|
+
reasoningEffort,
|
|
1278
1308
|
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1279
1309
|
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
1280
1310
|
timeout: config.timeout,
|
|
@@ -1297,6 +1327,10 @@ var PRICING_TABLE = {
|
|
|
1297
1327
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1298
1328
|
outputPricePerToken: 50 / PER_MILLION
|
|
1299
1329
|
},
|
|
1330
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
|
|
1331
|
+
inputPricePerToken: 4 / PER_MILLION,
|
|
1332
|
+
outputPricePerToken: 20 / PER_MILLION
|
|
1333
|
+
},
|
|
1300
1334
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
|
|
1301
1335
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1302
1336
|
outputPricePerToken: 25 / PER_MILLION
|
|
@@ -1305,6 +1339,10 @@ var PRICING_TABLE = {
|
|
|
1305
1339
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1306
1340
|
outputPricePerToken: 25 / PER_MILLION
|
|
1307
1341
|
},
|
|
1342
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
|
|
1343
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1344
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1345
|
+
},
|
|
1308
1346
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
|
|
1309
1347
|
inputPricePerToken: 3 / PER_MILLION,
|
|
1310
1348
|
outputPricePerToken: 15 / PER_MILLION
|
|
@@ -1331,6 +1369,22 @@ var PRICING_TABLE = {
|
|
|
1331
1369
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1332
1370
|
outputPricePerToken: 50 / PER_MILLION
|
|
1333
1371
|
},
|
|
1372
|
+
// Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
|
|
1373
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
|
|
1374
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1375
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1376
|
+
},
|
|
1377
|
+
// Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
|
|
1378
|
+
// bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
|
|
1379
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
|
|
1380
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1381
|
+
outputPricePerToken: 10 / PER_MILLION
|
|
1382
|
+
},
|
|
1383
|
+
// Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
|
|
1384
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
|
|
1385
|
+
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1386
|
+
outputPricePerToken: 0.5 / PER_MILLION
|
|
1387
|
+
},
|
|
1334
1388
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
|
|
1335
1389
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1336
1390
|
outputPricePerToken: 30 / PER_MILLION
|
|
@@ -1427,6 +1481,10 @@ var PRICING_TABLE = {
|
|
|
1427
1481
|
inputPricePerToken: 0.1 / PER_MILLION,
|
|
1428
1482
|
outputPricePerToken: 0.2 / PER_MILLION
|
|
1429
1483
|
},
|
|
1484
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
|
|
1485
|
+
inputPricePerToken: 1.6 / PER_MILLION,
|
|
1486
|
+
outputPricePerToken: 4.8 / PER_MILLION
|
|
1487
|
+
},
|
|
1430
1488
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
|
|
1431
1489
|
inputPricePerToken: 2 / PER_MILLION,
|
|
1432
1490
|
outputPricePerToken: 6 / PER_MILLION
|
|
@@ -1460,6 +1518,13 @@ var PRICING_TABLE = {
|
|
|
1460
1518
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
|
|
1461
1519
|
inputPricePerToken: 0.15 / PER_MILLION,
|
|
1462
1520
|
outputPricePerToken: 0.5 / PER_MILLION
|
|
1521
|
+
},
|
|
1522
|
+
// Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
|
|
1523
|
+
// upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
|
|
1524
|
+
// is $0.0036/MTok, not modelled (no provider gets a cache discount here).
|
|
1525
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
|
|
1526
|
+
inputPricePerToken: 0.435 / PER_MILLION,
|
|
1527
|
+
outputPricePerToken: 0.87 / PER_MILLION
|
|
1463
1528
|
}
|
|
1464
1529
|
};
|
|
1465
1530
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -2481,6 +2546,12 @@ function stripCodeFences(text) {
|
|
|
2481
2546
|
}
|
|
2482
2547
|
var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
|
|
2483
2548
|
var AskResponseSchema = AskResultSchema.omit({ usage: true });
|
|
2549
|
+
var AskImageResponseSchema = AskResponseSchema.omit({
|
|
2550
|
+
frameReferences: true,
|
|
2551
|
+
timestampReferences: true
|
|
2552
|
+
});
|
|
2553
|
+
var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
|
|
2554
|
+
var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
|
|
2484
2555
|
var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
|
|
2485
2556
|
var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
|
|
2486
2557
|
function parseJson(text) {
|
|
@@ -2573,7 +2644,11 @@ function createDriver(provider, config) {
|
|
|
2573
2644
|
return PROVIDER_REGISTRY[provider](config);
|
|
2574
2645
|
}
|
|
2575
2646
|
var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
|
|
2576
|
-
var
|
|
2647
|
+
var askSchemaOptionsByMedia = {
|
|
2648
|
+
image: toSchemaOptions(AskImageResponseSchema),
|
|
2649
|
+
video: toSchemaOptions(AskFramesResponseSchema),
|
|
2650
|
+
"native-video": toSchemaOptions(AskNativeVideoResponseSchema)
|
|
2651
|
+
};
|
|
2577
2652
|
var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
|
|
2578
2653
|
function mediaToProviderInputs(media) {
|
|
2579
2654
|
if (media.kind === "image") {
|
|
@@ -2718,7 +2793,12 @@ function visualAI(config = {}) {
|
|
|
2718
2793
|
media: dispatch.mediaContext
|
|
2719
2794
|
});
|
|
2720
2795
|
debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
|
|
2721
|
-
const { response, metadata } = await sendMedia(
|
|
2796
|
+
const { response, metadata } = await sendMedia(
|
|
2797
|
+
driver,
|
|
2798
|
+
dispatch,
|
|
2799
|
+
prompt,
|
|
2800
|
+
askSchemaOptionsByMedia[dispatch.mediaContext.kind]
|
|
2801
|
+
);
|
|
2722
2802
|
debugLog(resolvedConfig, "ask response", response.text, "response");
|
|
2723
2803
|
const result = parseAskResponse(response.text);
|
|
2724
2804
|
return {
|
|
@@ -2741,7 +2821,7 @@ function visualAI(config = {}) {
|
|
|
2741
2821
|
debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
|
|
2742
2822
|
const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
|
|
2743
2823
|
debugLog(resolvedConfig, "compare response", response.text, "response");
|
|
2744
|
-
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model
|
|
2824
|
+
const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
|
|
2745
2825
|
const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
|
|
2746
2826
|
let diffImage;
|
|
2747
2827
|
if (effectiveDiffImage) {
|