visual-ai-assertions 0.25.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -67,135 +67,6 @@ __export(index_exports, {
67
67
  });
68
68
  module.exports = __toCommonJS(index_exports);
69
69
 
70
- // src/constants.ts
71
- var ReasoningEffort = {
72
- LOW: "low",
73
- MEDIUM: "medium",
74
- HIGH: "high",
75
- XHIGH: "xhigh"
76
- };
77
- var ImageDetail = {
78
- AUTO: "auto",
79
- LOW: "low",
80
- HIGH: "high"
81
- };
82
- var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
83
- var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
84
- var Provider = {
85
- ANTHROPIC: "anthropic",
86
- OPENAI: "openai",
87
- GOOGLE: "google",
88
- OPENROUTER: "openrouter"
89
- };
90
- var Model = {
91
- Anthropic: {
92
- FABLE_5_1: "claude-fable-5-1",
93
- FABLE_5: "claude-fable-5",
94
- OPUS_5: "claude-opus-5",
95
- OPUS_4_8: "claude-opus-4-8",
96
- OPUS_4_7: "claude-opus-4-7",
97
- OPUS_4_6: "claude-opus-4-6",
98
- SONNET_5: "claude-sonnet-5",
99
- SONNET_4_6: "claude-sonnet-4-6",
100
- HAIKU_4_5: "claude-haiku-4-5"
101
- },
102
- OpenAI: {
103
- GPT_6_ASTRA: "gpt-6-astra",
104
- GPT_5_6_SOL: "gpt-5.6-sol",
105
- GPT_5_6_TERRA: "gpt-5.6-terra",
106
- GPT_5_6_LUNA: "gpt-5.6-luna",
107
- GPT_5_5: "gpt-5.5",
108
- GPT_5_4: "gpt-5.4",
109
- GPT_5_4_PRO: "gpt-5.4-pro",
110
- GPT_5_4_MINI: "gpt-5.4-mini",
111
- GPT_5_4_NANO: "gpt-5.4-nano",
112
- GPT_5_2: "gpt-5.2",
113
- GPT_5_MINI: "gpt-5-mini"
114
- },
115
- Google: {
116
- GEMINI_3_8_FLASH: "gemini-3.8-flash",
117
- GEMINI_3_7_FLASH: "gemini-3.7-flash",
118
- GEMINI_3_6_FLASH: "gemini-3.6-flash",
119
- GEMINI_3_5_FLASH: "gemini-3.5-flash",
120
- GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
121
- GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
122
- GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
123
- GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
124
- },
125
- /**
126
- * Models routed through OpenRouter (https://openrouter.ai). Slugs always
127
- * carry a vendor prefix (`vendor/model`), which is how provider inference
128
- * recognizes them. All listed models accept image input.
129
- */
130
- OpenRouter: {
131
- MUSE_SPARK_1_3: "meta/muse-spark-1.3",
132
- GROK_4_6: "x-ai/grok-4.6",
133
- GROK_4_5: "x-ai/grok-4.5",
134
- KIMI_K3: "moonshotai/kimi-k3",
135
- KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
136
- QWEN_3_8_MAX: "qwen/qwen3.8-max",
137
- QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
138
- QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
139
- GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
140
- }
141
- };
142
- var DEFAULT_MODELS = {
143
- [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
144
- [Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
145
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
146
- [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
147
- };
148
- var DEFAULT_MAX_TOKENS = 4096;
149
- var OPENAI_REASONING_MAX_TOKENS = 16384;
150
- var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
151
- var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
152
- Model.OpenAI.GPT_6_ASTRA
153
- ]);
154
- var MODEL_TO_PROVIDER = new Map([
155
- ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
156
- ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
157
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
158
- ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
159
- ]);
160
- var VALID_PROVIDERS = Object.values(Provider);
161
- var PROVIDER_DEFAULT_REASONING = {
162
- openai: "medium",
163
- anthropic: "off",
164
- google: "off",
165
- // Varies by upstream model; the driver sends no reasoning field unless configured.
166
- openrouter: "off"
167
- };
168
- var Content = {
169
- /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
170
- PLACEHOLDER_TEXT: "placeholder-text",
171
- /** Detects error messages, banners, stack traces, or error codes */
172
- ERROR_MESSAGES: "error-messages",
173
- /** Detects broken image icons or failed-to-load image indicators */
174
- BROKEN_IMAGES: "broken-images",
175
- /** Detects UI elements that unintentionally overlap and obscure content */
176
- OVERLAPPING_ELEMENTS: "overlapping-elements"
177
- };
178
- var Layout = {
179
- /** Detects elements that unintentionally overlap each other */
180
- OVERLAP: "overlap",
181
- /** Detects content cut off or extending beyond container boundaries */
182
- OVERFLOW: "overflow",
183
- /** Detects inconsistent alignment of text, images, and UI components */
184
- ALIGNMENT: "alignment"
185
- };
186
- var Accessibility = {
187
- /** Detects insufficient color contrast between text and backgrounds */
188
- CONTRAST: "contrast",
189
- /** Detects text that is cut off, overlapping, too small, or obscured */
190
- READABILITY: "readability",
191
- /** Detects interactive elements that are not visually distinct */
192
- INTERACTIVE_VISIBILITY: "interactive-visibility",
193
- /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
194
- COLOR_BLINDNESS: "color-blindness",
195
- /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
196
- COLOR_ALONE: "color-alone"
197
- };
198
-
199
70
  // src/errors.ts
200
71
  var VisualAIError = class extends Error {
201
72
  code;
@@ -562,7 +433,7 @@ function visibleRole(finalState, requireCorrectRendering) {
562
433
  var ELEMENTS_VISIBLE_CLIPPING_RULES = [
563
434
  "When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
564
435
  "An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
565
- "An element you cannot see at all is not visible, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
436
+ "The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
566
437
  ];
567
438
  var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
568
439
  var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
@@ -593,6 +464,145 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
593
464
  return buildCheckPrompt(statements, { role, instructions });
594
465
  }
595
466
 
467
+ // src/constants.ts
468
+ var ReasoningEffort = {
469
+ MINIMAL: "minimal",
470
+ LOW: "low",
471
+ MEDIUM: "medium",
472
+ HIGH: "high",
473
+ XHIGH: "xhigh"
474
+ };
475
+ var ImageDetail = {
476
+ AUTO: "auto",
477
+ LOW: "low",
478
+ HIGH: "high"
479
+ };
480
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
481
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
482
+ var Provider = {
483
+ ANTHROPIC: "anthropic",
484
+ OPENAI: "openai",
485
+ GOOGLE: "google",
486
+ OPENROUTER: "openrouter"
487
+ };
488
+ var Model = {
489
+ Anthropic: {
490
+ FABLE_5_1: "claude-fable-5-1",
491
+ FABLE_5: "claude-fable-5",
492
+ OPUS_5_5: "claude-opus-5-5",
493
+ OPUS_5: "claude-opus-5",
494
+ OPUS_4_8: "claude-opus-4-8",
495
+ OPUS_4_7: "claude-opus-4-7",
496
+ OPUS_4_6: "claude-opus-4-6",
497
+ SONNET_5_5: "claude-sonnet-5-5",
498
+ SONNET_5: "claude-sonnet-5",
499
+ SONNET_4_6: "claude-sonnet-4-6",
500
+ HAIKU_5_5: "claude-haiku-5-5",
501
+ HAIKU_4_5: "claude-haiku-4-5"
502
+ },
503
+ OpenAI: {
504
+ GPT_6_ASTRA: "gpt-6-astra",
505
+ GPT_6_1_SOL: "gpt-6.1-sol",
506
+ GPT_6_SOL: "gpt-6-sol",
507
+ GPT_6_LUNA: "gpt-6-luna",
508
+ GPT_5_6_SOL: "gpt-5.6-sol",
509
+ GPT_5_6_TERRA: "gpt-5.6-terra",
510
+ GPT_5_6_LUNA: "gpt-5.6-luna",
511
+ GPT_5_5: "gpt-5.5",
512
+ GPT_5_4: "gpt-5.4",
513
+ GPT_5_4_PRO: "gpt-5.4-pro",
514
+ GPT_5_4_MINI: "gpt-5.4-mini",
515
+ GPT_5_4_NANO: "gpt-5.4-nano",
516
+ GPT_5_2: "gpt-5.2",
517
+ GPT_5_MINI: "gpt-5-mini"
518
+ },
519
+ Google: {
520
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
521
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
522
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
523
+ GEMINI_3_5_FLASH: "gemini-3.5-flash",
524
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
525
+ GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
526
+ GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
527
+ GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
528
+ },
529
+ /**
530
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
531
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
532
+ * recognizes them. All listed models accept image input.
533
+ */
534
+ OpenRouter: {
535
+ MUSE_SPARK_1_3: "meta/muse-spark-1.3",
536
+ GROK_4_7: "x-ai/grok-4.7",
537
+ GROK_4_6: "x-ai/grok-4.6",
538
+ GROK_4_5: "x-ai/grok-4.5",
539
+ KIMI_K3: "moonshotai/kimi-k3",
540
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
541
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
542
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
543
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
544
+ GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
545
+ MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
546
+ }
547
+ };
548
+ var DEFAULT_MODELS = {
549
+ [Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
550
+ [Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
551
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
552
+ [Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
553
+ };
554
+ var DEFAULT_MAX_TOKENS = 4096;
555
+ var OPENAI_REASONING_MAX_TOKENS = 16384;
556
+ var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
557
+ var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
558
+ Model.OpenRouter.QWEN_3_8_MAX,
559
+ Model.OpenRouter.QWEN_3_7_PLUS
560
+ ]);
561
+ var MODEL_TO_PROVIDER = new Map([
562
+ ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
563
+ ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
564
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
565
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
566
+ ]);
567
+ var VALID_PROVIDERS = Object.values(Provider);
568
+ var PROVIDER_DEFAULT_REASONING = {
569
+ openai: "medium",
570
+ anthropic: "off",
571
+ google: "off",
572
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
573
+ openrouter: "off"
574
+ };
575
+ var Content = {
576
+ /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
577
+ PLACEHOLDER_TEXT: "placeholder-text",
578
+ /** Detects error messages, banners, stack traces, or error codes */
579
+ ERROR_MESSAGES: "error-messages",
580
+ /** Detects broken image icons or failed-to-load image indicators */
581
+ BROKEN_IMAGES: "broken-images",
582
+ /** Detects UI elements that unintentionally overlap and obscure content */
583
+ OVERLAPPING_ELEMENTS: "overlapping-elements"
584
+ };
585
+ var Layout = {
586
+ /** Detects elements that unintentionally overlap each other */
587
+ OVERLAP: "overlap",
588
+ /** Detects content cut off or extending beyond container boundaries */
589
+ OVERFLOW: "overflow",
590
+ /** Detects inconsistent alignment of text, images, and UI components */
591
+ ALIGNMENT: "alignment"
592
+ };
593
+ var Accessibility = {
594
+ /** Detects insufficient color contrast between text and backgrounds */
595
+ CONTRAST: "contrast",
596
+ /** Detects text that is cut off, overlapping, too small, or obscured */
597
+ READABILITY: "readability",
598
+ /** Detects interactive elements that are not visually distinct */
599
+ INTERACTIVE_VISIBILITY: "interactive-visibility",
600
+ /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
601
+ COLOR_BLINDNESS: "color-blindness",
602
+ /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
603
+ COLOR_ALONE: "color-alone"
604
+ };
605
+
596
606
  // src/templates/accessibility.ts
597
607
  var ALL_CHECKS = Object.values(Accessibility);
598
608
  var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
@@ -702,17 +712,23 @@ function parseRetryAfter(value) {
702
712
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
703
713
  Model.Anthropic.FABLE_5_1,
704
714
  Model.Anthropic.FABLE_5,
715
+ Model.Anthropic.OPUS_5_5,
705
716
  Model.Anthropic.OPUS_5,
706
717
  Model.Anthropic.OPUS_4_8,
707
718
  Model.Anthropic.OPUS_4_7,
708
- Model.Anthropic.SONNET_5
719
+ Model.Anthropic.SONNET_5_5,
720
+ Model.Anthropic.SONNET_5,
721
+ Model.Anthropic.HAIKU_5_5
709
722
  ]);
710
723
  function mapEffort(level, model) {
724
+ if (level === "minimal") return "low";
711
725
  if (level !== "xhigh") return level;
712
726
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
713
727
  }
714
728
  var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
715
729
  var EFFORT_TO_BUDGET_TOKENS = {
730
+ // 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
731
+ minimal: 1024,
716
732
  low: 1024,
717
733
  medium: 4096,
718
734
  high: 8192,
@@ -829,6 +845,9 @@ function sleep(ms) {
829
845
  return new Promise((resolve2) => setTimeout(resolve2, ms));
830
846
  }
831
847
  var GOOGLE_THINKING_LEVEL = {
848
+ // Gemini does define a "minimal" thinking level, but some models reject it
849
+ // (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
850
+ minimal: "low",
832
851
  low: "low",
833
852
  medium: "medium",
834
853
  high: "high",
@@ -1146,6 +1165,9 @@ var OpenAIDriver = class {
1146
1165
  // src/providers/openrouter.ts
1147
1166
  var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
1148
1167
  var OPENROUTER_REASONING_EFFORT = {
1168
+ // OpenRouter normalizes upstream vendors to low/medium/high only, so
1169
+ // "minimal" has no native equivalent and clamps to the floor.
1170
+ minimal: "low",
1149
1171
  low: "low",
1150
1172
  medium: "medium",
1151
1173
  high: "high",
@@ -1305,10 +1327,20 @@ function parseBooleanEnv(envName, value) {
1305
1327
  `Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
1306
1328
  );
1307
1329
  }
1330
+ function parseReasoningEffortEnv(envName, value) {
1331
+ if (value === void 0 || value === "") return void 0;
1332
+ const levels = Object.values(ReasoningEffort);
1333
+ const lower = value.toLowerCase();
1334
+ if (levels.includes(lower)) return lower;
1335
+ throw new VisualAIConfigError(
1336
+ `Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
1337
+ );
1338
+ }
1308
1339
  var debugDeprecationWarned = false;
1309
1340
  function resolveConfig(config) {
1310
1341
  const provider = resolveProvider(config);
1311
1342
  const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
1343
+ const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
1312
1344
  const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
1313
1345
  const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
1314
1346
  const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
@@ -1326,12 +1358,12 @@ function resolveConfig(config) {
1326
1358
  }
1327
1359
  const userSetMaxTokens = config.maxTokens !== void 0;
1328
1360
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
1329
- const effortNeedsLargeBudget = config.reasoningEffort === "high" || config.reasoningEffort === "xhigh";
1361
+ const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
1330
1362
  const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
1331
1363
  if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
1332
1364
  maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
1333
1365
  if (debug) {
1334
- const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${config.reasoningEffort}"`;
1366
+ const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
1335
1367
  process.stderr.write(
1336
1368
  `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
1337
1369
  `
@@ -1343,7 +1375,7 @@ function resolveConfig(config) {
1343
1375
  apiKey: config.apiKey,
1344
1376
  model,
1345
1377
  maxTokens,
1346
- reasoningEffort: config.reasoningEffort,
1378
+ reasoningEffort,
1347
1379
  maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1348
1380
  imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1349
1381
  timeout: config.timeout,
@@ -1366,6 +1398,10 @@ var PRICING_TABLE = {
1366
1398
  inputPricePerToken: 10 / PER_MILLION,
1367
1399
  outputPricePerToken: 50 / PER_MILLION
1368
1400
  },
1401
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
1402
+ inputPricePerToken: 4 / PER_MILLION,
1403
+ outputPricePerToken: 20 / PER_MILLION
1404
+ },
1369
1405
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1370
1406
  inputPricePerToken: 5 / PER_MILLION,
1371
1407
  outputPricePerToken: 25 / PER_MILLION
@@ -1374,6 +1410,10 @@ var PRICING_TABLE = {
1374
1410
  inputPricePerToken: 5 / PER_MILLION,
1375
1411
  outputPricePerToken: 25 / PER_MILLION
1376
1412
  },
1413
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
1414
+ inputPricePerToken: 2 / PER_MILLION,
1415
+ outputPricePerToken: 10 / PER_MILLION
1416
+ },
1377
1417
  [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
1378
1418
  inputPricePerToken: 3 / PER_MILLION,
1379
1419
  outputPricePerToken: 15 / PER_MILLION
@@ -1390,6 +1430,12 @@ var PRICING_TABLE = {
1390
1430
  inputPricePerToken: 3 / PER_MILLION,
1391
1431
  outputPricePerToken: 15 / PER_MILLION
1392
1432
  },
1433
+ // Prompts above 100K tokens bill at $0.50/$2.50, far beyond screenshot-sized
1434
+ // calls. Cached input is $0.01/MTok and cache writes $0.125/MTok (not modelled).
1435
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.HAIKU_5_5}`]: {
1436
+ inputPricePerToken: 0.1 / PER_MILLION,
1437
+ outputPricePerToken: 0.5 / PER_MILLION
1438
+ },
1393
1439
  [`${Provider.ANTHROPIC}:${Model.Anthropic.HAIKU_4_5}`]: {
1394
1440
  inputPricePerToken: 1 / PER_MILLION,
1395
1441
  outputPricePerToken: 5 / PER_MILLION
@@ -1400,6 +1446,22 @@ var PRICING_TABLE = {
1400
1446
  inputPricePerToken: 10 / PER_MILLION,
1401
1447
  outputPricePerToken: 50 / PER_MILLION
1402
1448
  },
1449
+ // Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
1450
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
1451
+ inputPricePerToken: 2 / PER_MILLION,
1452
+ outputPricePerToken: 10 / PER_MILLION
1453
+ },
1454
+ // Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
1455
+ // bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
1456
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
1457
+ inputPricePerToken: 2 / PER_MILLION,
1458
+ outputPricePerToken: 10 / PER_MILLION
1459
+ },
1460
+ // Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
1461
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
1462
+ inputPricePerToken: 0.1 / PER_MILLION,
1463
+ outputPricePerToken: 0.5 / PER_MILLION
1464
+ },
1403
1465
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
1404
1466
  inputPricePerToken: 5 / PER_MILLION,
1405
1467
  outputPricePerToken: 30 / PER_MILLION
@@ -1496,6 +1558,10 @@ var PRICING_TABLE = {
1496
1558
  inputPricePerToken: 0.1 / PER_MILLION,
1497
1559
  outputPricePerToken: 0.2 / PER_MILLION
1498
1560
  },
1561
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
1562
+ inputPricePerToken: 1.6 / PER_MILLION,
1563
+ outputPricePerToken: 4.8 / PER_MILLION
1564
+ },
1499
1565
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1500
1566
  inputPricePerToken: 2 / PER_MILLION,
1501
1567
  outputPricePerToken: 6 / PER_MILLION
@@ -1529,6 +1595,13 @@ var PRICING_TABLE = {
1529
1595
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
1530
1596
  inputPricePerToken: 0.15 / PER_MILLION,
1531
1597
  outputPricePerToken: 0.5 / PER_MILLION
1598
+ },
1599
+ // Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
1600
+ // upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
1601
+ // is $0.0036/MTok, not modelled (no provider gets a cache discount here).
1602
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
1603
+ inputPricePerToken: 0.435 / PER_MILLION,
1604
+ outputPricePerToken: 0.87 / PER_MILLION
1532
1605
  }
1533
1606
  };
1534
1607
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -2550,6 +2623,12 @@ function stripCodeFences(text) {
2550
2623
  }
2551
2624
  var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
2552
2625
  var AskResponseSchema = AskResultSchema.omit({ usage: true });
2626
+ var AskImageResponseSchema = AskResponseSchema.omit({
2627
+ frameReferences: true,
2628
+ timestampReferences: true
2629
+ });
2630
+ var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
2631
+ var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
2553
2632
  var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
2554
2633
  var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
2555
2634
  function parseJson(text) {
@@ -2642,7 +2721,11 @@ function createDriver(provider, config) {
2642
2721
  return PROVIDER_REGISTRY[provider](config);
2643
2722
  }
2644
2723
  var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
2645
- var askSchemaOptions = toSchemaOptions(AskResponseSchema);
2724
+ var askSchemaOptionsByMedia = {
2725
+ image: toSchemaOptions(AskImageResponseSchema),
2726
+ video: toSchemaOptions(AskFramesResponseSchema),
2727
+ "native-video": toSchemaOptions(AskNativeVideoResponseSchema)
2728
+ };
2646
2729
  var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2647
2730
  function mediaToProviderInputs(media) {
2648
2731
  if (media.kind === "image") {
@@ -2787,7 +2870,12 @@ function visualAI(config = {}) {
2787
2870
  media: dispatch.mediaContext
2788
2871
  });
2789
2872
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2790
- const { response, metadata } = await sendMedia(driver, dispatch, prompt, askSchemaOptions);
2873
+ const { response, metadata } = await sendMedia(
2874
+ driver,
2875
+ dispatch,
2876
+ prompt,
2877
+ askSchemaOptionsByMedia[dispatch.mediaContext.kind]
2878
+ );
2791
2879
  debugLog(resolvedConfig, "ask response", response.text, "response");
2792
2880
  const result = parseAskResponse(response.text);
2793
2881
  return {
@@ -2810,7 +2898,7 @@ function visualAI(config = {}) {
2810
2898
  debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
2811
2899
  const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
2812
2900
  debugLog(resolvedConfig, "compare response", response.text, "response");
2813
- const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model === Model.Google.GEMINI_3_FLASH_PREVIEW;
2901
+ const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
2814
2902
  const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
2815
2903
  let diffImage;
2816
2904
  if (effectiveDiffImage) {