visual-ai-assertions 0.25.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,132 +1,3 @@
1
- // src/constants.ts
2
- var ReasoningEffort = {
3
- LOW: "low",
4
- MEDIUM: "medium",
5
- HIGH: "high",
6
- XHIGH: "xhigh"
7
- };
8
- var ImageDetail = {
9
- AUTO: "auto",
10
- LOW: "low",
11
- HIGH: "high"
12
- };
13
- var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
14
- var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
15
- var Provider = {
16
- ANTHROPIC: "anthropic",
17
- OPENAI: "openai",
18
- GOOGLE: "google",
19
- OPENROUTER: "openrouter"
20
- };
21
- var Model = {
22
- Anthropic: {
23
- FABLE_5_1: "claude-fable-5-1",
24
- FABLE_5: "claude-fable-5",
25
- OPUS_5: "claude-opus-5",
26
- OPUS_4_8: "claude-opus-4-8",
27
- OPUS_4_7: "claude-opus-4-7",
28
- OPUS_4_6: "claude-opus-4-6",
29
- SONNET_5: "claude-sonnet-5",
30
- SONNET_4_6: "claude-sonnet-4-6",
31
- HAIKU_4_5: "claude-haiku-4-5"
32
- },
33
- OpenAI: {
34
- GPT_6_ASTRA: "gpt-6-astra",
35
- GPT_5_6_SOL: "gpt-5.6-sol",
36
- GPT_5_6_TERRA: "gpt-5.6-terra",
37
- GPT_5_6_LUNA: "gpt-5.6-luna",
38
- GPT_5_5: "gpt-5.5",
39
- GPT_5_4: "gpt-5.4",
40
- GPT_5_4_PRO: "gpt-5.4-pro",
41
- GPT_5_4_MINI: "gpt-5.4-mini",
42
- GPT_5_4_NANO: "gpt-5.4-nano",
43
- GPT_5_2: "gpt-5.2",
44
- GPT_5_MINI: "gpt-5-mini"
45
- },
46
- Google: {
47
- GEMINI_3_8_FLASH: "gemini-3.8-flash",
48
- GEMINI_3_7_FLASH: "gemini-3.7-flash",
49
- GEMINI_3_6_FLASH: "gemini-3.6-flash",
50
- GEMINI_3_5_FLASH: "gemini-3.5-flash",
51
- GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
52
- GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
53
- GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
54
- GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
55
- },
56
- /**
57
- * Models routed through OpenRouter (https://openrouter.ai). Slugs always
58
- * carry a vendor prefix (`vendor/model`), which is how provider inference
59
- * recognizes them. All listed models accept image input.
60
- */
61
- OpenRouter: {
62
- MUSE_SPARK_1_3: "meta/muse-spark-1.3",
63
- GROK_4_6: "x-ai/grok-4.6",
64
- GROK_4_5: "x-ai/grok-4.5",
65
- KIMI_K3: "moonshotai/kimi-k3",
66
- KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
67
- QWEN_3_8_MAX: "qwen/qwen3.8-max",
68
- QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
69
- QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
70
- GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
71
- }
72
- };
73
- var DEFAULT_MODELS = {
74
- [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
75
- [Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
76
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
77
- [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
78
- };
79
- var DEFAULT_MAX_TOKENS = 4096;
80
- var OPENAI_REASONING_MAX_TOKENS = 16384;
81
- var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
82
- var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
83
- Model.OpenAI.GPT_6_ASTRA
84
- ]);
85
- var MODEL_TO_PROVIDER = new Map([
86
- ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
87
- ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
88
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
89
- ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
90
- ]);
91
- var VALID_PROVIDERS = Object.values(Provider);
92
- var PROVIDER_DEFAULT_REASONING = {
93
- openai: "medium",
94
- anthropic: "off",
95
- google: "off",
96
- // Varies by upstream model; the driver sends no reasoning field unless configured.
97
- openrouter: "off"
98
- };
99
- var Content = {
100
- /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
101
- PLACEHOLDER_TEXT: "placeholder-text",
102
- /** Detects error messages, banners, stack traces, or error codes */
103
- ERROR_MESSAGES: "error-messages",
104
- /** Detects broken image icons or failed-to-load image indicators */
105
- BROKEN_IMAGES: "broken-images",
106
- /** Detects UI elements that unintentionally overlap and obscure content */
107
- OVERLAPPING_ELEMENTS: "overlapping-elements"
108
- };
109
- var Layout = {
110
- /** Detects elements that unintentionally overlap each other */
111
- OVERLAP: "overlap",
112
- /** Detects content cut off or extending beyond container boundaries */
113
- OVERFLOW: "overflow",
114
- /** Detects inconsistent alignment of text, images, and UI components */
115
- ALIGNMENT: "alignment"
116
- };
117
- var Accessibility = {
118
- /** Detects insufficient color contrast between text and backgrounds */
119
- CONTRAST: "contrast",
120
- /** Detects text that is cut off, overlapping, too small, or obscured */
121
- READABILITY: "readability",
122
- /** Detects interactive elements that are not visually distinct */
123
- INTERACTIVE_VISIBILITY: "interactive-visibility",
124
- /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
125
- COLOR_BLINDNESS: "color-blindness",
126
- /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
127
- COLOR_ALONE: "color-alone"
128
- };
129
-
130
1
  // src/errors.ts
131
2
  var VisualAIError = class extends Error {
132
3
  code;
@@ -493,7 +364,7 @@ function visibleRole(finalState, requireCorrectRendering) {
493
364
  var ELEMENTS_VISIBLE_CLIPPING_RULES = [
494
365
  "When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
495
366
  "An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
496
- "An element you cannot see at all is not visible, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
367
+ "The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
497
368
  ];
498
369
  var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
499
370
  var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
@@ -524,6 +395,145 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
524
395
  return buildCheckPrompt(statements, { role, instructions });
525
396
  }
526
397
 
398
+ // src/constants.ts
399
+ var ReasoningEffort = {
400
+ MINIMAL: "minimal",
401
+ LOW: "low",
402
+ MEDIUM: "medium",
403
+ HIGH: "high",
404
+ XHIGH: "xhigh"
405
+ };
406
+ var ImageDetail = {
407
+ AUTO: "auto",
408
+ LOW: "low",
409
+ HIGH: "high"
410
+ };
411
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
412
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
413
+ var Provider = {
414
+ ANTHROPIC: "anthropic",
415
+ OPENAI: "openai",
416
+ GOOGLE: "google",
417
+ OPENROUTER: "openrouter"
418
+ };
419
+ var Model = {
420
+ Anthropic: {
421
+ FABLE_5_1: "claude-fable-5-1",
422
+ FABLE_5: "claude-fable-5",
423
+ OPUS_5_5: "claude-opus-5-5",
424
+ OPUS_5: "claude-opus-5",
425
+ OPUS_4_8: "claude-opus-4-8",
426
+ OPUS_4_7: "claude-opus-4-7",
427
+ OPUS_4_6: "claude-opus-4-6",
428
+ SONNET_5_5: "claude-sonnet-5-5",
429
+ SONNET_5: "claude-sonnet-5",
430
+ SONNET_4_6: "claude-sonnet-4-6",
431
+ HAIKU_5_5: "claude-haiku-5-5",
432
+ HAIKU_4_5: "claude-haiku-4-5"
433
+ },
434
+ OpenAI: {
435
+ GPT_6_ASTRA: "gpt-6-astra",
436
+ GPT_6_1_SOL: "gpt-6.1-sol",
437
+ GPT_6_SOL: "gpt-6-sol",
438
+ GPT_6_LUNA: "gpt-6-luna",
439
+ GPT_5_6_SOL: "gpt-5.6-sol",
440
+ GPT_5_6_TERRA: "gpt-5.6-terra",
441
+ GPT_5_6_LUNA: "gpt-5.6-luna",
442
+ GPT_5_5: "gpt-5.5",
443
+ GPT_5_4: "gpt-5.4",
444
+ GPT_5_4_PRO: "gpt-5.4-pro",
445
+ GPT_5_4_MINI: "gpt-5.4-mini",
446
+ GPT_5_4_NANO: "gpt-5.4-nano",
447
+ GPT_5_2: "gpt-5.2",
448
+ GPT_5_MINI: "gpt-5-mini"
449
+ },
450
+ Google: {
451
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
452
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
453
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
454
+ GEMINI_3_5_FLASH: "gemini-3.5-flash",
455
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
456
+ GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
457
+ GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
458
+ GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
459
+ },
460
+ /**
461
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
462
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
463
+ * recognizes them. All listed models accept image input.
464
+ */
465
+ OpenRouter: {
466
+ MUSE_SPARK_1_3: "meta/muse-spark-1.3",
467
+ GROK_4_7: "x-ai/grok-4.7",
468
+ GROK_4_6: "x-ai/grok-4.6",
469
+ GROK_4_5: "x-ai/grok-4.5",
470
+ KIMI_K3: "moonshotai/kimi-k3",
471
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
472
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
473
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
474
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
475
+ GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
476
+ MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
477
+ }
478
+ };
479
+ var DEFAULT_MODELS = {
480
+ [Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
481
+ [Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
482
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
483
+ [Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
484
+ };
485
+ var DEFAULT_MAX_TOKENS = 4096;
486
+ var OPENAI_REASONING_MAX_TOKENS = 16384;
487
+ var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
488
+ var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
489
+ Model.OpenRouter.QWEN_3_8_MAX,
490
+ Model.OpenRouter.QWEN_3_7_PLUS
491
+ ]);
492
+ var MODEL_TO_PROVIDER = new Map([
493
+ ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
494
+ ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
495
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
496
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
497
+ ]);
498
+ var VALID_PROVIDERS = Object.values(Provider);
499
+ var PROVIDER_DEFAULT_REASONING = {
500
+ openai: "medium",
501
+ anthropic: "off",
502
+ google: "off",
503
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
504
+ openrouter: "off"
505
+ };
506
+ var Content = {
507
+ /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
508
+ PLACEHOLDER_TEXT: "placeholder-text",
509
+ /** Detects error messages, banners, stack traces, or error codes */
510
+ ERROR_MESSAGES: "error-messages",
511
+ /** Detects broken image icons or failed-to-load image indicators */
512
+ BROKEN_IMAGES: "broken-images",
513
+ /** Detects UI elements that unintentionally overlap and obscure content */
514
+ OVERLAPPING_ELEMENTS: "overlapping-elements"
515
+ };
516
+ var Layout = {
517
+ /** Detects elements that unintentionally overlap each other */
518
+ OVERLAP: "overlap",
519
+ /** Detects content cut off or extending beyond container boundaries */
520
+ OVERFLOW: "overflow",
521
+ /** Detects inconsistent alignment of text, images, and UI components */
522
+ ALIGNMENT: "alignment"
523
+ };
524
+ var Accessibility = {
525
+ /** Detects insufficient color contrast between text and backgrounds */
526
+ CONTRAST: "contrast",
527
+ /** Detects text that is cut off, overlapping, too small, or obscured */
528
+ READABILITY: "readability",
529
+ /** Detects interactive elements that are not visually distinct */
530
+ INTERACTIVE_VISIBILITY: "interactive-visibility",
531
+ /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
532
+ COLOR_BLINDNESS: "color-blindness",
533
+ /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
534
+ COLOR_ALONE: "color-alone"
535
+ };
536
+
527
537
  // src/templates/accessibility.ts
528
538
  var ALL_CHECKS = Object.values(Accessibility);
529
539
  var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
@@ -633,17 +643,23 @@ function parseRetryAfter(value) {
633
643
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
634
644
  Model.Anthropic.FABLE_5_1,
635
645
  Model.Anthropic.FABLE_5,
646
+ Model.Anthropic.OPUS_5_5,
636
647
  Model.Anthropic.OPUS_5,
637
648
  Model.Anthropic.OPUS_4_8,
638
649
  Model.Anthropic.OPUS_4_7,
639
- Model.Anthropic.SONNET_5
650
+ Model.Anthropic.SONNET_5_5,
651
+ Model.Anthropic.SONNET_5,
652
+ Model.Anthropic.HAIKU_5_5
640
653
  ]);
641
654
  function mapEffort(level, model) {
655
+ if (level === "minimal") return "low";
642
656
  if (level !== "xhigh") return level;
643
657
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
644
658
  }
645
659
  var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
646
660
  var EFFORT_TO_BUDGET_TOKENS = {
661
+ // 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
662
+ minimal: 1024,
647
663
  low: 1024,
648
664
  medium: 4096,
649
665
  high: 8192,
@@ -760,6 +776,9 @@ function sleep(ms) {
760
776
  return new Promise((resolve2) => setTimeout(resolve2, ms));
761
777
  }
762
778
  var GOOGLE_THINKING_LEVEL = {
779
+ // Gemini does define a "minimal" thinking level, but some models reject it
780
+ // (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
781
+ minimal: "low",
763
782
  low: "low",
764
783
  medium: "medium",
765
784
  high: "high",
@@ -1077,6 +1096,9 @@ var OpenAIDriver = class {
1077
1096
  // src/providers/openrouter.ts
1078
1097
  var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
1079
1098
  var OPENROUTER_REASONING_EFFORT = {
1099
+ // OpenRouter normalizes upstream vendors to low/medium/high only, so
1100
+ // "minimal" has no native equivalent and clamps to the floor.
1101
+ minimal: "low",
1080
1102
  low: "low",
1081
1103
  medium: "medium",
1082
1104
  high: "high",
@@ -1236,10 +1258,20 @@ function parseBooleanEnv(envName, value) {
1236
1258
  `Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
1237
1259
  );
1238
1260
  }
1261
+ function parseReasoningEffortEnv(envName, value) {
1262
+ if (value === void 0 || value === "") return void 0;
1263
+ const levels = Object.values(ReasoningEffort);
1264
+ const lower = value.toLowerCase();
1265
+ if (levels.includes(lower)) return lower;
1266
+ throw new VisualAIConfigError(
1267
+ `Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
1268
+ );
1269
+ }
1239
1270
  var debugDeprecationWarned = false;
1240
1271
  function resolveConfig(config) {
1241
1272
  const provider = resolveProvider(config);
1242
1273
  const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
1274
+ const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
1243
1275
  const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
1244
1276
  const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
1245
1277
  const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
@@ -1257,12 +1289,12 @@ function resolveConfig(config) {
1257
1289
  }
1258
1290
  const userSetMaxTokens = config.maxTokens !== void 0;
1259
1291
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
1260
- const effortNeedsLargeBudget = config.reasoningEffort === "high" || config.reasoningEffort === "xhigh";
1292
+ const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
1261
1293
  const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
1262
1294
  if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
1263
1295
  maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
1264
1296
  if (debug) {
1265
- const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${config.reasoningEffort}"`;
1297
+ const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
1266
1298
  process.stderr.write(
1267
1299
  `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
1268
1300
  `
@@ -1274,7 +1306,7 @@ function resolveConfig(config) {
1274
1306
  apiKey: config.apiKey,
1275
1307
  model,
1276
1308
  maxTokens,
1277
- reasoningEffort: config.reasoningEffort,
1309
+ reasoningEffort,
1278
1310
  maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1279
1311
  imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1280
1312
  timeout: config.timeout,
@@ -1297,6 +1329,10 @@ var PRICING_TABLE = {
1297
1329
  inputPricePerToken: 10 / PER_MILLION,
1298
1330
  outputPricePerToken: 50 / PER_MILLION
1299
1331
  },
1332
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
1333
+ inputPricePerToken: 4 / PER_MILLION,
1334
+ outputPricePerToken: 20 / PER_MILLION
1335
+ },
1300
1336
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1301
1337
  inputPricePerToken: 5 / PER_MILLION,
1302
1338
  outputPricePerToken: 25 / PER_MILLION
@@ -1305,6 +1341,10 @@ var PRICING_TABLE = {
1305
1341
  inputPricePerToken: 5 / PER_MILLION,
1306
1342
  outputPricePerToken: 25 / PER_MILLION
1307
1343
  },
1344
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
1345
+ inputPricePerToken: 2 / PER_MILLION,
1346
+ outputPricePerToken: 10 / PER_MILLION
1347
+ },
1308
1348
  [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
1309
1349
  inputPricePerToken: 3 / PER_MILLION,
1310
1350
  outputPricePerToken: 15 / PER_MILLION
@@ -1321,6 +1361,12 @@ var PRICING_TABLE = {
1321
1361
  inputPricePerToken: 3 / PER_MILLION,
1322
1362
  outputPricePerToken: 15 / PER_MILLION
1323
1363
  },
1364
+ // Prompts above 100K tokens bill at $0.50/$2.50, far beyond screenshot-sized
1365
+ // calls. Cached input is $0.01/MTok and cache writes $0.125/MTok (not modelled).
1366
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.HAIKU_5_5}`]: {
1367
+ inputPricePerToken: 0.1 / PER_MILLION,
1368
+ outputPricePerToken: 0.5 / PER_MILLION
1369
+ },
1324
1370
  [`${Provider.ANTHROPIC}:${Model.Anthropic.HAIKU_4_5}`]: {
1325
1371
  inputPricePerToken: 1 / PER_MILLION,
1326
1372
  outputPricePerToken: 5 / PER_MILLION
@@ -1331,6 +1377,22 @@ var PRICING_TABLE = {
1331
1377
  inputPricePerToken: 10 / PER_MILLION,
1332
1378
  outputPricePerToken: 50 / PER_MILLION
1333
1379
  },
1380
+ // Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
1381
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
1382
+ inputPricePerToken: 2 / PER_MILLION,
1383
+ outputPricePerToken: 10 / PER_MILLION
1384
+ },
1385
+ // Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
1386
+ // bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
1387
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
1388
+ inputPricePerToken: 2 / PER_MILLION,
1389
+ outputPricePerToken: 10 / PER_MILLION
1390
+ },
1391
+ // Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
1392
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
1393
+ inputPricePerToken: 0.1 / PER_MILLION,
1394
+ outputPricePerToken: 0.5 / PER_MILLION
1395
+ },
1334
1396
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
1335
1397
  inputPricePerToken: 5 / PER_MILLION,
1336
1398
  outputPricePerToken: 30 / PER_MILLION
@@ -1427,6 +1489,10 @@ var PRICING_TABLE = {
1427
1489
  inputPricePerToken: 0.1 / PER_MILLION,
1428
1490
  outputPricePerToken: 0.2 / PER_MILLION
1429
1491
  },
1492
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
1493
+ inputPricePerToken: 1.6 / PER_MILLION,
1494
+ outputPricePerToken: 4.8 / PER_MILLION
1495
+ },
1430
1496
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1431
1497
  inputPricePerToken: 2 / PER_MILLION,
1432
1498
  outputPricePerToken: 6 / PER_MILLION
@@ -1460,6 +1526,13 @@ var PRICING_TABLE = {
1460
1526
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
1461
1527
  inputPricePerToken: 0.15 / PER_MILLION,
1462
1528
  outputPricePerToken: 0.5 / PER_MILLION
1529
+ },
1530
+ // Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
1531
+ // upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
1532
+ // is $0.0036/MTok, not modelled (no provider gets a cache discount here).
1533
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
1534
+ inputPricePerToken: 0.435 / PER_MILLION,
1535
+ outputPricePerToken: 0.87 / PER_MILLION
1463
1536
  }
1464
1537
  };
1465
1538
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -2481,6 +2554,12 @@ function stripCodeFences(text) {
2481
2554
  }
2482
2555
  var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
2483
2556
  var AskResponseSchema = AskResultSchema.omit({ usage: true });
2557
+ var AskImageResponseSchema = AskResponseSchema.omit({
2558
+ frameReferences: true,
2559
+ timestampReferences: true
2560
+ });
2561
+ var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
2562
+ var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
2484
2563
  var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
2485
2564
  var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
2486
2565
  function parseJson(text) {
@@ -2573,7 +2652,11 @@ function createDriver(provider, config) {
2573
2652
  return PROVIDER_REGISTRY[provider](config);
2574
2653
  }
2575
2654
  var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
2576
- var askSchemaOptions = toSchemaOptions(AskResponseSchema);
2655
+ var askSchemaOptionsByMedia = {
2656
+ image: toSchemaOptions(AskImageResponseSchema),
2657
+ video: toSchemaOptions(AskFramesResponseSchema),
2658
+ "native-video": toSchemaOptions(AskNativeVideoResponseSchema)
2659
+ };
2577
2660
  var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2578
2661
  function mediaToProviderInputs(media) {
2579
2662
  if (media.kind === "image") {
@@ -2718,7 +2801,12 @@ function visualAI(config = {}) {
2718
2801
  media: dispatch.mediaContext
2719
2802
  });
2720
2803
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2721
- const { response, metadata } = await sendMedia(driver, dispatch, prompt, askSchemaOptions);
2804
+ const { response, metadata } = await sendMedia(
2805
+ driver,
2806
+ dispatch,
2807
+ prompt,
2808
+ askSchemaOptionsByMedia[dispatch.mediaContext.kind]
2809
+ );
2722
2810
  debugLog(resolvedConfig, "ask response", response.text, "response");
2723
2811
  const result = parseAskResponse(response.text);
2724
2812
  return {
@@ -2741,7 +2829,7 @@ function visualAI(config = {}) {
2741
2829
  debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
2742
2830
  const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
2743
2831
  debugLog(resolvedConfig, "compare response", response.text, "response");
2744
- const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model === Model.Google.GEMINI_3_FLASH_PREVIEW;
2832
+ const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
2745
2833
  const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
2746
2834
  let diffImage;
2747
2835
  if (effectiveDiffImage) {