visual-ai-assertions 0.25.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -67,135 +67,6 @@ __export(index_exports, {
67
67
  });
68
68
  module.exports = __toCommonJS(index_exports);
69
69
 
70
- // src/constants.ts
71
- var ReasoningEffort = {
72
- LOW: "low",
73
- MEDIUM: "medium",
74
- HIGH: "high",
75
- XHIGH: "xhigh"
76
- };
77
- var ImageDetail = {
78
- AUTO: "auto",
79
- LOW: "low",
80
- HIGH: "high"
81
- };
82
- var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
83
- var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
84
- var Provider = {
85
- ANTHROPIC: "anthropic",
86
- OPENAI: "openai",
87
- GOOGLE: "google",
88
- OPENROUTER: "openrouter"
89
- };
90
- var Model = {
91
- Anthropic: {
92
- FABLE_5_1: "claude-fable-5-1",
93
- FABLE_5: "claude-fable-5",
94
- OPUS_5: "claude-opus-5",
95
- OPUS_4_8: "claude-opus-4-8",
96
- OPUS_4_7: "claude-opus-4-7",
97
- OPUS_4_6: "claude-opus-4-6",
98
- SONNET_5: "claude-sonnet-5",
99
- SONNET_4_6: "claude-sonnet-4-6",
100
- HAIKU_4_5: "claude-haiku-4-5"
101
- },
102
- OpenAI: {
103
- GPT_6_ASTRA: "gpt-6-astra",
104
- GPT_5_6_SOL: "gpt-5.6-sol",
105
- GPT_5_6_TERRA: "gpt-5.6-terra",
106
- GPT_5_6_LUNA: "gpt-5.6-luna",
107
- GPT_5_5: "gpt-5.5",
108
- GPT_5_4: "gpt-5.4",
109
- GPT_5_4_PRO: "gpt-5.4-pro",
110
- GPT_5_4_MINI: "gpt-5.4-mini",
111
- GPT_5_4_NANO: "gpt-5.4-nano",
112
- GPT_5_2: "gpt-5.2",
113
- GPT_5_MINI: "gpt-5-mini"
114
- },
115
- Google: {
116
- GEMINI_3_8_FLASH: "gemini-3.8-flash",
117
- GEMINI_3_7_FLASH: "gemini-3.7-flash",
118
- GEMINI_3_6_FLASH: "gemini-3.6-flash",
119
- GEMINI_3_5_FLASH: "gemini-3.5-flash",
120
- GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
121
- GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
122
- GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
123
- GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
124
- },
125
- /**
126
- * Models routed through OpenRouter (https://openrouter.ai). Slugs always
127
- * carry a vendor prefix (`vendor/model`), which is how provider inference
128
- * recognizes them. All listed models accept image input.
129
- */
130
- OpenRouter: {
131
- MUSE_SPARK_1_3: "meta/muse-spark-1.3",
132
- GROK_4_6: "x-ai/grok-4.6",
133
- GROK_4_5: "x-ai/grok-4.5",
134
- KIMI_K3: "moonshotai/kimi-k3",
135
- KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
136
- QWEN_3_8_MAX: "qwen/qwen3.8-max",
137
- QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
138
- QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
139
- GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
140
- }
141
- };
142
- var DEFAULT_MODELS = {
143
- [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
144
- [Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
145
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
146
- [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
147
- };
148
- var DEFAULT_MAX_TOKENS = 4096;
149
- var OPENAI_REASONING_MAX_TOKENS = 16384;
150
- var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
151
- var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
152
- Model.OpenAI.GPT_6_ASTRA
153
- ]);
154
- var MODEL_TO_PROVIDER = new Map([
155
- ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
156
- ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
157
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
158
- ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
159
- ]);
160
- var VALID_PROVIDERS = Object.values(Provider);
161
- var PROVIDER_DEFAULT_REASONING = {
162
- openai: "medium",
163
- anthropic: "off",
164
- google: "off",
165
- // Varies by upstream model; the driver sends no reasoning field unless configured.
166
- openrouter: "off"
167
- };
168
- var Content = {
169
- /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
170
- PLACEHOLDER_TEXT: "placeholder-text",
171
- /** Detects error messages, banners, stack traces, or error codes */
172
- ERROR_MESSAGES: "error-messages",
173
- /** Detects broken image icons or failed-to-load image indicators */
174
- BROKEN_IMAGES: "broken-images",
175
- /** Detects UI elements that unintentionally overlap and obscure content */
176
- OVERLAPPING_ELEMENTS: "overlapping-elements"
177
- };
178
- var Layout = {
179
- /** Detects elements that unintentionally overlap each other */
180
- OVERLAP: "overlap",
181
- /** Detects content cut off or extending beyond container boundaries */
182
- OVERFLOW: "overflow",
183
- /** Detects inconsistent alignment of text, images, and UI components */
184
- ALIGNMENT: "alignment"
185
- };
186
- var Accessibility = {
187
- /** Detects insufficient color contrast between text and backgrounds */
188
- CONTRAST: "contrast",
189
- /** Detects text that is cut off, overlapping, too small, or obscured */
190
- READABILITY: "readability",
191
- /** Detects interactive elements that are not visually distinct */
192
- INTERACTIVE_VISIBILITY: "interactive-visibility",
193
- /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
194
- COLOR_BLINDNESS: "color-blindness",
195
- /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
196
- COLOR_ALONE: "color-alone"
197
- };
198
-
199
70
  // src/errors.ts
200
71
  var VisualAIError = class extends Error {
201
72
  code;
@@ -562,7 +433,7 @@ function visibleRole(finalState, requireCorrectRendering) {
562
433
  var ELEMENTS_VISIBLE_CLIPPING_RULES = [
563
434
  "When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
564
435
  "An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
565
- "An element you cannot see at all is not visible, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
436
+ "The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
566
437
  ];
567
438
  var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
568
439
  var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
@@ -593,6 +464,144 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
593
464
  return buildCheckPrompt(statements, { role, instructions });
594
465
  }
595
466
 
467
+ // src/constants.ts
468
+ var ReasoningEffort = {
469
+ MINIMAL: "minimal",
470
+ LOW: "low",
471
+ MEDIUM: "medium",
472
+ HIGH: "high",
473
+ XHIGH: "xhigh"
474
+ };
475
+ var ImageDetail = {
476
+ AUTO: "auto",
477
+ LOW: "low",
478
+ HIGH: "high"
479
+ };
480
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
481
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
482
+ var Provider = {
483
+ ANTHROPIC: "anthropic",
484
+ OPENAI: "openai",
485
+ GOOGLE: "google",
486
+ OPENROUTER: "openrouter"
487
+ };
488
+ var Model = {
489
+ Anthropic: {
490
+ FABLE_5_1: "claude-fable-5-1",
491
+ FABLE_5: "claude-fable-5",
492
+ OPUS_5_5: "claude-opus-5-5",
493
+ OPUS_5: "claude-opus-5",
494
+ OPUS_4_8: "claude-opus-4-8",
495
+ OPUS_4_7: "claude-opus-4-7",
496
+ OPUS_4_6: "claude-opus-4-6",
497
+ SONNET_5_5: "claude-sonnet-5-5",
498
+ SONNET_5: "claude-sonnet-5",
499
+ SONNET_4_6: "claude-sonnet-4-6",
500
+ HAIKU_4_5: "claude-haiku-4-5"
501
+ },
502
+ OpenAI: {
503
+ GPT_6_ASTRA: "gpt-6-astra",
504
+ GPT_6_1_SOL: "gpt-6.1-sol",
505
+ GPT_6_SOL: "gpt-6-sol",
506
+ GPT_6_LUNA: "gpt-6-luna",
507
+ GPT_5_6_SOL: "gpt-5.6-sol",
508
+ GPT_5_6_TERRA: "gpt-5.6-terra",
509
+ GPT_5_6_LUNA: "gpt-5.6-luna",
510
+ GPT_5_5: "gpt-5.5",
511
+ GPT_5_4: "gpt-5.4",
512
+ GPT_5_4_PRO: "gpt-5.4-pro",
513
+ GPT_5_4_MINI: "gpt-5.4-mini",
514
+ GPT_5_4_NANO: "gpt-5.4-nano",
515
+ GPT_5_2: "gpt-5.2",
516
+ GPT_5_MINI: "gpt-5-mini"
517
+ },
518
+ Google: {
519
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
520
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
521
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
522
+ GEMINI_3_5_FLASH: "gemini-3.5-flash",
523
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
524
+ GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
525
+ GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
526
+ GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
527
+ },
528
+ /**
529
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
530
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
531
+ * recognizes them. All listed models accept image input.
532
+ */
533
+ OpenRouter: {
534
+ MUSE_SPARK_1_3: "meta/muse-spark-1.3",
535
+ GROK_4_7: "x-ai/grok-4.7",
536
+ GROK_4_6: "x-ai/grok-4.6",
537
+ GROK_4_5: "x-ai/grok-4.5",
538
+ KIMI_K3: "moonshotai/kimi-k3",
539
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
540
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
541
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
542
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
543
+ GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
544
+ MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
545
+ }
546
+ };
547
+ var DEFAULT_MODELS = {
548
+ [Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
549
+ [Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
550
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
551
+ [Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
552
+ };
553
+ var DEFAULT_MAX_TOKENS = 4096;
554
+ var OPENAI_REASONING_MAX_TOKENS = 16384;
555
+ var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
556
+ var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
557
+ Model.OpenRouter.QWEN_3_8_MAX,
558
+ Model.OpenRouter.QWEN_3_7_PLUS
559
+ ]);
560
+ var MODEL_TO_PROVIDER = new Map([
561
+ ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
562
+ ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
563
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
564
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
565
+ ]);
566
+ var VALID_PROVIDERS = Object.values(Provider);
567
+ var PROVIDER_DEFAULT_REASONING = {
568
+ openai: "medium",
569
+ anthropic: "off",
570
+ google: "off",
571
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
572
+ openrouter: "off"
573
+ };
574
+ var Content = {
575
+ /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
576
+ PLACEHOLDER_TEXT: "placeholder-text",
577
+ /** Detects error messages, banners, stack traces, or error codes */
578
+ ERROR_MESSAGES: "error-messages",
579
+ /** Detects broken image icons or failed-to-load image indicators */
580
+ BROKEN_IMAGES: "broken-images",
581
+ /** Detects UI elements that unintentionally overlap and obscure content */
582
+ OVERLAPPING_ELEMENTS: "overlapping-elements"
583
+ };
584
+ var Layout = {
585
+ /** Detects elements that unintentionally overlap each other */
586
+ OVERLAP: "overlap",
587
+ /** Detects content cut off or extending beyond container boundaries */
588
+ OVERFLOW: "overflow",
589
+ /** Detects inconsistent alignment of text, images, and UI components */
590
+ ALIGNMENT: "alignment"
591
+ };
592
+ var Accessibility = {
593
+ /** Detects insufficient color contrast between text and backgrounds */
594
+ CONTRAST: "contrast",
595
+ /** Detects text that is cut off, overlapping, too small, or obscured */
596
+ READABILITY: "readability",
597
+ /** Detects interactive elements that are not visually distinct */
598
+ INTERACTIVE_VISIBILITY: "interactive-visibility",
599
+ /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
600
+ COLOR_BLINDNESS: "color-blindness",
601
+ /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
602
+ COLOR_ALONE: "color-alone"
603
+ };
604
+
596
605
  // src/templates/accessibility.ts
597
606
  var ALL_CHECKS = Object.values(Accessibility);
598
607
  var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
@@ -702,17 +711,22 @@ function parseRetryAfter(value) {
702
711
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
703
712
  Model.Anthropic.FABLE_5_1,
704
713
  Model.Anthropic.FABLE_5,
714
+ Model.Anthropic.OPUS_5_5,
705
715
  Model.Anthropic.OPUS_5,
706
716
  Model.Anthropic.OPUS_4_8,
707
717
  Model.Anthropic.OPUS_4_7,
718
+ Model.Anthropic.SONNET_5_5,
708
719
  Model.Anthropic.SONNET_5
709
720
  ]);
710
721
  function mapEffort(level, model) {
722
+ if (level === "minimal") return "low";
711
723
  if (level !== "xhigh") return level;
712
724
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
713
725
  }
714
726
  var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
715
727
  var EFFORT_TO_BUDGET_TOKENS = {
728
+ // 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
729
+ minimal: 1024,
716
730
  low: 1024,
717
731
  medium: 4096,
718
732
  high: 8192,
@@ -829,6 +843,9 @@ function sleep(ms) {
829
843
  return new Promise((resolve2) => setTimeout(resolve2, ms));
830
844
  }
831
845
  var GOOGLE_THINKING_LEVEL = {
846
+ // Gemini does define a "minimal" thinking level, but some models reject it
847
+ // (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
848
+ minimal: "low",
832
849
  low: "low",
833
850
  medium: "medium",
834
851
  high: "high",
@@ -1146,6 +1163,9 @@ var OpenAIDriver = class {
1146
1163
  // src/providers/openrouter.ts
1147
1164
  var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
1148
1165
  var OPENROUTER_REASONING_EFFORT = {
1166
+ // OpenRouter normalizes upstream vendors to low/medium/high only, so
1167
+ // "minimal" has no native equivalent and clamps to the floor.
1168
+ minimal: "low",
1149
1169
  low: "low",
1150
1170
  medium: "medium",
1151
1171
  high: "high",
@@ -1305,10 +1325,20 @@ function parseBooleanEnv(envName, value) {
1305
1325
  `Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
1306
1326
  );
1307
1327
  }
1328
+ function parseReasoningEffortEnv(envName, value) {
1329
+ if (value === void 0 || value === "") return void 0;
1330
+ const levels = Object.values(ReasoningEffort);
1331
+ const lower = value.toLowerCase();
1332
+ if (levels.includes(lower)) return lower;
1333
+ throw new VisualAIConfigError(
1334
+ `Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
1335
+ );
1336
+ }
1308
1337
  var debugDeprecationWarned = false;
1309
1338
  function resolveConfig(config) {
1310
1339
  const provider = resolveProvider(config);
1311
1340
  const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
1341
+ const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
1312
1342
  const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
1313
1343
  const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
1314
1344
  const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
@@ -1326,12 +1356,12 @@ function resolveConfig(config) {
1326
1356
  }
1327
1357
  const userSetMaxTokens = config.maxTokens !== void 0;
1328
1358
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
1329
- const effortNeedsLargeBudget = config.reasoningEffort === "high" || config.reasoningEffort === "xhigh";
1359
+ const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
1330
1360
  const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
1331
1361
  if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
1332
1362
  maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
1333
1363
  if (debug) {
1334
- const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${config.reasoningEffort}"`;
1364
+ const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
1335
1365
  process.stderr.write(
1336
1366
  `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
1337
1367
  `
@@ -1343,7 +1373,7 @@ function resolveConfig(config) {
1343
1373
  apiKey: config.apiKey,
1344
1374
  model,
1345
1375
  maxTokens,
1346
- reasoningEffort: config.reasoningEffort,
1376
+ reasoningEffort,
1347
1377
  maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1348
1378
  imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1349
1379
  timeout: config.timeout,
@@ -1366,6 +1396,10 @@ var PRICING_TABLE = {
1366
1396
  inputPricePerToken: 10 / PER_MILLION,
1367
1397
  outputPricePerToken: 50 / PER_MILLION
1368
1398
  },
1399
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
1400
+ inputPricePerToken: 4 / PER_MILLION,
1401
+ outputPricePerToken: 20 / PER_MILLION
1402
+ },
1369
1403
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1370
1404
  inputPricePerToken: 5 / PER_MILLION,
1371
1405
  outputPricePerToken: 25 / PER_MILLION
@@ -1374,6 +1408,10 @@ var PRICING_TABLE = {
1374
1408
  inputPricePerToken: 5 / PER_MILLION,
1375
1409
  outputPricePerToken: 25 / PER_MILLION
1376
1410
  },
1411
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
1412
+ inputPricePerToken: 2 / PER_MILLION,
1413
+ outputPricePerToken: 10 / PER_MILLION
1414
+ },
1377
1415
  [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
1378
1416
  inputPricePerToken: 3 / PER_MILLION,
1379
1417
  outputPricePerToken: 15 / PER_MILLION
@@ -1400,6 +1438,22 @@ var PRICING_TABLE = {
1400
1438
  inputPricePerToken: 10 / PER_MILLION,
1401
1439
  outputPricePerToken: 50 / PER_MILLION
1402
1440
  },
1441
+ // Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
1442
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
1443
+ inputPricePerToken: 2 / PER_MILLION,
1444
+ outputPricePerToken: 10 / PER_MILLION
1445
+ },
1446
+ // Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
1447
+ // bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
1448
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
1449
+ inputPricePerToken: 2 / PER_MILLION,
1450
+ outputPricePerToken: 10 / PER_MILLION
1451
+ },
1452
+ // Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
1453
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
1454
+ inputPricePerToken: 0.1 / PER_MILLION,
1455
+ outputPricePerToken: 0.5 / PER_MILLION
1456
+ },
1403
1457
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
1404
1458
  inputPricePerToken: 5 / PER_MILLION,
1405
1459
  outputPricePerToken: 30 / PER_MILLION
@@ -1496,6 +1550,10 @@ var PRICING_TABLE = {
1496
1550
  inputPricePerToken: 0.1 / PER_MILLION,
1497
1551
  outputPricePerToken: 0.2 / PER_MILLION
1498
1552
  },
1553
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
1554
+ inputPricePerToken: 1.6 / PER_MILLION,
1555
+ outputPricePerToken: 4.8 / PER_MILLION
1556
+ },
1499
1557
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1500
1558
  inputPricePerToken: 2 / PER_MILLION,
1501
1559
  outputPricePerToken: 6 / PER_MILLION
@@ -1529,6 +1587,13 @@ var PRICING_TABLE = {
1529
1587
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
1530
1588
  inputPricePerToken: 0.15 / PER_MILLION,
1531
1589
  outputPricePerToken: 0.5 / PER_MILLION
1590
+ },
1591
+ // Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
1592
+ // upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
1593
+ // is $0.0036/MTok, not modelled (no provider gets a cache discount here).
1594
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
1595
+ inputPricePerToken: 0.435 / PER_MILLION,
1596
+ outputPricePerToken: 0.87 / PER_MILLION
1532
1597
  }
1533
1598
  };
1534
1599
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -2550,6 +2615,12 @@ function stripCodeFences(text) {
2550
2615
  }
2551
2616
  var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
2552
2617
  var AskResponseSchema = AskResultSchema.omit({ usage: true });
2618
+ var AskImageResponseSchema = AskResponseSchema.omit({
2619
+ frameReferences: true,
2620
+ timestampReferences: true
2621
+ });
2622
+ var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
2623
+ var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
2553
2624
  var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
2554
2625
  var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
2555
2626
  function parseJson(text) {
@@ -2642,7 +2713,11 @@ function createDriver(provider, config) {
2642
2713
  return PROVIDER_REGISTRY[provider](config);
2643
2714
  }
2644
2715
  var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
2645
- var askSchemaOptions = toSchemaOptions(AskResponseSchema);
2716
+ var askSchemaOptionsByMedia = {
2717
+ image: toSchemaOptions(AskImageResponseSchema),
2718
+ video: toSchemaOptions(AskFramesResponseSchema),
2719
+ "native-video": toSchemaOptions(AskNativeVideoResponseSchema)
2720
+ };
2646
2721
  var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2647
2722
  function mediaToProviderInputs(media) {
2648
2723
  if (media.kind === "image") {
@@ -2787,7 +2862,12 @@ function visualAI(config = {}) {
2787
2862
  media: dispatch.mediaContext
2788
2863
  });
2789
2864
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2790
- const { response, metadata } = await sendMedia(driver, dispatch, prompt, askSchemaOptions);
2865
+ const { response, metadata } = await sendMedia(
2866
+ driver,
2867
+ dispatch,
2868
+ prompt,
2869
+ askSchemaOptionsByMedia[dispatch.mediaContext.kind]
2870
+ );
2791
2871
  debugLog(resolvedConfig, "ask response", response.text, "response");
2792
2872
  const result = parseAskResponse(response.text);
2793
2873
  return {
@@ -2810,7 +2890,7 @@ function visualAI(config = {}) {
2810
2890
  debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
2811
2891
  const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
2812
2892
  debugLog(resolvedConfig, "compare response", response.text, "response");
2813
- const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model === Model.Google.GEMINI_3_FLASH_PREVIEW;
2893
+ const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
2814
2894
  const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
2815
2895
  let diffImage;
2816
2896
  if (effectiveDiffImage) {