visual-ai-assertions 0.25.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,132 +1,3 @@
1
- // src/constants.ts
2
- var ReasoningEffort = {
3
- LOW: "low",
4
- MEDIUM: "medium",
5
- HIGH: "high",
6
- XHIGH: "xhigh"
7
- };
8
- var ImageDetail = {
9
- AUTO: "auto",
10
- LOW: "low",
11
- HIGH: "high"
12
- };
13
- var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
14
- var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
15
- var Provider = {
16
- ANTHROPIC: "anthropic",
17
- OPENAI: "openai",
18
- GOOGLE: "google",
19
- OPENROUTER: "openrouter"
20
- };
21
- var Model = {
22
- Anthropic: {
23
- FABLE_5_1: "claude-fable-5-1",
24
- FABLE_5: "claude-fable-5",
25
- OPUS_5: "claude-opus-5",
26
- OPUS_4_8: "claude-opus-4-8",
27
- OPUS_4_7: "claude-opus-4-7",
28
- OPUS_4_6: "claude-opus-4-6",
29
- SONNET_5: "claude-sonnet-5",
30
- SONNET_4_6: "claude-sonnet-4-6",
31
- HAIKU_4_5: "claude-haiku-4-5"
32
- },
33
- OpenAI: {
34
- GPT_6_ASTRA: "gpt-6-astra",
35
- GPT_5_6_SOL: "gpt-5.6-sol",
36
- GPT_5_6_TERRA: "gpt-5.6-terra",
37
- GPT_5_6_LUNA: "gpt-5.6-luna",
38
- GPT_5_5: "gpt-5.5",
39
- GPT_5_4: "gpt-5.4",
40
- GPT_5_4_PRO: "gpt-5.4-pro",
41
- GPT_5_4_MINI: "gpt-5.4-mini",
42
- GPT_5_4_NANO: "gpt-5.4-nano",
43
- GPT_5_2: "gpt-5.2",
44
- GPT_5_MINI: "gpt-5-mini"
45
- },
46
- Google: {
47
- GEMINI_3_8_FLASH: "gemini-3.8-flash",
48
- GEMINI_3_7_FLASH: "gemini-3.7-flash",
49
- GEMINI_3_6_FLASH: "gemini-3.6-flash",
50
- GEMINI_3_5_FLASH: "gemini-3.5-flash",
51
- GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
52
- GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
53
- GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
54
- GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
55
- },
56
- /**
57
- * Models routed through OpenRouter (https://openrouter.ai). Slugs always
58
- * carry a vendor prefix (`vendor/model`), which is how provider inference
59
- * recognizes them. All listed models accept image input.
60
- */
61
- OpenRouter: {
62
- MUSE_SPARK_1_3: "meta/muse-spark-1.3",
63
- GROK_4_6: "x-ai/grok-4.6",
64
- GROK_4_5: "x-ai/grok-4.5",
65
- KIMI_K3: "moonshotai/kimi-k3",
66
- KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
67
- QWEN_3_8_MAX: "qwen/qwen3.8-max",
68
- QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
69
- QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
70
- GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
71
- }
72
- };
73
- var DEFAULT_MODELS = {
74
- [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
75
- [Provider.OPENAI]: Model.OpenAI.GPT_5_6_LUNA,
76
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
77
- [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
78
- };
79
- var DEFAULT_MAX_TOKENS = 4096;
80
- var OPENAI_REASONING_MAX_TOKENS = 16384;
81
- var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
82
- var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
83
- Model.OpenAI.GPT_6_ASTRA
84
- ]);
85
- var MODEL_TO_PROVIDER = new Map([
86
- ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
87
- ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
88
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
89
- ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
90
- ]);
91
- var VALID_PROVIDERS = Object.values(Provider);
92
- var PROVIDER_DEFAULT_REASONING = {
93
- openai: "medium",
94
- anthropic: "off",
95
- google: "off",
96
- // Varies by upstream model; the driver sends no reasoning field unless configured.
97
- openrouter: "off"
98
- };
99
- var Content = {
100
- /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
101
- PLACEHOLDER_TEXT: "placeholder-text",
102
- /** Detects error messages, banners, stack traces, or error codes */
103
- ERROR_MESSAGES: "error-messages",
104
- /** Detects broken image icons or failed-to-load image indicators */
105
- BROKEN_IMAGES: "broken-images",
106
- /** Detects UI elements that unintentionally overlap and obscure content */
107
- OVERLAPPING_ELEMENTS: "overlapping-elements"
108
- };
109
- var Layout = {
110
- /** Detects elements that unintentionally overlap each other */
111
- OVERLAP: "overlap",
112
- /** Detects content cut off or extending beyond container boundaries */
113
- OVERFLOW: "overflow",
114
- /** Detects inconsistent alignment of text, images, and UI components */
115
- ALIGNMENT: "alignment"
116
- };
117
- var Accessibility = {
118
- /** Detects insufficient color contrast between text and backgrounds */
119
- CONTRAST: "contrast",
120
- /** Detects text that is cut off, overlapping, too small, or obscured */
121
- READABILITY: "readability",
122
- /** Detects interactive elements that are not visually distinct */
123
- INTERACTIVE_VISIBILITY: "interactive-visibility",
124
- /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
125
- COLOR_BLINDNESS: "color-blindness",
126
- /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
127
- COLOR_ALONE: "color-alone"
128
- };
129
-
130
1
  // src/errors.ts
131
2
  var VisualAIError = class extends Error {
132
3
  code;
@@ -493,7 +364,7 @@ function visibleRole(finalState, requireCorrectRendering) {
493
364
  var ELEMENTS_VISIBLE_CLIPPING_RULES = [
494
365
  "When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
495
366
  "An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
496
- "An element you cannot see at all is not visible, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
367
+ "The scrolling allowance above applies only to elements that are at least partly rendered. If no part of an element is on screen, the check for that element FAILS: do not infer that it exists below the fold. Judge only what this screenshot actually shows."
497
368
  ];
498
369
  var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
499
370
  var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
@@ -524,6 +395,144 @@ function buildElementsVisibilityPrompt(elements, visible, options) {
524
395
  return buildCheckPrompt(statements, { role, instructions });
525
396
  }
526
397
 
398
+ // src/constants.ts
399
+ var ReasoningEffort = {
400
+ MINIMAL: "minimal",
401
+ LOW: "low",
402
+ MEDIUM: "medium",
403
+ HIGH: "high",
404
+ XHIGH: "xhigh"
405
+ };
406
+ var ImageDetail = {
407
+ AUTO: "auto",
408
+ LOW: "low",
409
+ HIGH: "high"
410
+ };
411
+ var DEFAULT_IMAGE_DETAIL = ImageDetail.AUTO;
412
+ var DEFAULT_MAX_IMAGE_DIMENSION = 1568;
413
+ var Provider = {
414
+ ANTHROPIC: "anthropic",
415
+ OPENAI: "openai",
416
+ GOOGLE: "google",
417
+ OPENROUTER: "openrouter"
418
+ };
419
+ var Model = {
420
+ Anthropic: {
421
+ FABLE_5_1: "claude-fable-5-1",
422
+ FABLE_5: "claude-fable-5",
423
+ OPUS_5_5: "claude-opus-5-5",
424
+ OPUS_5: "claude-opus-5",
425
+ OPUS_4_8: "claude-opus-4-8",
426
+ OPUS_4_7: "claude-opus-4-7",
427
+ OPUS_4_6: "claude-opus-4-6",
428
+ SONNET_5_5: "claude-sonnet-5-5",
429
+ SONNET_5: "claude-sonnet-5",
430
+ SONNET_4_6: "claude-sonnet-4-6",
431
+ HAIKU_4_5: "claude-haiku-4-5"
432
+ },
433
+ OpenAI: {
434
+ GPT_6_ASTRA: "gpt-6-astra",
435
+ GPT_6_1_SOL: "gpt-6.1-sol",
436
+ GPT_6_SOL: "gpt-6-sol",
437
+ GPT_6_LUNA: "gpt-6-luna",
438
+ GPT_5_6_SOL: "gpt-5.6-sol",
439
+ GPT_5_6_TERRA: "gpt-5.6-terra",
440
+ GPT_5_6_LUNA: "gpt-5.6-luna",
441
+ GPT_5_5: "gpt-5.5",
442
+ GPT_5_4: "gpt-5.4",
443
+ GPT_5_4_PRO: "gpt-5.4-pro",
444
+ GPT_5_4_MINI: "gpt-5.4-mini",
445
+ GPT_5_4_NANO: "gpt-5.4-nano",
446
+ GPT_5_2: "gpt-5.2",
447
+ GPT_5_MINI: "gpt-5-mini"
448
+ },
449
+ Google: {
450
+ GEMINI_3_8_FLASH: "gemini-3.8-flash",
451
+ GEMINI_3_7_FLASH: "gemini-3.7-flash",
452
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
453
+ GEMINI_3_5_FLASH: "gemini-3.5-flash",
454
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
455
+ GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
456
+ GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
457
+ GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
458
+ },
459
+ /**
460
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
461
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
462
+ * recognizes them. All listed models accept image input.
463
+ */
464
+ OpenRouter: {
465
+ MUSE_SPARK_1_3: "meta/muse-spark-1.3",
466
+ GROK_4_7: "x-ai/grok-4.7",
467
+ GROK_4_6: "x-ai/grok-4.6",
468
+ GROK_4_5: "x-ai/grok-4.5",
469
+ KIMI_K3: "moonshotai/kimi-k3",
470
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
471
+ QWEN_3_8_MAX: "qwen/qwen3.8-max",
472
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
473
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
474
+ GLM_5_3_FLASH: "z-ai/glm-5.3-flash",
475
+ MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro"
476
+ }
477
+ };
478
+ var DEFAULT_MODELS = {
479
+ [Provider.ANTHROPIC]: Model.Anthropic.SONNET_5_5,
480
+ [Provider.OPENAI]: Model.OpenAI.GPT_6_1_SOL,
481
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_8_FLASH,
482
+ [Provider.OPENROUTER]: Model.OpenRouter.MUSE_SPARK_1_3
483
+ };
484
+ var DEFAULT_MAX_TOKENS = 4096;
485
+ var OPENAI_REASONING_MAX_TOKENS = 16384;
486
+ var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
487
+ var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
488
+ Model.OpenRouter.QWEN_3_8_MAX,
489
+ Model.OpenRouter.QWEN_3_7_PLUS
490
+ ]);
491
+ var MODEL_TO_PROVIDER = new Map([
492
+ ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
493
+ ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
494
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
495
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
496
+ ]);
497
+ var VALID_PROVIDERS = Object.values(Provider);
498
+ var PROVIDER_DEFAULT_REASONING = {
499
+ openai: "medium",
500
+ anthropic: "off",
501
+ google: "off",
502
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
503
+ openrouter: "off"
504
+ };
505
+ var Content = {
506
+ /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
507
+ PLACEHOLDER_TEXT: "placeholder-text",
508
+ /** Detects error messages, banners, stack traces, or error codes */
509
+ ERROR_MESSAGES: "error-messages",
510
+ /** Detects broken image icons or failed-to-load image indicators */
511
+ BROKEN_IMAGES: "broken-images",
512
+ /** Detects UI elements that unintentionally overlap and obscure content */
513
+ OVERLAPPING_ELEMENTS: "overlapping-elements"
514
+ };
515
+ var Layout = {
516
+ /** Detects elements that unintentionally overlap each other */
517
+ OVERLAP: "overlap",
518
+ /** Detects content cut off or extending beyond container boundaries */
519
+ OVERFLOW: "overflow",
520
+ /** Detects inconsistent alignment of text, images, and UI components */
521
+ ALIGNMENT: "alignment"
522
+ };
523
+ var Accessibility = {
524
+ /** Detects insufficient color contrast between text and backgrounds */
525
+ CONTRAST: "contrast",
526
+ /** Detects text that is cut off, overlapping, too small, or obscured */
527
+ READABILITY: "readability",
528
+ /** Detects interactive elements that are not visually distinct */
529
+ INTERACTIVE_VISIBILITY: "interactive-visibility",
530
+ /** Detects color choices likely to be indistinguishable to viewers with common color vision deficiencies */
531
+ COLOR_BLINDNESS: "color-blindness",
532
+ /** Detects information conveyed by color alone, without a non-color cue (icon, text, pattern, position) */
533
+ COLOR_ALONE: "color-alone"
534
+ };
535
+
527
536
  // src/templates/accessibility.ts
528
537
  var ALL_CHECKS = Object.values(Accessibility);
529
538
  var ACCESSIBILITY_ROLE = "Evaluate this screenshot for visual accessibility. Focus on what you can actually perceive \u2014 apparent contrast levels, text legibility, and visual distinctiveness of interactive elements.";
@@ -633,17 +642,22 @@ function parseRetryAfter(value) {
633
642
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
634
643
  Model.Anthropic.FABLE_5_1,
635
644
  Model.Anthropic.FABLE_5,
645
+ Model.Anthropic.OPUS_5_5,
636
646
  Model.Anthropic.OPUS_5,
637
647
  Model.Anthropic.OPUS_4_8,
638
648
  Model.Anthropic.OPUS_4_7,
649
+ Model.Anthropic.SONNET_5_5,
639
650
  Model.Anthropic.SONNET_5
640
651
  ]);
641
652
  function mapEffort(level, model) {
653
+ if (level === "minimal") return "low";
642
654
  if (level !== "xhigh") return level;
643
655
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
644
656
  }
645
657
  var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
646
658
  var EFFORT_TO_BUDGET_TOKENS = {
659
+ // 1024 is Anthropic's minimum thinking budget, so minimal and low coincide.
660
+ minimal: 1024,
647
661
  low: 1024,
648
662
  medium: 4096,
649
663
  high: 8192,
@@ -760,6 +774,9 @@ function sleep(ms) {
760
774
  return new Promise((resolve2) => setTimeout(resolve2, ms));
761
775
  }
762
776
  var GOOGLE_THINKING_LEVEL = {
777
+ // Gemini does define a "minimal" thinking level, but some models reject it
778
+ // (e.g. Gemini 3.1 Pro), so "minimal" clamps to "low" here as well.
779
+ minimal: "low",
763
780
  low: "low",
764
781
  medium: "medium",
765
782
  high: "high",
@@ -1077,6 +1094,9 @@ var OpenAIDriver = class {
1077
1094
  // src/providers/openrouter.ts
1078
1095
  var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
1079
1096
  var OPENROUTER_REASONING_EFFORT = {
1097
+ // OpenRouter normalizes upstream vendors to low/medium/high only, so
1098
+ // "minimal" has no native equivalent and clamps to the floor.
1099
+ minimal: "low",
1080
1100
  low: "low",
1081
1101
  medium: "medium",
1082
1102
  high: "high",
@@ -1236,10 +1256,20 @@ function parseBooleanEnv(envName, value) {
1236
1256
  `Invalid ${envName} value: "${value}". Use "true", "1", "false", or "0".`
1237
1257
  );
1238
1258
  }
1259
+ function parseReasoningEffortEnv(envName, value) {
1260
+ if (value === void 0 || value === "") return void 0;
1261
+ const levels = Object.values(ReasoningEffort);
1262
+ const lower = value.toLowerCase();
1263
+ if (levels.includes(lower)) return lower;
1264
+ throw new VisualAIConfigError(
1265
+ `Invalid ${envName} value: "${value}". Use one of: ${levels.join(", ")}.`
1266
+ );
1267
+ }
1239
1268
  var debugDeprecationWarned = false;
1240
1269
  function resolveConfig(config) {
1241
1270
  const provider = resolveProvider(config);
1242
1271
  const model = config.model ?? process.env.VISUAL_AI_MODEL ?? DEFAULT_MODELS[provider];
1272
+ const reasoningEffort = config.reasoningEffort ?? parseReasoningEffortEnv("VISUAL_AI_REASONING_EFFORT", process.env.VISUAL_AI_REASONING_EFFORT);
1243
1273
  const debug = config.debug ?? parseBooleanEnv("VISUAL_AI_DEBUG", process.env.VISUAL_AI_DEBUG) ?? false;
1244
1274
  const debugPrompt = config.debugPrompt ?? parseBooleanEnv("VISUAL_AI_DEBUG_PROMPT", process.env.VISUAL_AI_DEBUG_PROMPT) ?? false;
1245
1275
  const debugResponse = config.debugResponse ?? parseBooleanEnv("VISUAL_AI_DEBUG_RESPONSE", process.env.VISUAL_AI_DEBUG_RESPONSE) ?? false;
@@ -1257,12 +1287,12 @@ function resolveConfig(config) {
1257
1287
  }
1258
1288
  const userSetMaxTokens = config.maxTokens !== void 0;
1259
1289
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
1260
- const effortNeedsLargeBudget = config.reasoningEffort === "high" || config.reasoningEffort === "xhigh";
1290
+ const effortNeedsLargeBudget = reasoningEffort === "high" || reasoningEffort === "xhigh";
1261
1291
  const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
1262
1292
  if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
1263
1293
  maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
1264
1294
  if (debug) {
1265
- const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${config.reasoningEffort}"`;
1295
+ const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${reasoningEffort}"`;
1266
1296
  process.stderr.write(
1267
1297
  `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
1268
1298
  `
@@ -1274,7 +1304,7 @@ function resolveConfig(config) {
1274
1304
  apiKey: config.apiKey,
1275
1305
  model,
1276
1306
  maxTokens,
1277
- reasoningEffort: config.reasoningEffort,
1307
+ reasoningEffort,
1278
1308
  maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1279
1309
  imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1280
1310
  timeout: config.timeout,
@@ -1297,6 +1327,10 @@ var PRICING_TABLE = {
1297
1327
  inputPricePerToken: 10 / PER_MILLION,
1298
1328
  outputPricePerToken: 50 / PER_MILLION
1299
1329
  },
1330
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5_5}`]: {
1331
+ inputPricePerToken: 4 / PER_MILLION,
1332
+ outputPricePerToken: 20 / PER_MILLION
1333
+ },
1300
1334
  [`${Provider.ANTHROPIC}:${Model.Anthropic.OPUS_5}`]: {
1301
1335
  inputPricePerToken: 5 / PER_MILLION,
1302
1336
  outputPricePerToken: 25 / PER_MILLION
@@ -1305,6 +1339,10 @@ var PRICING_TABLE = {
1305
1339
  inputPricePerToken: 5 / PER_MILLION,
1306
1340
  outputPricePerToken: 25 / PER_MILLION
1307
1341
  },
1342
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5_5}`]: {
1343
+ inputPricePerToken: 2 / PER_MILLION,
1344
+ outputPricePerToken: 10 / PER_MILLION
1345
+ },
1308
1346
  [`${Provider.ANTHROPIC}:${Model.Anthropic.SONNET_5}`]: {
1309
1347
  inputPricePerToken: 3 / PER_MILLION,
1310
1348
  outputPricePerToken: 15 / PER_MILLION
@@ -1331,6 +1369,22 @@ var PRICING_TABLE = {
1331
1369
  inputPricePerToken: 10 / PER_MILLION,
1332
1370
  outputPricePerToken: 50 / PER_MILLION
1333
1371
  },
1372
+ // Cached input is $0.10/MTok (not modelled), half GPT-6 Sol's cached rate.
1373
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_1_SOL}`]: {
1374
+ inputPricePerToken: 2 / PER_MILLION,
1375
+ outputPricePerToken: 10 / PER_MILLION
1376
+ },
1377
+ // Cached input is $0.20/MTok (not modelled). Prompts above 272K input tokens
1378
+ // bill at 2x input / 1.5x output, which is far beyond screenshot-sized calls.
1379
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_SOL}`]: {
1380
+ inputPricePerToken: 2 / PER_MILLION,
1381
+ outputPricePerToken: 10 / PER_MILLION
1382
+ },
1383
+ // Cached input is $0.01/MTok and cache writes $0.125/MTok; neither is modelled.
1384
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_LUNA}`]: {
1385
+ inputPricePerToken: 0.1 / PER_MILLION,
1386
+ outputPricePerToken: 0.5 / PER_MILLION
1387
+ },
1334
1388
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
1335
1389
  inputPricePerToken: 5 / PER_MILLION,
1336
1390
  outputPricePerToken: 30 / PER_MILLION
@@ -1427,6 +1481,10 @@ var PRICING_TABLE = {
1427
1481
  inputPricePerToken: 0.1 / PER_MILLION,
1428
1482
  outputPricePerToken: 0.2 / PER_MILLION
1429
1483
  },
1484
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_7}`]: {
1485
+ inputPricePerToken: 1.6 / PER_MILLION,
1486
+ outputPricePerToken: 4.8 / PER_MILLION
1487
+ },
1430
1488
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_6}`]: {
1431
1489
  inputPricePerToken: 2 / PER_MILLION,
1432
1490
  outputPricePerToken: 6 / PER_MILLION
@@ -1460,6 +1518,13 @@ var PRICING_TABLE = {
1460
1518
  [`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
1461
1519
  inputPricePerToken: 0.15 / PER_MILLION,
1462
1520
  outputPricePerToken: 0.5 / PER_MILLION
1521
+ },
1522
+ // Verified 2026-09-23 against https://openrouter.ai/api/v1/models; both
1523
+ // upstream endpoints (Xiaomi, DeepInfra) charge the same rate. Cached input
1524
+ // is $0.0036/MTok, not modelled (no provider gets a cache discount here).
1525
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.MIMO_V2_6_PRO}`]: {
1526
+ inputPricePerToken: 0.435 / PER_MILLION,
1527
+ outputPricePerToken: 0.87 / PER_MILLION
1463
1528
  }
1464
1529
  };
1465
1530
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -2481,6 +2546,12 @@ function stripCodeFences(text) {
2481
2546
  }
2482
2547
  var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
2483
2548
  var AskResponseSchema = AskResultSchema.omit({ usage: true });
2549
+ var AskImageResponseSchema = AskResponseSchema.omit({
2550
+ frameReferences: true,
2551
+ timestampReferences: true
2552
+ });
2553
+ var AskFramesResponseSchema = AskResponseSchema.omit({ timestampReferences: true });
2554
+ var AskNativeVideoResponseSchema = AskResponseSchema.omit({ frameReferences: true });
2484
2555
  var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
2485
2556
  var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
2486
2557
  function parseJson(text) {
@@ -2573,7 +2644,11 @@ function createDriver(provider, config) {
2573
2644
  return PROVIDER_REGISTRY[provider](config);
2574
2645
  }
2575
2646
  var checkSchemaOptions = toSchemaOptions(CheckResponseSchema);
2576
- var askSchemaOptions = toSchemaOptions(AskResponseSchema);
2647
+ var askSchemaOptionsByMedia = {
2648
+ image: toSchemaOptions(AskImageResponseSchema),
2649
+ video: toSchemaOptions(AskFramesResponseSchema),
2650
+ "native-video": toSchemaOptions(AskNativeVideoResponseSchema)
2651
+ };
2577
2652
  var compareSchemaOptions = toSchemaOptions(CompareResponseSchema);
2578
2653
  function mediaToProviderInputs(media) {
2579
2654
  if (media.kind === "image") {
@@ -2718,7 +2793,12 @@ function visualAI(config = {}) {
2718
2793
  media: dispatch.mediaContext
2719
2794
  });
2720
2795
  debugLog(resolvedConfig, "ask prompt", prompt, "prompt");
2721
- const { response, metadata } = await sendMedia(driver, dispatch, prompt, askSchemaOptions);
2796
+ const { response, metadata } = await sendMedia(
2797
+ driver,
2798
+ dispatch,
2799
+ prompt,
2800
+ askSchemaOptionsByMedia[dispatch.mediaContext.kind]
2801
+ );
2722
2802
  debugLog(resolvedConfig, "ask response", response.text, "response");
2723
2803
  const result = parseAskResponse(response.text);
2724
2804
  return {
@@ -2741,7 +2821,7 @@ function visualAI(config = {}) {
2741
2821
  debugLog(resolvedConfig, "compare prompt", prompt, "prompt");
2742
2822
  const response = await timedSendMessage(driver, [imgA, imgB], prompt, compareSchemaOptions);
2743
2823
  debugLog(resolvedConfig, "compare response", response.text, "response");
2744
- const supportsAnnotatedDiff = resolvedConfig.provider === "google" && resolvedConfig.model === Model.Google.GEMINI_3_FLASH_PREVIEW;
2824
+ const supportsAnnotatedDiff = resolvedConfig.provider === "google" && DIFF_ALLOWED_MODELS.has(resolvedConfig.model);
2745
2825
  const effectiveDiffImage = options?.diffImage ?? (supportsAnnotatedDiff ? true : false);
2746
2826
  let diffImage;
2747
2827
  if (effectiveDiffImage) {