visual-ai-assertions 0.21.0 → 0.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # visual-ai-assertions
2
2
 
3
- AI-powered visual assertions for E2E tests. Send screenshots — or short video recordings — to Claude, GPT, Gemini — or Grok, Kimi, and Qwen via OpenRouter — and get structured, typed results.
3
+ AI-powered visual assertions for E2E tests. Send screenshots — or short video recordings — to Claude, GPT, Gemini — or Grok, Kimi, Qwen, and GLM via OpenRouter — and get structured, typed results.
4
4
 
5
5
  ## Installation
6
6
 
@@ -11,7 +11,7 @@ npm install visual-ai-assertions
11
11
  # Optional: install additional provider SDKs
12
12
  npm install @anthropic-ai/sdk # for Claude
13
13
  npm install @google/genai # for Gemini
14
- # OpenRouter (Grok, Kimi, Qwen, ...) uses the OpenAI SDK — no extra install
14
+ # OpenRouter (Grok, Kimi, Qwen, GLM, ...) uses the OpenAI SDK — no extra install
15
15
 
16
16
  # Zod is a peer dependency
17
17
  npm install zod
@@ -104,6 +104,7 @@ const ai = visualAI({
104
104
  debug: true, // optional, logs prompts/responses to stderr
105
105
  maxTokens: 4096, // optional, default 4096
106
106
  reasoningEffort: "high", // optional, "low" | "medium" | "high" | "xhigh"
107
+ timeout: 120_000, // optional, ms — defaults to the provider SDK's own timeout
107
108
  trackUsage: false, // optional, defaults to false — usage stats to stderr
108
109
  });
109
110
 
@@ -255,6 +256,33 @@ await ai.elementsVisible(screenshot, ["Submit button", "Nav bar", "Footer"]);
255
256
  // Check that UI elements are hidden
256
257
  await ai.elementsHidden(screenshot, ["Loading spinner", "Error modal"]);
257
258
 
259
+ // Clipping is judged the way a tester would. An element cut off at the trailing
260
+ // edge of a scrollable row or feed passes, because scrolling reaches it. One
261
+ // sliced by the screen edge or by fixed chrome such as the status bar or a
262
+ // sticky nav fails, because scrolling cannot. An element you cannot see at all
263
+ // fails: only what the screenshot shows is judged.
264
+ //
265
+ // Overlays are judged by what they say about the element's state, not by how
266
+ // much they cover. A badge, favourite icon or price pill coexists with finished
267
+ // content and leaves the element visible. A loading spinner, skeleton, progress
268
+ // bar or error overlay says it is not ready, so the check fails even though you
269
+ // can still see what is underneath. A modal, dialog or cookie banner that a user
270
+ // could not read or use past also fails.
271
+ //
272
+ // Pass finalState: false when the screenshot was deliberately captured mid-load,
273
+ // so loading chrome is expected rather than a defect. The finished-state rules
274
+ // are left out; presence, clipping and blocking overlays are still judged.
275
+ await ai.elementsVisible(screenshot, ["Product image"], { finalState: false });
276
+
277
+ // By default this is a presence check: an element counts as visible if it is
278
+ // there at all, whatever it looks like. Pass requireCorrectRendering: true to
279
+ // also fail an element that is present but clearly badly rendered — unreadable
280
+ // contrast, overlapping or misaligned elements, text cut off mid-word — with
281
+ // the model saying the element is present before naming the defect. Opt in per
282
+ // assertion: measured on the bench, it catches exactly that case and makes
283
+ // models flakier on plain presence questions.
284
+ await ai.elementsVisible(screenshot, ["Promo banner"], { requireCorrectRendering: true });
285
+
258
286
  // Accessibility checks (contrast, readability, interactive visibility, color blindness, color-alone meaning)
259
287
  await ai.accessibility(screenshot);
260
288
  await ai.accessibility(screenshot, {
@@ -463,16 +491,17 @@ The `VisualAIKnownError` union and `isVisualAIKnownError()` helper are useful wh
463
491
 
464
492
  ## Configuration
465
493
 
466
- | Option | Type | Default | Description |
467
- | ----------------- | ------- | ---------------- | ----------------------------------------------------------------------------- |
468
- | `apiKey` | string | env var | API key for the provider |
469
- | `model` | string | provider default | Model to use |
470
- | `debug` | boolean | `false` | Enable error diagnostic logging to stderr |
471
- | `debugPrompt` | boolean | `false` | Log prompts to stderr |
472
- | `debugResponse` | boolean | `false` | Log responses to stderr |
473
- | `maxTokens` | number | `4096` | Max tokens for AI response |
474
- | `reasoningEffort` | string | `undefined` | `"low"` `"medium"` `"high"` `"xhigh"` — controls how deeply the model reasons |
475
- | `trackUsage` | boolean | `false` | Log token usage and estimated cost to stderr |
494
+ | Option | Type | Default | Description |
495
+ | ----------------- | ------- | ---------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
496
+ | `apiKey` | string | env var | API key for the provider |
497
+ | `model` | string | provider default | Model to use |
498
+ | `debug` | boolean | `false` | Enable error diagnostic logging to stderr |
499
+ | `debugPrompt` | boolean | `false` | Log prompts to stderr |
500
+ | `debugResponse` | boolean | `false` | Log responses to stderr |
501
+ | `maxTokens` | number | `4096` | Max tokens for AI response |
502
+ | `reasoningEffort` | string | `undefined` | `"low"` `"medium"` `"high"` `"xhigh"` — controls how deeply the model reasons |
503
+ | `timeout` | number | SDK default | Per-request timeout in ms. Unset leaves each SDK's own default (OpenAI/OpenRouter 10 min, Google 1 min). SDKs retry timeouts, so total wall time can be a multiple of this. |
504
+ | `trackUsage` | boolean | `false` | Log token usage and estimated cost to stderr |
476
505
 
477
506
  ## Exported Types
478
507
 
@@ -535,7 +564,8 @@ All listed models support image/vision input. Pass any model ID to the `model` c
535
564
 
536
565
  | Model | Model ID | Input $/MTok | Output $/MTok | Notes |
537
566
  | ----------------- | ------------------- | ------------ | ------------- | ------------------------------------------- |
538
- | Claude Fable 5 | `claude-fable-5` | $10 | $50 | Most capable; long-horizon agentic work |
567
+ | Claude Fable 5.1 | `claude-fable-5-1` | $10 | $50 | Most capable; long-horizon agentic work |
568
+ | Claude Fable 5 | `claude-fable-5` | $10 | $50 | Predecessor to Fable 5.1, same price |
539
569
  | Claude Opus 4.8 | `claude-opus-4-8` | $5 | $25 | Most capable Opus tier; supports `xhigh` |
540
570
  | Claude Opus 4.7 | `claude-opus-4-7` | $5 | $25 | Previous Opus; supports `xhigh` effort tier |
541
571
  | Claude Opus 4.6 | `claude-opus-4-6` | $5 | $25 | Previous flagship, 128K max output |
@@ -547,6 +577,7 @@ All listed models support image/vision input. Pass any model ID to the `model` c
547
577
 
548
578
  | Model | Model ID | Input $/MTok | Output $/MTok | Notes |
549
579
  | ------------- | --------------- | ------------ | ------------- | -------------------------------------- |
580
+ | GPT-6 Astra | `gpt-6-astra` | $10 | $50 | Most capable; restricted access¹ |
550
581
  | GPT-5.6 Sol | `gpt-5.6-sol` | $5 | $30 | Newest flagship, frontier tier |
551
582
  | GPT-5.6 Terra | `gpt-5.6-terra` | $2 | $12 | Newest balanced, everyday tier |
552
583
  | GPT-5.6 Luna | `gpt-5.6-luna` | $0.20 | $1.20 | **Default** — newest, fastest/cheapest |
@@ -558,6 +589,10 @@ All listed models support image/vision input. Pass any model ID to the `model` c
558
589
  | GPT-5.4 nano | `gpt-5.4-nano` | $0.20 | $1.25 | Cheapest older-generation option |
559
590
  | GPT-5 mini | `gpt-5-mini` | $0.25 | $2 | Fast and cheap |
560
591
 
592
+ ¹ GPT-6 Astra is rolling out through OpenAI's Trusted Access Program, so many API keys cannot reach it yet — expect a `VisualAIProviderError` naming the model until your account is enabled.
593
+
594
+ Astra reasons heavily enough to spend the entire 4096-token default output budget before emitting an answer, so **it is given a 32768-token budget automatically**, at every reasoning effort rather than only at `high`/`xhigh` like other OpenAI models. That follows OpenAI's guidance to reserve at least 25,000 tokens for reasoning and output. Its output length is erratic — identical calls have used anywhere from 0 to 16384+ reasoning tokens — so a large budget reduces truncation without eliminating it; a call that exhausts the budget still bills for the tokens it burned. Passing `maxTokens` explicitly still wins. It also accepts a fifth reasoning level, `max`, above `xhigh`; this library's `reasoningEffort` stops at `xhigh`, which is passed through unchanged, so `max` is not currently reachable.
595
+
561
596
  ### Google
562
597
 
563
598
  | Model | Model ID | Input $/MTok | Output $/MTok | Notes |
@@ -587,9 +622,12 @@ Any [OpenRouter](https://openrouter.ai/models) model slug (always `vendor/model`
587
622
  | Qwen3.8 Max | `qwen/qwen3.8-max` | $2 | $6 | First Max tier with image input |
588
623
  | Qwen3.7 Plus | `qwen/qwen3.7-plus` | $0.32 | $1.28 | Cost-effective, GUI/screen-reading |
589
624
  | Qwen3.6 Flash | `qwen/qwen3.6-flash` | $0.19 | $1.13 | **Default** — cheap flash vision tier |
625
+ | GLM 5.3 Flash | `z-ai/glm-5.3-flash` | $0.15 | $0.50 | Z.ai flash tier, 1.3M context² |
590
626
 
591
627
  ¹ Muse Spark 1.3 is age-gated by OpenRouter: calls return HTTP 403 (`VisualAIAuthError`) until the account completes the 18+ confirmation at [openrouter.ai/settings/preferences](https://openrouter.ai/settings/preferences). It also reasons by default — expect several hundred reasoning tokens per call even with no `reasoningEffort` set.
592
628
 
629
+ ² GLM 5.3 Flash reasons by default — expect one to two hundred reasoning tokens per call even with no `reasoningEffort` set, billed at the output rate. OpenRouter's own context cap for it is 1,048,576 tokens (Z.ai lists 1,310,720) and its output ceiling is 131,072.
630
+
593
631
  Meta also publishes `meta/muse-spark-1.3-contributor`, the same model at $0.10 / $0.20 per MTok — about 12x cheaper — because Meta uses everything submitted through it for product improvement. It has **no named constant** (`Model.OpenRouter` does not expose it) and never appears by default anywhere in this library, so using it takes a deliberate, explicit choice: pass the slug directly as a plain string, `visualAI({ model: "meta/muse-spark-1.3-contributor" })`. Any OpenRouter slug works this way — see the note above the table — and cost tracking works correctly once you opt in. OpenRouter itself blocks it with HTTP 404 (`paid-model-training-violation-by-account`) until the account's privacy settings allow training endpoints, at [openrouter.ai/settings/privacy](https://openrouter.ai/settings/privacy). Only use it if sending your screenshots to Meta for training is a trade you've deliberately made.
594
632
 
595
633
  `qwen/qwen3.7-max` and the DeepSeek V4 family (`deepseek/deepseek-v4-pro`, `deepseek/deepseek-v4-flash`, and dated variants such as `deepseek/deepseek-v4-pro-0813`) are not listed because they accept no image input on OpenRouter.
package/dist/index.cjs CHANGED
@@ -89,6 +89,7 @@ var Provider = {
89
89
  };
90
90
  var Model = {
91
91
  Anthropic: {
92
+ FABLE_5_1: "claude-fable-5-1",
92
93
  FABLE_5: "claude-fable-5",
93
94
  OPUS_5: "claude-opus-5",
94
95
  OPUS_4_8: "claude-opus-4-8",
@@ -99,6 +100,7 @@ var Model = {
99
100
  HAIKU_4_5: "claude-haiku-4-5"
100
101
  },
101
102
  OpenAI: {
103
+ GPT_6_ASTRA: "gpt-6-astra",
102
104
  GPT_5_6_SOL: "gpt-5.6-sol",
103
105
  GPT_5_6_TERRA: "gpt-5.6-terra",
104
106
  GPT_5_6_LUNA: "gpt-5.6-luna",
@@ -133,7 +135,8 @@ var Model = {
133
135
  KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
134
136
  QWEN_3_8_MAX: "qwen/qwen3.8-max",
135
137
  QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
136
- QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
138
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
139
+ GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
137
140
  }
138
141
  };
139
142
  var DEFAULT_MODELS = {
@@ -144,6 +147,10 @@ var DEFAULT_MODELS = {
144
147
  };
145
148
  var DEFAULT_MAX_TOKENS = 4096;
146
149
  var OPENAI_REASONING_MAX_TOKENS = 16384;
150
+ var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
151
+ var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
152
+ Model.OpenAI.GPT_6_ASTRA
153
+ ]);
147
154
  var MODEL_TO_PROVIDER = new Map([
148
155
  ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
149
156
  ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
@@ -488,22 +495,50 @@ function buildComparePrompt(options) {
488
495
  }
489
496
 
490
497
  // src/templates/elements-visibility.ts
491
- var ELEMENTS_VISIBLE_ROLE = "Check whether specific UI elements are present and fully visible in this screenshot.";
492
498
  var ELEMENTS_HIDDEN_ROLE = "Check whether specific UI elements are absent or hidden in this screenshot.";
493
- var ELEMENTS_VISIBLE_EDGE_RULES = [
494
- "If an element is partially visible (cut off by screenshot boundary), it is NOT considered fully visible \u2014 the check for that element should fail. Note the partial visibility in your reasoning."
499
+ function joinClauses(clauses) {
500
+ if (clauses.length <= 1) return clauses[0] ?? "";
501
+ if (clauses.length === 2) return `${clauses[0]} and ${clauses[1]}`;
502
+ return `${clauses.slice(0, -1).join(", ")}, and ${clauses[clauses.length - 1]}`;
503
+ }
504
+ function visibleRole(finalState, requireCorrectRendering) {
505
+ const clauses = ["present", "properly visible"];
506
+ if (requireCorrectRendering) clauses.push("correctly rendered");
507
+ if (finalState) clauses.push("in their finished state");
508
+ return `Check whether specific UI elements are ${joinClauses(clauses)} in this screenshot.`;
509
+ }
510
+ var ELEMENTS_VISIBLE_CLIPPING_RULES = [
511
+ "When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
512
+ "An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
513
+ "An element you cannot see at all is not visible, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
495
514
  ];
496
- var ELEMENTS_HIDDEN_EDGE_RULES = [
497
- "If an element is partially visible (cut off by screenshot boundary), it is NOT considered hidden \u2014 the check for that element should fail. Note the partial visibility in your reasoning."
515
+ var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
516
+ var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
517
+ var ELEMENTS_VISIBLE_OCCLUSION_RULE = "An element a user could not read or use because a modal, dialog, cookie banner, toast or similar overlay covers it is NOT visible: the check for that element FAILS.";
518
+ function visibleRules(finalState, requireCorrectRendering) {
519
+ return [
520
+ ...ELEMENTS_VISIBLE_CLIPPING_RULES,
521
+ ...finalState ? [ELEMENTS_VISIBLE_FINAL_STATE_RULE] : [],
522
+ ...requireCorrectRendering ? [ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE] : [],
523
+ ELEMENTS_VISIBLE_OCCLUSION_RULE
524
+ ];
525
+ }
526
+ var ELEMENTS_HIDDEN_BASE_RULES = [
527
+ "An element that is rendered at all, even partly, is not hidden, so the check for that element FAILS. This includes one peeking past the edge of a scrollable row or feed, which the user reaches by scrolling normally. Note the partial visibility in your reasoning.",
528
+ "An element that appears nowhere in this screenshot counts as hidden, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
498
529
  ];
530
+ var ELEMENTS_HIDDEN_FINAL_STATE_RULE = "An element sitting under a loading spinner, skeleton, progress bar or error overlay is still rendered, so it is not hidden and the check for that element FAILS. It is not properly visible either; that is what the visible check is for.";
531
+ function hiddenRules(finalState) {
532
+ return finalState ? [...ELEMENTS_HIDDEN_BASE_RULES, ELEMENTS_HIDDEN_FINAL_STATE_RULE] : ELEMENTS_HIDDEN_BASE_RULES;
533
+ }
499
534
  function buildElementsVisibilityPrompt(elements, visible, options) {
500
- const statements = visible ? elements.map((el) => `The element "${el}" is fully visible on the page`) : elements.map((el) => `The element "${el}" is NOT visible on the page`);
501
- const defaultRules = visible ? ELEMENTS_VISIBLE_EDGE_RULES : ELEMENTS_HIDDEN_EDGE_RULES;
535
+ const statements = visible ? elements.map((el) => `The element "${el}" is visible on the page`) : elements.map((el) => `The element "${el}" is NOT visible on the page`);
536
+ const finalState = options?.finalState ?? true;
537
+ const correctRendering = visible && (options?.requireCorrectRendering ?? false);
538
+ const defaultRules = visible ? visibleRules(finalState, correctRendering) : hiddenRules(finalState);
502
539
  const instructions = options?.instructions ? [...defaultRules, ...options.instructions] : defaultRules;
503
- return buildCheckPrompt(statements, {
504
- role: visible ? ELEMENTS_VISIBLE_ROLE : ELEMENTS_HIDDEN_ROLE,
505
- instructions
506
- });
540
+ const role = visible ? visibleRole(finalState, correctRendering) : ELEMENTS_HIDDEN_ROLE;
541
+ return buildCheckPrompt(statements, { role, instructions });
507
542
  }
508
543
 
509
544
  // src/templates/accessibility.ts
@@ -613,6 +648,7 @@ function parseRetryAfter(value) {
613
648
 
614
649
  // src/providers/anthropic.ts
615
650
  var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
651
+ Model.Anthropic.FABLE_5_1,
616
652
  Model.Anthropic.FABLE_5,
617
653
  Model.Anthropic.OPUS_5,
618
654
  Model.Anthropic.OPUS_4_8,
@@ -636,12 +672,14 @@ var AnthropicDriver = class {
636
672
  maxTokens;
637
673
  apiKeyOrEnv;
638
674
  reasoningEffort;
675
+ timeout;
639
676
  constructor(config) {
640
677
  this.model = config.model;
641
678
  this.maxTokens = config.maxTokens;
642
679
  this.client = null;
643
680
  this.apiKeyOrEnv = config.apiKey;
644
681
  this.reasoningEffort = config.reasoningEffort;
682
+ this.timeout = config.timeout;
645
683
  }
646
684
  async getClient() {
647
685
  if (this.client) return this.client;
@@ -660,7 +698,10 @@ var AnthropicDriver = class {
660
698
  "Anthropic API key not found. Set ANTHROPIC_API_KEY or pass apiKey in config."
661
699
  );
662
700
  }
663
- this.client = new Anthropic({ apiKey });
701
+ this.client = new Anthropic({
702
+ apiKey,
703
+ ...this.timeout !== void 0 && { timeout: this.timeout }
704
+ });
664
705
  return this.client;
665
706
  }
666
707
  async sendMessage(images, prompt, _options) {
@@ -758,6 +799,7 @@ var GoogleDriver = class {
758
799
  apiKeyOrEnv;
759
800
  reasoningEffort;
760
801
  imageDetail;
802
+ timeout;
761
803
  constructor(config) {
762
804
  this.model = config.model;
763
805
  this.maxTokens = config.maxTokens;
@@ -765,6 +807,7 @@ var GoogleDriver = class {
765
807
  this.apiKeyOrEnv = config.apiKey;
766
808
  this.reasoningEffort = config.reasoningEffort;
767
809
  this.imageDetail = config.imageDetail;
810
+ this.timeout = config.timeout;
768
811
  }
769
812
  toGeminiParts(images) {
770
813
  return images.map((img) => ({
@@ -788,7 +831,10 @@ var GoogleDriver = class {
788
831
  "Google API key not found. Set GOOGLE_API_KEY or pass apiKey in config."
789
832
  );
790
833
  }
791
- this.client = new GoogleGenAI({ apiKey });
834
+ this.client = new GoogleGenAI({
835
+ apiKey,
836
+ ...this.timeout !== void 0 && { httpOptions: { timeout: this.timeout } }
837
+ });
792
838
  return this.client;
793
839
  }
794
840
  async sendMessage(images, prompt, _options) {
@@ -874,6 +920,7 @@ var OpenAIDriver = class {
874
920
  apiKeyOrEnv;
875
921
  reasoningEffort;
876
922
  imageDetail;
923
+ timeout;
877
924
  constructor(config) {
878
925
  this.model = config.model;
879
926
  this.maxTokens = config.maxTokens;
@@ -881,6 +928,7 @@ var OpenAIDriver = class {
881
928
  this.apiKeyOrEnv = config.apiKey;
882
929
  this.reasoningEffort = config.reasoningEffort;
883
930
  this.imageDetail = config.imageDetail;
931
+ this.timeout = config.timeout;
884
932
  }
885
933
  async getClient() {
886
934
  if (this.client) return this.client;
@@ -897,7 +945,10 @@ var OpenAIDriver = class {
897
945
  "OpenAI API key not found. Set OPENAI_API_KEY or pass apiKey in config."
898
946
  );
899
947
  }
900
- this.client = new OpenAI({ apiKey });
948
+ this.client = new OpenAI({
949
+ apiKey,
950
+ ...this.timeout !== void 0 && { timeout: this.timeout }
951
+ });
901
952
  return this.client;
902
953
  }
903
954
  async sendMessage(images, prompt, options) {
@@ -972,6 +1023,7 @@ var OpenRouterDriver = class {
972
1023
  apiKeyOrEnv;
973
1024
  reasoningEffort;
974
1025
  imageDetail;
1026
+ timeout;
975
1027
  constructor(config) {
976
1028
  this.model = config.model;
977
1029
  this.maxTokens = config.maxTokens;
@@ -979,6 +1031,7 @@ var OpenRouterDriver = class {
979
1031
  this.apiKeyOrEnv = config.apiKey;
980
1032
  this.reasoningEffort = config.reasoningEffort;
981
1033
  this.imageDetail = config.imageDetail;
1034
+ this.timeout = config.timeout;
982
1035
  }
983
1036
  async getClient() {
984
1037
  if (this.client) return this.client;
@@ -997,7 +1050,11 @@ var OpenRouterDriver = class {
997
1050
  "OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
998
1051
  );
999
1052
  }
1000
- this.client = new OpenAI({ apiKey, baseURL: OPENROUTER_BASE_URL });
1053
+ this.client = new OpenAI({
1054
+ apiKey,
1055
+ baseURL: OPENROUTER_BASE_URL,
1056
+ ...this.timeout !== void 0 && { timeout: this.timeout }
1057
+ });
1001
1058
  return this.client;
1002
1059
  }
1003
1060
  async sendMessage(images, prompt, options) {
@@ -1127,13 +1184,21 @@ function resolveConfig(config) {
1127
1184
  `
1128
1185
  );
1129
1186
  }
1187
+ if (config.timeout !== void 0 && (!Number.isFinite(config.timeout) || config.timeout <= 0)) {
1188
+ throw new VisualAIConfigError(
1189
+ `Invalid timeout: ${config.timeout}. Must be a positive number of milliseconds.`
1190
+ );
1191
+ }
1130
1192
  const userSetMaxTokens = config.maxTokens !== void 0;
1131
1193
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
1132
- if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
1133
- maxTokens = OPENAI_REASONING_MAX_TOKENS;
1194
+ const effortNeedsLargeBudget = config.reasoningEffort === "high" || config.reasoningEffort === "xhigh";
1195
+ const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
1196
+ if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
1197
+ maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
1134
1198
  if (debug) {
1199
+ const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${config.reasoningEffort}"`;
1135
1200
  process.stderr.write(
1136
- `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for provider "${provider}" with reasoningEffort "${config.reasoningEffort}".
1201
+ `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
1137
1202
  `
1138
1203
  );
1139
1204
  }
@@ -1146,6 +1211,7 @@ function resolveConfig(config) {
1146
1211
  reasoningEffort: config.reasoningEffort,
1147
1212
  maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
1148
1213
  imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
1214
+ timeout: config.timeout,
1149
1215
  debug,
1150
1216
  debugPrompt,
1151
1217
  debugResponse,
@@ -1156,6 +1222,11 @@ function resolveConfig(config) {
1156
1222
  // src/core/pricing.ts
1157
1223
  var PER_MILLION = 1e6;
1158
1224
  var PRICING_TABLE = {
1225
+ // Fable 5.1 is priced identically to Fable 5, the tier it succeeds.
1226
+ [`${Provider.ANTHROPIC}:${Model.Anthropic.FABLE_5_1}`]: {
1227
+ inputPricePerToken: 10 / PER_MILLION,
1228
+ outputPricePerToken: 50 / PER_MILLION
1229
+ },
1159
1230
  [`${Provider.ANTHROPIC}:${Model.Anthropic.FABLE_5}`]: {
1160
1231
  inputPricePerToken: 10 / PER_MILLION,
1161
1232
  outputPricePerToken: 50 / PER_MILLION
@@ -1188,6 +1259,12 @@ var PRICING_TABLE = {
1188
1259
  inputPricePerToken: 1 / PER_MILLION,
1189
1260
  outputPricePerToken: 5 / PER_MILLION
1190
1261
  },
1262
+ // Cached input is $1/MTok and cache writes $12.50/MTok; neither is modelled
1263
+ // here, since `calculateCost` applies no cache discount on any provider.
1264
+ [`${Provider.OPENAI}:${Model.OpenAI.GPT_6_ASTRA}`]: {
1265
+ inputPricePerToken: 10 / PER_MILLION,
1266
+ outputPricePerToken: 50 / PER_MILLION
1267
+ },
1191
1268
  [`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
1192
1269
  inputPricePerToken: 5 / PER_MILLION,
1193
1270
  outputPricePerToken: 30 / PER_MILLION
@@ -1311,6 +1388,12 @@ var PRICING_TABLE = {
1311
1388
  [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
1312
1389
  inputPricePerToken: 0.1875 / PER_MILLION,
1313
1390
  outputPricePerToken: 1.125 / PER_MILLION
1391
+ },
1392
+ // Verified 2026-09-10 against https://openrouter.ai/api/v1/models. Cached
1393
+ // input is $0.03/MTok, not modelled (no provider gets a cache discount here).
1394
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
1395
+ inputPricePerToken: 0.15 / PER_MILLION,
1396
+ outputPricePerToken: 0.5 / PER_MILLION
1314
1397
  }
1315
1398
  };
1316
1399
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -2196,10 +2279,34 @@ function stripCodeFences(text) {
2196
2279
  var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
2197
2280
  var AskResponseSchema = AskResultSchema.omit({ usage: true });
2198
2281
  var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
2282
+ var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
2283
+ function parseJson(text) {
2284
+ try {
2285
+ return JSON.parse(text);
2286
+ } catch (first) {
2287
+ try {
2288
+ return JSON.parse(text.replace(/[\u0000-\u001f]/g, " "));
2289
+ } catch {
2290
+ throw first;
2291
+ }
2292
+ }
2293
+ }
2294
+ function stripControlCharacters(value) {
2295
+ if (typeof value === "string") return value.replace(STRAY_CONTROL_CHARS, "");
2296
+ if (Array.isArray(value)) return value.map((item) => stripControlCharacters(item));
2297
+ if (value !== null && typeof value === "object") {
2298
+ const out = {};
2299
+ for (const [key, item] of Object.entries(value)) {
2300
+ out[key] = stripControlCharacters(item);
2301
+ }
2302
+ return out;
2303
+ }
2304
+ return value;
2305
+ }
2199
2306
  function parseResponse(raw, schema) {
2200
2307
  let parsed;
2201
2308
  try {
2202
- parsed = JSON.parse(stripCodeFences(raw));
2309
+ parsed = stripControlCharacters(parseJson(stripCodeFences(raw)));
2203
2310
  } catch {
2204
2311
  throw new VisualAIResponseParseError(
2205
2312
  `Failed to parse AI response as JSON: ${raw.slice(0, 200)}`,
@@ -2294,7 +2401,8 @@ function visualAI(config = {}) {
2294
2401
  model: resolvedConfig.model,
2295
2402
  maxTokens: resolvedConfig.maxTokens,
2296
2403
  reasoningEffort: resolvedConfig.reasoningEffort,
2297
- imageDetail: resolvedConfig.imageDetail
2404
+ imageDetail: resolvedConfig.imageDetail,
2405
+ timeout: resolvedConfig.timeout
2298
2406
  };
2299
2407
  const driver = createDriver(resolvedConfig.provider, driverConfig);
2300
2408
  const maxImageDimension = resolvedConfig.maxImageDimension;