visual-ai-assertions 0.21.0 → 0.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +51 -13
- package/dist/index.cjs +129 -21
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +45 -0
- package/dist/index.d.ts +45 -0
- package/dist/index.js +129 -21
- package/dist/index.js.map +1 -1
- package/package.json +3 -2
package/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# visual-ai-assertions
|
|
2
2
|
|
|
3
|
-
AI-powered visual assertions for E2E tests. Send screenshots — or short video recordings — to Claude, GPT, Gemini — or Grok, Kimi, and
|
|
3
|
+
AI-powered visual assertions for E2E tests. Send screenshots — or short video recordings — to Claude, GPT, Gemini — or Grok, Kimi, Qwen, and GLM via OpenRouter — and get structured, typed results.
|
|
4
4
|
|
|
5
5
|
## Installation
|
|
6
6
|
|
|
@@ -11,7 +11,7 @@ npm install visual-ai-assertions
|
|
|
11
11
|
# Optional: install additional provider SDKs
|
|
12
12
|
npm install @anthropic-ai/sdk # for Claude
|
|
13
13
|
npm install @google/genai # for Gemini
|
|
14
|
-
# OpenRouter (Grok, Kimi, Qwen, ...) uses the OpenAI SDK — no extra install
|
|
14
|
+
# OpenRouter (Grok, Kimi, Qwen, GLM, ...) uses the OpenAI SDK — no extra install
|
|
15
15
|
|
|
16
16
|
# Zod is a peer dependency
|
|
17
17
|
npm install zod
|
|
@@ -104,6 +104,7 @@ const ai = visualAI({
|
|
|
104
104
|
debug: true, // optional, logs prompts/responses to stderr
|
|
105
105
|
maxTokens: 4096, // optional, default 4096
|
|
106
106
|
reasoningEffort: "high", // optional, "low" | "medium" | "high" | "xhigh"
|
|
107
|
+
timeout: 120_000, // optional, ms — defaults to the provider SDK's own timeout
|
|
107
108
|
trackUsage: false, // optional, defaults to false — usage stats to stderr
|
|
108
109
|
});
|
|
109
110
|
|
|
@@ -255,6 +256,33 @@ await ai.elementsVisible(screenshot, ["Submit button", "Nav bar", "Footer"]);
|
|
|
255
256
|
// Check that UI elements are hidden
|
|
256
257
|
await ai.elementsHidden(screenshot, ["Loading spinner", "Error modal"]);
|
|
257
258
|
|
|
259
|
+
// Clipping is judged the way a tester would. An element cut off at the trailing
|
|
260
|
+
// edge of a scrollable row or feed passes, because scrolling reaches it. One
|
|
261
|
+
// sliced by the screen edge or by fixed chrome such as the status bar or a
|
|
262
|
+
// sticky nav fails, because scrolling cannot. An element you cannot see at all
|
|
263
|
+
// fails: only what the screenshot shows is judged.
|
|
264
|
+
//
|
|
265
|
+
// Overlays are judged by what they say about the element's state, not by how
|
|
266
|
+
// much they cover. A badge, favourite icon or price pill coexists with finished
|
|
267
|
+
// content and leaves the element visible. A loading spinner, skeleton, progress
|
|
268
|
+
// bar or error overlay says it is not ready, so the check fails even though you
|
|
269
|
+
// can still see what is underneath. A modal, dialog or cookie banner that a user
|
|
270
|
+
// could not read or use past also fails.
|
|
271
|
+
//
|
|
272
|
+
// Pass finalState: false when the screenshot was deliberately captured mid-load,
|
|
273
|
+
// so loading chrome is expected rather than a defect. The finished-state rules
|
|
274
|
+
// are left out; presence, clipping and blocking overlays are still judged.
|
|
275
|
+
await ai.elementsVisible(screenshot, ["Product image"], { finalState: false });
|
|
276
|
+
|
|
277
|
+
// By default this is a presence check: an element counts as visible if it is
|
|
278
|
+
// there at all, whatever it looks like. Pass requireCorrectRendering: true to
|
|
279
|
+
// also fail an element that is present but clearly badly rendered — unreadable
|
|
280
|
+
// contrast, overlapping or misaligned elements, text cut off mid-word — with
|
|
281
|
+
// the model saying the element is present before naming the defect. Opt in per
|
|
282
|
+
// assertion: measured on the bench, it catches exactly that case and makes
|
|
283
|
+
// models flakier on plain presence questions.
|
|
284
|
+
await ai.elementsVisible(screenshot, ["Promo banner"], { requireCorrectRendering: true });
|
|
285
|
+
|
|
258
286
|
// Accessibility checks (contrast, readability, interactive visibility, color blindness, color-alone meaning)
|
|
259
287
|
await ai.accessibility(screenshot);
|
|
260
288
|
await ai.accessibility(screenshot, {
|
|
@@ -463,16 +491,17 @@ The `VisualAIKnownError` union and `isVisualAIKnownError()` helper are useful wh
|
|
|
463
491
|
|
|
464
492
|
## Configuration
|
|
465
493
|
|
|
466
|
-
| Option | Type | Default | Description
|
|
467
|
-
| ----------------- | ------- | ---------------- |
|
|
468
|
-
| `apiKey` | string | env var | API key for the provider
|
|
469
|
-
| `model` | string | provider default | Model to use
|
|
470
|
-
| `debug` | boolean | `false` | Enable error diagnostic logging to stderr
|
|
471
|
-
| `debugPrompt` | boolean | `false` | Log prompts to stderr
|
|
472
|
-
| `debugResponse` | boolean | `false` | Log responses to stderr
|
|
473
|
-
| `maxTokens` | number | `4096` | Max tokens for AI response
|
|
474
|
-
| `reasoningEffort` | string | `undefined` | `"low"` `"medium"` `"high"` `"xhigh"` — controls how deeply the model reasons
|
|
475
|
-
| `
|
|
494
|
+
| Option | Type | Default | Description |
|
|
495
|
+
| ----------------- | ------- | ---------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
496
|
+
| `apiKey` | string | env var | API key for the provider |
|
|
497
|
+
| `model` | string | provider default | Model to use |
|
|
498
|
+
| `debug` | boolean | `false` | Enable error diagnostic logging to stderr |
|
|
499
|
+
| `debugPrompt` | boolean | `false` | Log prompts to stderr |
|
|
500
|
+
| `debugResponse` | boolean | `false` | Log responses to stderr |
|
|
501
|
+
| `maxTokens` | number | `4096` | Max tokens for AI response |
|
|
502
|
+
| `reasoningEffort` | string | `undefined` | `"low"` `"medium"` `"high"` `"xhigh"` — controls how deeply the model reasons |
|
|
503
|
+
| `timeout` | number | SDK default | Per-request timeout in ms. Unset leaves each SDK's own default (OpenAI/OpenRouter 10 min, Google 1 min). SDKs retry timeouts, so total wall time can be a multiple of this. |
|
|
504
|
+
| `trackUsage` | boolean | `false` | Log token usage and estimated cost to stderr |
|
|
476
505
|
|
|
477
506
|
## Exported Types
|
|
478
507
|
|
|
@@ -535,7 +564,8 @@ All listed models support image/vision input. Pass any model ID to the `model` c
|
|
|
535
564
|
|
|
536
565
|
| Model | Model ID | Input $/MTok | Output $/MTok | Notes |
|
|
537
566
|
| ----------------- | ------------------- | ------------ | ------------- | ------------------------------------------- |
|
|
538
|
-
| Claude Fable 5
|
|
567
|
+
| Claude Fable 5.1 | `claude-fable-5-1` | $10 | $50 | Most capable; long-horizon agentic work |
|
|
568
|
+
| Claude Fable 5 | `claude-fable-5` | $10 | $50 | Predecessor to Fable 5.1, same price |
|
|
539
569
|
| Claude Opus 4.8 | `claude-opus-4-8` | $5 | $25 | Most capable Opus tier; supports `xhigh` |
|
|
540
570
|
| Claude Opus 4.7 | `claude-opus-4-7` | $5 | $25 | Previous Opus; supports `xhigh` effort tier |
|
|
541
571
|
| Claude Opus 4.6 | `claude-opus-4-6` | $5 | $25 | Previous flagship, 128K max output |
|
|
@@ -547,6 +577,7 @@ All listed models support image/vision input. Pass any model ID to the `model` c
|
|
|
547
577
|
|
|
548
578
|
| Model | Model ID | Input $/MTok | Output $/MTok | Notes |
|
|
549
579
|
| ------------- | --------------- | ------------ | ------------- | -------------------------------------- |
|
|
580
|
+
| GPT-6 Astra | `gpt-6-astra` | $10 | $50 | Most capable; restricted access¹ |
|
|
550
581
|
| GPT-5.6 Sol | `gpt-5.6-sol` | $5 | $30 | Newest flagship, frontier tier |
|
|
551
582
|
| GPT-5.6 Terra | `gpt-5.6-terra` | $2 | $12 | Newest balanced, everyday tier |
|
|
552
583
|
| GPT-5.6 Luna | `gpt-5.6-luna` | $0.20 | $1.20 | **Default** — newest, fastest/cheapest |
|
|
@@ -558,6 +589,10 @@ All listed models support image/vision input. Pass any model ID to the `model` c
|
|
|
558
589
|
| GPT-5.4 nano | `gpt-5.4-nano` | $0.20 | $1.25 | Cheapest older-generation option |
|
|
559
590
|
| GPT-5 mini | `gpt-5-mini` | $0.25 | $2 | Fast and cheap |
|
|
560
591
|
|
|
592
|
+
¹ GPT-6 Astra is rolling out through OpenAI's Trusted Access Program, so many API keys cannot reach it yet — expect a `VisualAIProviderError` naming the model until your account is enabled.
|
|
593
|
+
|
|
594
|
+
Astra reasons heavily enough to spend the entire 4096-token default output budget before emitting an answer, so **it is given a 32768-token budget automatically**, at every reasoning effort rather than only at `high`/`xhigh` like other OpenAI models. That follows OpenAI's guidance to reserve at least 25,000 tokens for reasoning and output. Its output length is erratic — identical calls have used anywhere from 0 to 16384+ reasoning tokens — so a large budget reduces truncation without eliminating it; a call that exhausts the budget still bills for the tokens it burned. Passing `maxTokens` explicitly still wins. It also accepts a fifth reasoning level, `max`, above `xhigh`; this library's `reasoningEffort` stops at `xhigh`, which is passed through unchanged, so `max` is not currently reachable.
|
|
595
|
+
|
|
561
596
|
### Google
|
|
562
597
|
|
|
563
598
|
| Model | Model ID | Input $/MTok | Output $/MTok | Notes |
|
|
@@ -587,9 +622,12 @@ Any [OpenRouter](https://openrouter.ai/models) model slug (always `vendor/model`
|
|
|
587
622
|
| Qwen3.8 Max | `qwen/qwen3.8-max` | $2 | $6 | First Max tier with image input |
|
|
588
623
|
| Qwen3.7 Plus | `qwen/qwen3.7-plus` | $0.32 | $1.28 | Cost-effective, GUI/screen-reading |
|
|
589
624
|
| Qwen3.6 Flash | `qwen/qwen3.6-flash` | $0.19 | $1.13 | **Default** — cheap flash vision tier |
|
|
625
|
+
| GLM 5.3 Flash | `z-ai/glm-5.3-flash` | $0.15 | $0.50 | Z.ai flash tier, 1.3M context² |
|
|
590
626
|
|
|
591
627
|
¹ Muse Spark 1.3 is age-gated by OpenRouter: calls return HTTP 403 (`VisualAIAuthError`) until the account completes the 18+ confirmation at [openrouter.ai/settings/preferences](https://openrouter.ai/settings/preferences). It also reasons by default — expect several hundred reasoning tokens per call even with no `reasoningEffort` set.
|
|
592
628
|
|
|
629
|
+
² GLM 5.3 Flash reasons by default — expect one to two hundred reasoning tokens per call even with no `reasoningEffort` set, billed at the output rate. OpenRouter's own context cap for it is 1,048,576 tokens (Z.ai lists 1,310,720) and its output ceiling is 131,072.
|
|
630
|
+
|
|
593
631
|
Meta also publishes `meta/muse-spark-1.3-contributor`, the same model at $0.10 / $0.20 per MTok — about 12x cheaper — because Meta uses everything submitted through it for product improvement. It has **no named constant** (`Model.OpenRouter` does not expose it) and never appears by default anywhere in this library, so using it takes a deliberate, explicit choice: pass the slug directly as a plain string, `visualAI({ model: "meta/muse-spark-1.3-contributor" })`. Any OpenRouter slug works this way — see the note above the table — and cost tracking works correctly once you opt in. OpenRouter itself blocks it with HTTP 404 (`paid-model-training-violation-by-account`) until the account's privacy settings allow training endpoints, at [openrouter.ai/settings/privacy](https://openrouter.ai/settings/privacy). Only use it if sending your screenshots to Meta for training is a trade you've deliberately made.
|
|
594
632
|
|
|
595
633
|
`qwen/qwen3.7-max` and the DeepSeek V4 family (`deepseek/deepseek-v4-pro`, `deepseek/deepseek-v4-flash`, and dated variants such as `deepseek/deepseek-v4-pro-0813`) are not listed because they accept no image input on OpenRouter.
|
package/dist/index.cjs
CHANGED
|
@@ -89,6 +89,7 @@ var Provider = {
|
|
|
89
89
|
};
|
|
90
90
|
var Model = {
|
|
91
91
|
Anthropic: {
|
|
92
|
+
FABLE_5_1: "claude-fable-5-1",
|
|
92
93
|
FABLE_5: "claude-fable-5",
|
|
93
94
|
OPUS_5: "claude-opus-5",
|
|
94
95
|
OPUS_4_8: "claude-opus-4-8",
|
|
@@ -99,6 +100,7 @@ var Model = {
|
|
|
99
100
|
HAIKU_4_5: "claude-haiku-4-5"
|
|
100
101
|
},
|
|
101
102
|
OpenAI: {
|
|
103
|
+
GPT_6_ASTRA: "gpt-6-astra",
|
|
102
104
|
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
103
105
|
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
104
106
|
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
@@ -133,7 +135,8 @@ var Model = {
|
|
|
133
135
|
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
134
136
|
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
135
137
|
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
136
|
-
QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
|
|
138
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
139
|
+
GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
|
|
137
140
|
}
|
|
138
141
|
};
|
|
139
142
|
var DEFAULT_MODELS = {
|
|
@@ -144,6 +147,10 @@ var DEFAULT_MODELS = {
|
|
|
144
147
|
};
|
|
145
148
|
var DEFAULT_MAX_TOKENS = 4096;
|
|
146
149
|
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
150
|
+
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
151
|
+
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
152
|
+
Model.OpenAI.GPT_6_ASTRA
|
|
153
|
+
]);
|
|
147
154
|
var MODEL_TO_PROVIDER = new Map([
|
|
148
155
|
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
149
156
|
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
@@ -488,22 +495,50 @@ function buildComparePrompt(options) {
|
|
|
488
495
|
}
|
|
489
496
|
|
|
490
497
|
// src/templates/elements-visibility.ts
|
|
491
|
-
var ELEMENTS_VISIBLE_ROLE = "Check whether specific UI elements are present and fully visible in this screenshot.";
|
|
492
498
|
var ELEMENTS_HIDDEN_ROLE = "Check whether specific UI elements are absent or hidden in this screenshot.";
|
|
493
|
-
|
|
494
|
-
|
|
499
|
+
function joinClauses(clauses) {
|
|
500
|
+
if (clauses.length <= 1) return clauses[0] ?? "";
|
|
501
|
+
if (clauses.length === 2) return `${clauses[0]} and ${clauses[1]}`;
|
|
502
|
+
return `${clauses.slice(0, -1).join(", ")}, and ${clauses[clauses.length - 1]}`;
|
|
503
|
+
}
|
|
504
|
+
function visibleRole(finalState, requireCorrectRendering) {
|
|
505
|
+
const clauses = ["present", "properly visible"];
|
|
506
|
+
if (requireCorrectRendering) clauses.push("correctly rendered");
|
|
507
|
+
if (finalState) clauses.push("in their finished state");
|
|
508
|
+
return `Check whether specific UI elements are ${joinClauses(clauses)} in this screenshot.`;
|
|
509
|
+
}
|
|
510
|
+
var ELEMENTS_VISIBLE_CLIPPING_RULES = [
|
|
511
|
+
"When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
|
|
512
|
+
"An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
|
|
513
|
+
"An element you cannot see at all is not visible, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
|
|
495
514
|
];
|
|
496
|
-
var
|
|
497
|
-
|
|
515
|
+
var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
|
|
516
|
+
var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
|
|
517
|
+
var ELEMENTS_VISIBLE_OCCLUSION_RULE = "An element a user could not read or use because a modal, dialog, cookie banner, toast or similar overlay covers it is NOT visible: the check for that element FAILS.";
|
|
518
|
+
function visibleRules(finalState, requireCorrectRendering) {
|
|
519
|
+
return [
|
|
520
|
+
...ELEMENTS_VISIBLE_CLIPPING_RULES,
|
|
521
|
+
...finalState ? [ELEMENTS_VISIBLE_FINAL_STATE_RULE] : [],
|
|
522
|
+
...requireCorrectRendering ? [ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE] : [],
|
|
523
|
+
ELEMENTS_VISIBLE_OCCLUSION_RULE
|
|
524
|
+
];
|
|
525
|
+
}
|
|
526
|
+
var ELEMENTS_HIDDEN_BASE_RULES = [
|
|
527
|
+
"An element that is rendered at all, even partly, is not hidden, so the check for that element FAILS. This includes one peeking past the edge of a scrollable row or feed, which the user reaches by scrolling normally. Note the partial visibility in your reasoning.",
|
|
528
|
+
"An element that appears nowhere in this screenshot counts as hidden, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
|
|
498
529
|
];
|
|
530
|
+
var ELEMENTS_HIDDEN_FINAL_STATE_RULE = "An element sitting under a loading spinner, skeleton, progress bar or error overlay is still rendered, so it is not hidden and the check for that element FAILS. It is not properly visible either; that is what the visible check is for.";
|
|
531
|
+
function hiddenRules(finalState) {
|
|
532
|
+
return finalState ? [...ELEMENTS_HIDDEN_BASE_RULES, ELEMENTS_HIDDEN_FINAL_STATE_RULE] : ELEMENTS_HIDDEN_BASE_RULES;
|
|
533
|
+
}
|
|
499
534
|
function buildElementsVisibilityPrompt(elements, visible, options) {
|
|
500
|
-
const statements = visible ? elements.map((el) => `The element "${el}" is
|
|
501
|
-
const
|
|
535
|
+
const statements = visible ? elements.map((el) => `The element "${el}" is visible on the page`) : elements.map((el) => `The element "${el}" is NOT visible on the page`);
|
|
536
|
+
const finalState = options?.finalState ?? true;
|
|
537
|
+
const correctRendering = visible && (options?.requireCorrectRendering ?? false);
|
|
538
|
+
const defaultRules = visible ? visibleRules(finalState, correctRendering) : hiddenRules(finalState);
|
|
502
539
|
const instructions = options?.instructions ? [...defaultRules, ...options.instructions] : defaultRules;
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
instructions
|
|
506
|
-
});
|
|
540
|
+
const role = visible ? visibleRole(finalState, correctRendering) : ELEMENTS_HIDDEN_ROLE;
|
|
541
|
+
return buildCheckPrompt(statements, { role, instructions });
|
|
507
542
|
}
|
|
508
543
|
|
|
509
544
|
// src/templates/accessibility.ts
|
|
@@ -613,6 +648,7 @@ function parseRetryAfter(value) {
|
|
|
613
648
|
|
|
614
649
|
// src/providers/anthropic.ts
|
|
615
650
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
651
|
+
Model.Anthropic.FABLE_5_1,
|
|
616
652
|
Model.Anthropic.FABLE_5,
|
|
617
653
|
Model.Anthropic.OPUS_5,
|
|
618
654
|
Model.Anthropic.OPUS_4_8,
|
|
@@ -636,12 +672,14 @@ var AnthropicDriver = class {
|
|
|
636
672
|
maxTokens;
|
|
637
673
|
apiKeyOrEnv;
|
|
638
674
|
reasoningEffort;
|
|
675
|
+
timeout;
|
|
639
676
|
constructor(config) {
|
|
640
677
|
this.model = config.model;
|
|
641
678
|
this.maxTokens = config.maxTokens;
|
|
642
679
|
this.client = null;
|
|
643
680
|
this.apiKeyOrEnv = config.apiKey;
|
|
644
681
|
this.reasoningEffort = config.reasoningEffort;
|
|
682
|
+
this.timeout = config.timeout;
|
|
645
683
|
}
|
|
646
684
|
async getClient() {
|
|
647
685
|
if (this.client) return this.client;
|
|
@@ -660,7 +698,10 @@ var AnthropicDriver = class {
|
|
|
660
698
|
"Anthropic API key not found. Set ANTHROPIC_API_KEY or pass apiKey in config."
|
|
661
699
|
);
|
|
662
700
|
}
|
|
663
|
-
this.client = new Anthropic({
|
|
701
|
+
this.client = new Anthropic({
|
|
702
|
+
apiKey,
|
|
703
|
+
...this.timeout !== void 0 && { timeout: this.timeout }
|
|
704
|
+
});
|
|
664
705
|
return this.client;
|
|
665
706
|
}
|
|
666
707
|
async sendMessage(images, prompt, _options) {
|
|
@@ -758,6 +799,7 @@ var GoogleDriver = class {
|
|
|
758
799
|
apiKeyOrEnv;
|
|
759
800
|
reasoningEffort;
|
|
760
801
|
imageDetail;
|
|
802
|
+
timeout;
|
|
761
803
|
constructor(config) {
|
|
762
804
|
this.model = config.model;
|
|
763
805
|
this.maxTokens = config.maxTokens;
|
|
@@ -765,6 +807,7 @@ var GoogleDriver = class {
|
|
|
765
807
|
this.apiKeyOrEnv = config.apiKey;
|
|
766
808
|
this.reasoningEffort = config.reasoningEffort;
|
|
767
809
|
this.imageDetail = config.imageDetail;
|
|
810
|
+
this.timeout = config.timeout;
|
|
768
811
|
}
|
|
769
812
|
toGeminiParts(images) {
|
|
770
813
|
return images.map((img) => ({
|
|
@@ -788,7 +831,10 @@ var GoogleDriver = class {
|
|
|
788
831
|
"Google API key not found. Set GOOGLE_API_KEY or pass apiKey in config."
|
|
789
832
|
);
|
|
790
833
|
}
|
|
791
|
-
this.client = new GoogleGenAI({
|
|
834
|
+
this.client = new GoogleGenAI({
|
|
835
|
+
apiKey,
|
|
836
|
+
...this.timeout !== void 0 && { httpOptions: { timeout: this.timeout } }
|
|
837
|
+
});
|
|
792
838
|
return this.client;
|
|
793
839
|
}
|
|
794
840
|
async sendMessage(images, prompt, _options) {
|
|
@@ -874,6 +920,7 @@ var OpenAIDriver = class {
|
|
|
874
920
|
apiKeyOrEnv;
|
|
875
921
|
reasoningEffort;
|
|
876
922
|
imageDetail;
|
|
923
|
+
timeout;
|
|
877
924
|
constructor(config) {
|
|
878
925
|
this.model = config.model;
|
|
879
926
|
this.maxTokens = config.maxTokens;
|
|
@@ -881,6 +928,7 @@ var OpenAIDriver = class {
|
|
|
881
928
|
this.apiKeyOrEnv = config.apiKey;
|
|
882
929
|
this.reasoningEffort = config.reasoningEffort;
|
|
883
930
|
this.imageDetail = config.imageDetail;
|
|
931
|
+
this.timeout = config.timeout;
|
|
884
932
|
}
|
|
885
933
|
async getClient() {
|
|
886
934
|
if (this.client) return this.client;
|
|
@@ -897,7 +945,10 @@ var OpenAIDriver = class {
|
|
|
897
945
|
"OpenAI API key not found. Set OPENAI_API_KEY or pass apiKey in config."
|
|
898
946
|
);
|
|
899
947
|
}
|
|
900
|
-
this.client = new OpenAI({
|
|
948
|
+
this.client = new OpenAI({
|
|
949
|
+
apiKey,
|
|
950
|
+
...this.timeout !== void 0 && { timeout: this.timeout }
|
|
951
|
+
});
|
|
901
952
|
return this.client;
|
|
902
953
|
}
|
|
903
954
|
async sendMessage(images, prompt, options) {
|
|
@@ -972,6 +1023,7 @@ var OpenRouterDriver = class {
|
|
|
972
1023
|
apiKeyOrEnv;
|
|
973
1024
|
reasoningEffort;
|
|
974
1025
|
imageDetail;
|
|
1026
|
+
timeout;
|
|
975
1027
|
constructor(config) {
|
|
976
1028
|
this.model = config.model;
|
|
977
1029
|
this.maxTokens = config.maxTokens;
|
|
@@ -979,6 +1031,7 @@ var OpenRouterDriver = class {
|
|
|
979
1031
|
this.apiKeyOrEnv = config.apiKey;
|
|
980
1032
|
this.reasoningEffort = config.reasoningEffort;
|
|
981
1033
|
this.imageDetail = config.imageDetail;
|
|
1034
|
+
this.timeout = config.timeout;
|
|
982
1035
|
}
|
|
983
1036
|
async getClient() {
|
|
984
1037
|
if (this.client) return this.client;
|
|
@@ -997,7 +1050,11 @@ var OpenRouterDriver = class {
|
|
|
997
1050
|
"OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
|
|
998
1051
|
);
|
|
999
1052
|
}
|
|
1000
|
-
this.client = new OpenAI({
|
|
1053
|
+
this.client = new OpenAI({
|
|
1054
|
+
apiKey,
|
|
1055
|
+
baseURL: OPENROUTER_BASE_URL,
|
|
1056
|
+
...this.timeout !== void 0 && { timeout: this.timeout }
|
|
1057
|
+
});
|
|
1001
1058
|
return this.client;
|
|
1002
1059
|
}
|
|
1003
1060
|
async sendMessage(images, prompt, options) {
|
|
@@ -1127,13 +1184,21 @@ function resolveConfig(config) {
|
|
|
1127
1184
|
`
|
|
1128
1185
|
);
|
|
1129
1186
|
}
|
|
1187
|
+
if (config.timeout !== void 0 && (!Number.isFinite(config.timeout) || config.timeout <= 0)) {
|
|
1188
|
+
throw new VisualAIConfigError(
|
|
1189
|
+
`Invalid timeout: ${config.timeout}. Must be a positive number of milliseconds.`
|
|
1190
|
+
);
|
|
1191
|
+
}
|
|
1130
1192
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
1131
1193
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
1132
|
-
|
|
1133
|
-
|
|
1194
|
+
const effortNeedsLargeBudget = config.reasoningEffort === "high" || config.reasoningEffort === "xhigh";
|
|
1195
|
+
const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
|
|
1196
|
+
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
|
|
1197
|
+
maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
|
|
1134
1198
|
if (debug) {
|
|
1199
|
+
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${config.reasoningEffort}"`;
|
|
1135
1200
|
process.stderr.write(
|
|
1136
|
-
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${
|
|
1201
|
+
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
|
|
1137
1202
|
`
|
|
1138
1203
|
);
|
|
1139
1204
|
}
|
|
@@ -1146,6 +1211,7 @@ function resolveConfig(config) {
|
|
|
1146
1211
|
reasoningEffort: config.reasoningEffort,
|
|
1147
1212
|
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1148
1213
|
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
1214
|
+
timeout: config.timeout,
|
|
1149
1215
|
debug,
|
|
1150
1216
|
debugPrompt,
|
|
1151
1217
|
debugResponse,
|
|
@@ -1156,6 +1222,11 @@ function resolveConfig(config) {
|
|
|
1156
1222
|
// src/core/pricing.ts
|
|
1157
1223
|
var PER_MILLION = 1e6;
|
|
1158
1224
|
var PRICING_TABLE = {
|
|
1225
|
+
// Fable 5.1 is priced identically to Fable 5, the tier it succeeds.
|
|
1226
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.FABLE_5_1}`]: {
|
|
1227
|
+
inputPricePerToken: 10 / PER_MILLION,
|
|
1228
|
+
outputPricePerToken: 50 / PER_MILLION
|
|
1229
|
+
},
|
|
1159
1230
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.FABLE_5}`]: {
|
|
1160
1231
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1161
1232
|
outputPricePerToken: 50 / PER_MILLION
|
|
@@ -1188,6 +1259,12 @@ var PRICING_TABLE = {
|
|
|
1188
1259
|
inputPricePerToken: 1 / PER_MILLION,
|
|
1189
1260
|
outputPricePerToken: 5 / PER_MILLION
|
|
1190
1261
|
},
|
|
1262
|
+
// Cached input is $1/MTok and cache writes $12.50/MTok; neither is modelled
|
|
1263
|
+
// here, since `calculateCost` applies no cache discount on any provider.
|
|
1264
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_ASTRA}`]: {
|
|
1265
|
+
inputPricePerToken: 10 / PER_MILLION,
|
|
1266
|
+
outputPricePerToken: 50 / PER_MILLION
|
|
1267
|
+
},
|
|
1191
1268
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
|
|
1192
1269
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1193
1270
|
outputPricePerToken: 30 / PER_MILLION
|
|
@@ -1311,6 +1388,12 @@ var PRICING_TABLE = {
|
|
|
1311
1388
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
|
|
1312
1389
|
inputPricePerToken: 0.1875 / PER_MILLION,
|
|
1313
1390
|
outputPricePerToken: 1.125 / PER_MILLION
|
|
1391
|
+
},
|
|
1392
|
+
// Verified 2026-09-10 against https://openrouter.ai/api/v1/models. Cached
|
|
1393
|
+
// input is $0.03/MTok, not modelled (no provider gets a cache discount here).
|
|
1394
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
|
|
1395
|
+
inputPricePerToken: 0.15 / PER_MILLION,
|
|
1396
|
+
outputPricePerToken: 0.5 / PER_MILLION
|
|
1314
1397
|
}
|
|
1315
1398
|
};
|
|
1316
1399
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -2196,10 +2279,34 @@ function stripCodeFences(text) {
|
|
|
2196
2279
|
var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
|
|
2197
2280
|
var AskResponseSchema = AskResultSchema.omit({ usage: true });
|
|
2198
2281
|
var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
|
|
2282
|
+
var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
|
|
2283
|
+
function parseJson(text) {
|
|
2284
|
+
try {
|
|
2285
|
+
return JSON.parse(text);
|
|
2286
|
+
} catch (first) {
|
|
2287
|
+
try {
|
|
2288
|
+
return JSON.parse(text.replace(/[\u0000-\u001f]/g, " "));
|
|
2289
|
+
} catch {
|
|
2290
|
+
throw first;
|
|
2291
|
+
}
|
|
2292
|
+
}
|
|
2293
|
+
}
|
|
2294
|
+
function stripControlCharacters(value) {
|
|
2295
|
+
if (typeof value === "string") return value.replace(STRAY_CONTROL_CHARS, "");
|
|
2296
|
+
if (Array.isArray(value)) return value.map((item) => stripControlCharacters(item));
|
|
2297
|
+
if (value !== null && typeof value === "object") {
|
|
2298
|
+
const out = {};
|
|
2299
|
+
for (const [key, item] of Object.entries(value)) {
|
|
2300
|
+
out[key] = stripControlCharacters(item);
|
|
2301
|
+
}
|
|
2302
|
+
return out;
|
|
2303
|
+
}
|
|
2304
|
+
return value;
|
|
2305
|
+
}
|
|
2199
2306
|
function parseResponse(raw, schema) {
|
|
2200
2307
|
let parsed;
|
|
2201
2308
|
try {
|
|
2202
|
-
parsed =
|
|
2309
|
+
parsed = stripControlCharacters(parseJson(stripCodeFences(raw)));
|
|
2203
2310
|
} catch {
|
|
2204
2311
|
throw new VisualAIResponseParseError(
|
|
2205
2312
|
`Failed to parse AI response as JSON: ${raw.slice(0, 200)}`,
|
|
@@ -2294,7 +2401,8 @@ function visualAI(config = {}) {
|
|
|
2294
2401
|
model: resolvedConfig.model,
|
|
2295
2402
|
maxTokens: resolvedConfig.maxTokens,
|
|
2296
2403
|
reasoningEffort: resolvedConfig.reasoningEffort,
|
|
2297
|
-
imageDetail: resolvedConfig.imageDetail
|
|
2404
|
+
imageDetail: resolvedConfig.imageDetail,
|
|
2405
|
+
timeout: resolvedConfig.timeout
|
|
2298
2406
|
};
|
|
2299
2407
|
const driver = createDriver(resolvedConfig.provider, driverConfig);
|
|
2300
2408
|
const maxImageDimension = resolvedConfig.maxImageDimension;
|