visual-ai-assertions 0.21.0 → 0.23.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +51 -13
- package/dist/index.cjs +129 -21
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +45 -0
- package/dist/index.d.ts +45 -0
- package/dist/index.js +129 -21
- package/dist/index.js.map +1 -1
- package/package.json +3 -2
package/dist/index.d.cts
CHANGED
|
@@ -32,6 +32,7 @@ declare const Provider: {
|
|
|
32
32
|
/** Known model names grouped by provider. */
|
|
33
33
|
declare const Model: {
|
|
34
34
|
readonly Anthropic: {
|
|
35
|
+
readonly FABLE_5_1: "claude-fable-5-1";
|
|
35
36
|
readonly FABLE_5: "claude-fable-5";
|
|
36
37
|
readonly OPUS_5: "claude-opus-5";
|
|
37
38
|
readonly OPUS_4_8: "claude-opus-4-8";
|
|
@@ -42,6 +43,7 @@ declare const Model: {
|
|
|
42
43
|
readonly HAIKU_4_5: "claude-haiku-4-5";
|
|
43
44
|
};
|
|
44
45
|
readonly OpenAI: {
|
|
46
|
+
readonly GPT_6_ASTRA: "gpt-6-astra";
|
|
45
47
|
readonly GPT_5_6_SOL: "gpt-5.6-sol";
|
|
46
48
|
readonly GPT_5_6_TERRA: "gpt-5.6-terra";
|
|
47
49
|
readonly GPT_5_6_LUNA: "gpt-5.6-luna";
|
|
@@ -77,6 +79,7 @@ declare const Model: {
|
|
|
77
79
|
readonly QWEN_3_8_MAX: "qwen/qwen3.8-max";
|
|
78
80
|
readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
|
|
79
81
|
readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
|
|
82
|
+
readonly GLM_5_3_FLASH: "z-ai/glm-5.3-flash";
|
|
80
83
|
};
|
|
81
84
|
};
|
|
82
85
|
/** Union of all built-in model name literals exposed by `Model`. */
|
|
@@ -667,6 +670,17 @@ interface VisualAIConfig {
|
|
|
667
670
|
* Anthropic (Claude auto-downscales images).
|
|
668
671
|
*/
|
|
669
672
|
imageDetail?: ImageDetailLevel;
|
|
673
|
+
/**
|
|
674
|
+
* Per-request timeout in milliseconds, forwarded to the provider SDK.
|
|
675
|
+
* Omitted by default, so each SDK's own default applies (OpenAI and
|
|
676
|
+
* OpenRouter 10 minutes, Google 1 minute, Anthropic per its own rules).
|
|
677
|
+
*
|
|
678
|
+
* Worth setting for heavy reasoning models: they can spend many minutes on a
|
|
679
|
+
* single call, and the SDK default lets a request hang far longer than most
|
|
680
|
+
* test suites should tolerate. Note that provider SDKs retry timed-out
|
|
681
|
+
* requests, so total wall time can exceed this value by a multiple.
|
|
682
|
+
*/
|
|
683
|
+
timeout?: number;
|
|
670
684
|
trackUsage?: boolean;
|
|
671
685
|
}
|
|
672
686
|
/** Optional instructions for `check()`. */
|
|
@@ -703,6 +717,37 @@ interface CompareOptions {
|
|
|
703
717
|
/** Optional instructions for `elementsVisible()` and `elementsHidden()`. */
|
|
704
718
|
interface ElementsVisibilityOptions {
|
|
705
719
|
instructions?: readonly string[];
|
|
720
|
+
/**
|
|
721
|
+
* Whether the screenshot shows the interface in its finished state. Defaults
|
|
722
|
+
* to `true`, which is what a test asserting on a settled screen wants: an
|
|
723
|
+
* element under a loading spinner, skeleton or error overlay is reported as
|
|
724
|
+
* not properly visible, because the overlay says the content is not ready.
|
|
725
|
+
*
|
|
726
|
+
* Set to `false` when the screenshot was deliberately captured mid-load, so
|
|
727
|
+
* loading chrome is expected rather than a defect. The finished-state rules
|
|
728
|
+
* are then left out of the prompt entirely, and only presence and clipping
|
|
729
|
+
* are judged. Note this omits the guidance rather than inverting it — to have
|
|
730
|
+
* the model actively disregard loading indicators, say so in `instructions`.
|
|
731
|
+
*/
|
|
732
|
+
finalState?: boolean;
|
|
733
|
+
/**
|
|
734
|
+
* Whether an element that is present but clearly badly rendered should fail.
|
|
735
|
+
* Defaults to `false`: `elementsVisible()` is a presence check, and an element
|
|
736
|
+
* counts as visible if it is there at all, whatever it looks like.
|
|
737
|
+
*
|
|
738
|
+
* Set to `true` to judge presentation as well. Text at contrast too low to
|
|
739
|
+
* read, elements overlapping, an element out of alignment with its siblings,
|
|
740
|
+
* or text cut off mid-word then fail, with the model told to say the element
|
|
741
|
+
* is present before naming the defect. Opt in per assertion where a rendering
|
|
742
|
+
* defect is the thing being checked: measured with the rule on and off, it
|
|
743
|
+
* caught exactly the bullet naming a present-but-overlapping element and
|
|
744
|
+
* nothing else, while making the better model noticeably flakier on plain
|
|
745
|
+
* presence questions.
|
|
746
|
+
*
|
|
747
|
+
* Ignored by `elementsHidden()`, where the question is absence and a rendering
|
|
748
|
+
* defect cannot change the answer.
|
|
749
|
+
*/
|
|
750
|
+
requireCorrectRendering?: boolean;
|
|
706
751
|
}
|
|
707
752
|
/** Options for the built-in accessibility template. */
|
|
708
753
|
interface AccessibilityOptions {
|
package/dist/index.d.ts
CHANGED
|
@@ -32,6 +32,7 @@ declare const Provider: {
|
|
|
32
32
|
/** Known model names grouped by provider. */
|
|
33
33
|
declare const Model: {
|
|
34
34
|
readonly Anthropic: {
|
|
35
|
+
readonly FABLE_5_1: "claude-fable-5-1";
|
|
35
36
|
readonly FABLE_5: "claude-fable-5";
|
|
36
37
|
readonly OPUS_5: "claude-opus-5";
|
|
37
38
|
readonly OPUS_4_8: "claude-opus-4-8";
|
|
@@ -42,6 +43,7 @@ declare const Model: {
|
|
|
42
43
|
readonly HAIKU_4_5: "claude-haiku-4-5";
|
|
43
44
|
};
|
|
44
45
|
readonly OpenAI: {
|
|
46
|
+
readonly GPT_6_ASTRA: "gpt-6-astra";
|
|
45
47
|
readonly GPT_5_6_SOL: "gpt-5.6-sol";
|
|
46
48
|
readonly GPT_5_6_TERRA: "gpt-5.6-terra";
|
|
47
49
|
readonly GPT_5_6_LUNA: "gpt-5.6-luna";
|
|
@@ -77,6 +79,7 @@ declare const Model: {
|
|
|
77
79
|
readonly QWEN_3_8_MAX: "qwen/qwen3.8-max";
|
|
78
80
|
readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
|
|
79
81
|
readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
|
|
82
|
+
readonly GLM_5_3_FLASH: "z-ai/glm-5.3-flash";
|
|
80
83
|
};
|
|
81
84
|
};
|
|
82
85
|
/** Union of all built-in model name literals exposed by `Model`. */
|
|
@@ -667,6 +670,17 @@ interface VisualAIConfig {
|
|
|
667
670
|
* Anthropic (Claude auto-downscales images).
|
|
668
671
|
*/
|
|
669
672
|
imageDetail?: ImageDetailLevel;
|
|
673
|
+
/**
|
|
674
|
+
* Per-request timeout in milliseconds, forwarded to the provider SDK.
|
|
675
|
+
* Omitted by default, so each SDK's own default applies (OpenAI and
|
|
676
|
+
* OpenRouter 10 minutes, Google 1 minute, Anthropic per its own rules).
|
|
677
|
+
*
|
|
678
|
+
* Worth setting for heavy reasoning models: they can spend many minutes on a
|
|
679
|
+
* single call, and the SDK default lets a request hang far longer than most
|
|
680
|
+
* test suites should tolerate. Note that provider SDKs retry timed-out
|
|
681
|
+
* requests, so total wall time can exceed this value by a multiple.
|
|
682
|
+
*/
|
|
683
|
+
timeout?: number;
|
|
670
684
|
trackUsage?: boolean;
|
|
671
685
|
}
|
|
672
686
|
/** Optional instructions for `check()`. */
|
|
@@ -703,6 +717,37 @@ interface CompareOptions {
|
|
|
703
717
|
/** Optional instructions for `elementsVisible()` and `elementsHidden()`. */
|
|
704
718
|
interface ElementsVisibilityOptions {
|
|
705
719
|
instructions?: readonly string[];
|
|
720
|
+
/**
|
|
721
|
+
* Whether the screenshot shows the interface in its finished state. Defaults
|
|
722
|
+
* to `true`, which is what a test asserting on a settled screen wants: an
|
|
723
|
+
* element under a loading spinner, skeleton or error overlay is reported as
|
|
724
|
+
* not properly visible, because the overlay says the content is not ready.
|
|
725
|
+
*
|
|
726
|
+
* Set to `false` when the screenshot was deliberately captured mid-load, so
|
|
727
|
+
* loading chrome is expected rather than a defect. The finished-state rules
|
|
728
|
+
* are then left out of the prompt entirely, and only presence and clipping
|
|
729
|
+
* are judged. Note this omits the guidance rather than inverting it — to have
|
|
730
|
+
* the model actively disregard loading indicators, say so in `instructions`.
|
|
731
|
+
*/
|
|
732
|
+
finalState?: boolean;
|
|
733
|
+
/**
|
|
734
|
+
* Whether an element that is present but clearly badly rendered should fail.
|
|
735
|
+
* Defaults to `false`: `elementsVisible()` is a presence check, and an element
|
|
736
|
+
* counts as visible if it is there at all, whatever it looks like.
|
|
737
|
+
*
|
|
738
|
+
* Set to `true` to judge presentation as well. Text at contrast too low to
|
|
739
|
+
* read, elements overlapping, an element out of alignment with its siblings,
|
|
740
|
+
* or text cut off mid-word then fail, with the model told to say the element
|
|
741
|
+
* is present before naming the defect. Opt in per assertion where a rendering
|
|
742
|
+
* defect is the thing being checked: measured with the rule on and off, it
|
|
743
|
+
* caught exactly the bullet naming a present-but-overlapping element and
|
|
744
|
+
* nothing else, while making the better model noticeably flakier on plain
|
|
745
|
+
* presence questions.
|
|
746
|
+
*
|
|
747
|
+
* Ignored by `elementsHidden()`, where the question is absence and a rendering
|
|
748
|
+
* defect cannot change the answer.
|
|
749
|
+
*/
|
|
750
|
+
requireCorrectRendering?: boolean;
|
|
706
751
|
}
|
|
707
752
|
/** Options for the built-in accessibility template. */
|
|
708
753
|
interface AccessibilityOptions {
|
package/dist/index.js
CHANGED
|
@@ -20,6 +20,7 @@ var Provider = {
|
|
|
20
20
|
};
|
|
21
21
|
var Model = {
|
|
22
22
|
Anthropic: {
|
|
23
|
+
FABLE_5_1: "claude-fable-5-1",
|
|
23
24
|
FABLE_5: "claude-fable-5",
|
|
24
25
|
OPUS_5: "claude-opus-5",
|
|
25
26
|
OPUS_4_8: "claude-opus-4-8",
|
|
@@ -30,6 +31,7 @@ var Model = {
|
|
|
30
31
|
HAIKU_4_5: "claude-haiku-4-5"
|
|
31
32
|
},
|
|
32
33
|
OpenAI: {
|
|
34
|
+
GPT_6_ASTRA: "gpt-6-astra",
|
|
33
35
|
GPT_5_6_SOL: "gpt-5.6-sol",
|
|
34
36
|
GPT_5_6_TERRA: "gpt-5.6-terra",
|
|
35
37
|
GPT_5_6_LUNA: "gpt-5.6-luna",
|
|
@@ -64,7 +66,8 @@ var Model = {
|
|
|
64
66
|
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
65
67
|
QWEN_3_8_MAX: "qwen/qwen3.8-max",
|
|
66
68
|
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
67
|
-
QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
|
|
69
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash",
|
|
70
|
+
GLM_5_3_FLASH: "z-ai/glm-5.3-flash"
|
|
68
71
|
}
|
|
69
72
|
};
|
|
70
73
|
var DEFAULT_MODELS = {
|
|
@@ -75,6 +78,10 @@ var DEFAULT_MODELS = {
|
|
|
75
78
|
};
|
|
76
79
|
var DEFAULT_MAX_TOKENS = 4096;
|
|
77
80
|
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
81
|
+
var OPENAI_HEAVY_REASONING_MAX_TOKENS = 32768;
|
|
82
|
+
var MODELS_REQUIRING_LARGE_OUTPUT_BUDGET = /* @__PURE__ */ new Set([
|
|
83
|
+
Model.OpenAI.GPT_6_ASTRA
|
|
84
|
+
]);
|
|
78
85
|
var MODEL_TO_PROVIDER = new Map([
|
|
79
86
|
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
80
87
|
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
@@ -419,22 +426,50 @@ function buildComparePrompt(options) {
|
|
|
419
426
|
}
|
|
420
427
|
|
|
421
428
|
// src/templates/elements-visibility.ts
|
|
422
|
-
var ELEMENTS_VISIBLE_ROLE = "Check whether specific UI elements are present and fully visible in this screenshot.";
|
|
423
429
|
var ELEMENTS_HIDDEN_ROLE = "Check whether specific UI elements are absent or hidden in this screenshot.";
|
|
424
|
-
|
|
425
|
-
|
|
430
|
+
function joinClauses(clauses) {
|
|
431
|
+
if (clauses.length <= 1) return clauses[0] ?? "";
|
|
432
|
+
if (clauses.length === 2) return `${clauses[0]} and ${clauses[1]}`;
|
|
433
|
+
return `${clauses.slice(0, -1).join(", ")}, and ${clauses[clauses.length - 1]}`;
|
|
434
|
+
}
|
|
435
|
+
function visibleRole(finalState, requireCorrectRendering) {
|
|
436
|
+
const clauses = ["present", "properly visible"];
|
|
437
|
+
if (requireCorrectRendering) clauses.push("correctly rendered");
|
|
438
|
+
if (finalState) clauses.push("in their finished state");
|
|
439
|
+
return `Check whether specific UI elements are ${joinClauses(clauses)} in this screenshot.`;
|
|
440
|
+
}
|
|
441
|
+
var ELEMENTS_VISIBLE_CLIPPING_RULES = [
|
|
442
|
+
"When an element is partly rendered but cut off at an edge, decide whether ordinary scrolling would bring it fully into view. For example, a card peeking past the end of a horizontal carousel, a filter chip in a row that continues past the screen edge, or a list item partly below the bottom of a scrolling feed is reachable that way, so the check for that element PASSES. Say in your reasoning that it is reached by scrolling.",
|
|
443
|
+
"An element that scrolling cannot bring into view is NOT properly visible: one sliced by the screen edge itself, or cut off or overlapped by fixed chrome such as the status bar, a notch, a home indicator, a sticky header, or a fixed bottom navigation bar. That is a layout fault, so the check for that element FAILS. Describe the clipping in your reasoning.",
|
|
444
|
+
"An element you cannot see at all is not visible, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
|
|
426
445
|
];
|
|
427
|
-
var
|
|
428
|
-
|
|
446
|
+
var ELEMENTS_VISIBLE_FINAL_STATE_RULE = "Judge each element in its finished, presented state. Things a design draws on top of an element \u2014 a badge, a favourite icon, a duration or price pill, a gradient scrim \u2014 coexist with finished content and leave it visible. An overlay that says the element is NOT ready \u2014 a loading spinner, a skeleton placeholder, a shimmer, a progress bar, an error or retry overlay \u2014 means the element is not properly visible even when you can still make out what sits underneath, so the check for that element FAILS. Name which of the two you are seeing in your reasoning.";
|
|
447
|
+
var ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE = "An element that is present but clearly defective in how it is rendered is NOT properly visible: text at contrast too low to read, elements overlapping or colliding with one another, an element visibly out of alignment with the siblings it should line up with, or text cut off mid-word inside its own container. The check for that element FAILS. In your reasoning, say that the element is present and then name the defect. Only clear, unambiguous defects count: do not fail an element for tight spacing, stylistic choices, or anything you would have to argue for.";
|
|
448
|
+
var ELEMENTS_VISIBLE_OCCLUSION_RULE = "An element a user could not read or use because a modal, dialog, cookie banner, toast or similar overlay covers it is NOT visible: the check for that element FAILS.";
|
|
449
|
+
function visibleRules(finalState, requireCorrectRendering) {
|
|
450
|
+
return [
|
|
451
|
+
...ELEMENTS_VISIBLE_CLIPPING_RULES,
|
|
452
|
+
...finalState ? [ELEMENTS_VISIBLE_FINAL_STATE_RULE] : [],
|
|
453
|
+
...requireCorrectRendering ? [ELEMENTS_VISIBLE_CORRECT_RENDERING_RULE] : [],
|
|
454
|
+
ELEMENTS_VISIBLE_OCCLUSION_RULE
|
|
455
|
+
];
|
|
456
|
+
}
|
|
457
|
+
var ELEMENTS_HIDDEN_BASE_RULES = [
|
|
458
|
+
"An element that is rendered at all, even partly, is not hidden, so the check for that element FAILS. This includes one peeking past the edge of a scrollable row or feed, which the user reaches by scrolling normally. Note the partial visibility in your reasoning.",
|
|
459
|
+
"An element that appears nowhere in this screenshot counts as hidden, even if the page might reveal it after scrolling. Judge only what this screenshot actually shows."
|
|
429
460
|
];
|
|
461
|
+
var ELEMENTS_HIDDEN_FINAL_STATE_RULE = "An element sitting under a loading spinner, skeleton, progress bar or error overlay is still rendered, so it is not hidden and the check for that element FAILS. It is not properly visible either; that is what the visible check is for.";
|
|
462
|
+
function hiddenRules(finalState) {
|
|
463
|
+
return finalState ? [...ELEMENTS_HIDDEN_BASE_RULES, ELEMENTS_HIDDEN_FINAL_STATE_RULE] : ELEMENTS_HIDDEN_BASE_RULES;
|
|
464
|
+
}
|
|
430
465
|
function buildElementsVisibilityPrompt(elements, visible, options) {
|
|
431
|
-
const statements = visible ? elements.map((el) => `The element "${el}" is
|
|
432
|
-
const
|
|
466
|
+
const statements = visible ? elements.map((el) => `The element "${el}" is visible on the page`) : elements.map((el) => `The element "${el}" is NOT visible on the page`);
|
|
467
|
+
const finalState = options?.finalState ?? true;
|
|
468
|
+
const correctRendering = visible && (options?.requireCorrectRendering ?? false);
|
|
469
|
+
const defaultRules = visible ? visibleRules(finalState, correctRendering) : hiddenRules(finalState);
|
|
433
470
|
const instructions = options?.instructions ? [...defaultRules, ...options.instructions] : defaultRules;
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
instructions
|
|
437
|
-
});
|
|
471
|
+
const role = visible ? visibleRole(finalState, correctRendering) : ELEMENTS_HIDDEN_ROLE;
|
|
472
|
+
return buildCheckPrompt(statements, { role, instructions });
|
|
438
473
|
}
|
|
439
474
|
|
|
440
475
|
// src/templates/accessibility.ts
|
|
@@ -544,6 +579,7 @@ function parseRetryAfter(value) {
|
|
|
544
579
|
|
|
545
580
|
// src/providers/anthropic.ts
|
|
546
581
|
var XHIGH_CAPABLE_MODELS = /* @__PURE__ */ new Set([
|
|
582
|
+
Model.Anthropic.FABLE_5_1,
|
|
547
583
|
Model.Anthropic.FABLE_5,
|
|
548
584
|
Model.Anthropic.OPUS_5,
|
|
549
585
|
Model.Anthropic.OPUS_4_8,
|
|
@@ -567,12 +603,14 @@ var AnthropicDriver = class {
|
|
|
567
603
|
maxTokens;
|
|
568
604
|
apiKeyOrEnv;
|
|
569
605
|
reasoningEffort;
|
|
606
|
+
timeout;
|
|
570
607
|
constructor(config) {
|
|
571
608
|
this.model = config.model;
|
|
572
609
|
this.maxTokens = config.maxTokens;
|
|
573
610
|
this.client = null;
|
|
574
611
|
this.apiKeyOrEnv = config.apiKey;
|
|
575
612
|
this.reasoningEffort = config.reasoningEffort;
|
|
613
|
+
this.timeout = config.timeout;
|
|
576
614
|
}
|
|
577
615
|
async getClient() {
|
|
578
616
|
if (this.client) return this.client;
|
|
@@ -591,7 +629,10 @@ var AnthropicDriver = class {
|
|
|
591
629
|
"Anthropic API key not found. Set ANTHROPIC_API_KEY or pass apiKey in config."
|
|
592
630
|
);
|
|
593
631
|
}
|
|
594
|
-
this.client = new Anthropic({
|
|
632
|
+
this.client = new Anthropic({
|
|
633
|
+
apiKey,
|
|
634
|
+
...this.timeout !== void 0 && { timeout: this.timeout }
|
|
635
|
+
});
|
|
595
636
|
return this.client;
|
|
596
637
|
}
|
|
597
638
|
async sendMessage(images, prompt, _options) {
|
|
@@ -689,6 +730,7 @@ var GoogleDriver = class {
|
|
|
689
730
|
apiKeyOrEnv;
|
|
690
731
|
reasoningEffort;
|
|
691
732
|
imageDetail;
|
|
733
|
+
timeout;
|
|
692
734
|
constructor(config) {
|
|
693
735
|
this.model = config.model;
|
|
694
736
|
this.maxTokens = config.maxTokens;
|
|
@@ -696,6 +738,7 @@ var GoogleDriver = class {
|
|
|
696
738
|
this.apiKeyOrEnv = config.apiKey;
|
|
697
739
|
this.reasoningEffort = config.reasoningEffort;
|
|
698
740
|
this.imageDetail = config.imageDetail;
|
|
741
|
+
this.timeout = config.timeout;
|
|
699
742
|
}
|
|
700
743
|
toGeminiParts(images) {
|
|
701
744
|
return images.map((img) => ({
|
|
@@ -719,7 +762,10 @@ var GoogleDriver = class {
|
|
|
719
762
|
"Google API key not found. Set GOOGLE_API_KEY or pass apiKey in config."
|
|
720
763
|
);
|
|
721
764
|
}
|
|
722
|
-
this.client = new GoogleGenAI({
|
|
765
|
+
this.client = new GoogleGenAI({
|
|
766
|
+
apiKey,
|
|
767
|
+
...this.timeout !== void 0 && { httpOptions: { timeout: this.timeout } }
|
|
768
|
+
});
|
|
723
769
|
return this.client;
|
|
724
770
|
}
|
|
725
771
|
async sendMessage(images, prompt, _options) {
|
|
@@ -805,6 +851,7 @@ var OpenAIDriver = class {
|
|
|
805
851
|
apiKeyOrEnv;
|
|
806
852
|
reasoningEffort;
|
|
807
853
|
imageDetail;
|
|
854
|
+
timeout;
|
|
808
855
|
constructor(config) {
|
|
809
856
|
this.model = config.model;
|
|
810
857
|
this.maxTokens = config.maxTokens;
|
|
@@ -812,6 +859,7 @@ var OpenAIDriver = class {
|
|
|
812
859
|
this.apiKeyOrEnv = config.apiKey;
|
|
813
860
|
this.reasoningEffort = config.reasoningEffort;
|
|
814
861
|
this.imageDetail = config.imageDetail;
|
|
862
|
+
this.timeout = config.timeout;
|
|
815
863
|
}
|
|
816
864
|
async getClient() {
|
|
817
865
|
if (this.client) return this.client;
|
|
@@ -828,7 +876,10 @@ var OpenAIDriver = class {
|
|
|
828
876
|
"OpenAI API key not found. Set OPENAI_API_KEY or pass apiKey in config."
|
|
829
877
|
);
|
|
830
878
|
}
|
|
831
|
-
this.client = new OpenAI({
|
|
879
|
+
this.client = new OpenAI({
|
|
880
|
+
apiKey,
|
|
881
|
+
...this.timeout !== void 0 && { timeout: this.timeout }
|
|
882
|
+
});
|
|
832
883
|
return this.client;
|
|
833
884
|
}
|
|
834
885
|
async sendMessage(images, prompt, options) {
|
|
@@ -903,6 +954,7 @@ var OpenRouterDriver = class {
|
|
|
903
954
|
apiKeyOrEnv;
|
|
904
955
|
reasoningEffort;
|
|
905
956
|
imageDetail;
|
|
957
|
+
timeout;
|
|
906
958
|
constructor(config) {
|
|
907
959
|
this.model = config.model;
|
|
908
960
|
this.maxTokens = config.maxTokens;
|
|
@@ -910,6 +962,7 @@ var OpenRouterDriver = class {
|
|
|
910
962
|
this.apiKeyOrEnv = config.apiKey;
|
|
911
963
|
this.reasoningEffort = config.reasoningEffort;
|
|
912
964
|
this.imageDetail = config.imageDetail;
|
|
965
|
+
this.timeout = config.timeout;
|
|
913
966
|
}
|
|
914
967
|
async getClient() {
|
|
915
968
|
if (this.client) return this.client;
|
|
@@ -928,7 +981,11 @@ var OpenRouterDriver = class {
|
|
|
928
981
|
"OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
|
|
929
982
|
);
|
|
930
983
|
}
|
|
931
|
-
this.client = new OpenAI({
|
|
984
|
+
this.client = new OpenAI({
|
|
985
|
+
apiKey,
|
|
986
|
+
baseURL: OPENROUTER_BASE_URL,
|
|
987
|
+
...this.timeout !== void 0 && { timeout: this.timeout }
|
|
988
|
+
});
|
|
932
989
|
return this.client;
|
|
933
990
|
}
|
|
934
991
|
async sendMessage(images, prompt, options) {
|
|
@@ -1058,13 +1115,21 @@ function resolveConfig(config) {
|
|
|
1058
1115
|
`
|
|
1059
1116
|
);
|
|
1060
1117
|
}
|
|
1118
|
+
if (config.timeout !== void 0 && (!Number.isFinite(config.timeout) || config.timeout <= 0)) {
|
|
1119
|
+
throw new VisualAIConfigError(
|
|
1120
|
+
`Invalid timeout: ${config.timeout}. Must be a positive number of milliseconds.`
|
|
1121
|
+
);
|
|
1122
|
+
}
|
|
1061
1123
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
1062
1124
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
1063
|
-
|
|
1064
|
-
|
|
1125
|
+
const effortNeedsLargeBudget = config.reasoningEffort === "high" || config.reasoningEffort === "xhigh";
|
|
1126
|
+
const modelNeedsLargeBudget = MODELS_REQUIRING_LARGE_OUTPUT_BUDGET.has(model);
|
|
1127
|
+
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (effortNeedsLargeBudget || modelNeedsLargeBudget)) {
|
|
1128
|
+
maxTokens = modelNeedsLargeBudget ? OPENAI_HEAVY_REASONING_MAX_TOKENS : OPENAI_REASONING_MAX_TOKENS;
|
|
1065
1129
|
if (debug) {
|
|
1130
|
+
const reason = modelNeedsLargeBudget ? `model "${model}", which exhausts smaller budgets on reasoning at any effort` : `provider "${provider}" with reasoningEffort "${config.reasoningEffort}"`;
|
|
1066
1131
|
process.stderr.write(
|
|
1067
|
-
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${
|
|
1132
|
+
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${maxTokens} for ${reason}.
|
|
1068
1133
|
`
|
|
1069
1134
|
);
|
|
1070
1135
|
}
|
|
@@ -1077,6 +1142,7 @@ function resolveConfig(config) {
|
|
|
1077
1142
|
reasoningEffort: config.reasoningEffort,
|
|
1078
1143
|
maxImageDimension: config.maxImageDimension ?? DEFAULT_MAX_IMAGE_DIMENSION,
|
|
1079
1144
|
imageDetail: config.imageDetail ?? DEFAULT_IMAGE_DETAIL,
|
|
1145
|
+
timeout: config.timeout,
|
|
1080
1146
|
debug,
|
|
1081
1147
|
debugPrompt,
|
|
1082
1148
|
debugResponse,
|
|
@@ -1087,6 +1153,11 @@ function resolveConfig(config) {
|
|
|
1087
1153
|
// src/core/pricing.ts
|
|
1088
1154
|
var PER_MILLION = 1e6;
|
|
1089
1155
|
var PRICING_TABLE = {
|
|
1156
|
+
// Fable 5.1 is priced identically to Fable 5, the tier it succeeds.
|
|
1157
|
+
[`${Provider.ANTHROPIC}:${Model.Anthropic.FABLE_5_1}`]: {
|
|
1158
|
+
inputPricePerToken: 10 / PER_MILLION,
|
|
1159
|
+
outputPricePerToken: 50 / PER_MILLION
|
|
1160
|
+
},
|
|
1090
1161
|
[`${Provider.ANTHROPIC}:${Model.Anthropic.FABLE_5}`]: {
|
|
1091
1162
|
inputPricePerToken: 10 / PER_MILLION,
|
|
1092
1163
|
outputPricePerToken: 50 / PER_MILLION
|
|
@@ -1119,6 +1190,12 @@ var PRICING_TABLE = {
|
|
|
1119
1190
|
inputPricePerToken: 1 / PER_MILLION,
|
|
1120
1191
|
outputPricePerToken: 5 / PER_MILLION
|
|
1121
1192
|
},
|
|
1193
|
+
// Cached input is $1/MTok and cache writes $12.50/MTok; neither is modelled
|
|
1194
|
+
// here, since `calculateCost` applies no cache discount on any provider.
|
|
1195
|
+
[`${Provider.OPENAI}:${Model.OpenAI.GPT_6_ASTRA}`]: {
|
|
1196
|
+
inputPricePerToken: 10 / PER_MILLION,
|
|
1197
|
+
outputPricePerToken: 50 / PER_MILLION
|
|
1198
|
+
},
|
|
1122
1199
|
[`${Provider.OPENAI}:${Model.OpenAI.GPT_5_6_SOL}`]: {
|
|
1123
1200
|
inputPricePerToken: 5 / PER_MILLION,
|
|
1124
1201
|
outputPricePerToken: 30 / PER_MILLION
|
|
@@ -1242,6 +1319,12 @@ var PRICING_TABLE = {
|
|
|
1242
1319
|
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
|
|
1243
1320
|
inputPricePerToken: 0.1875 / PER_MILLION,
|
|
1244
1321
|
outputPricePerToken: 1.125 / PER_MILLION
|
|
1322
|
+
},
|
|
1323
|
+
// Verified 2026-09-10 against https://openrouter.ai/api/v1/models. Cached
|
|
1324
|
+
// input is $0.03/MTok, not modelled (no provider gets a cache discount here).
|
|
1325
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GLM_5_3_FLASH}`]: {
|
|
1326
|
+
inputPricePerToken: 0.15 / PER_MILLION,
|
|
1327
|
+
outputPricePerToken: 0.5 / PER_MILLION
|
|
1245
1328
|
}
|
|
1246
1329
|
};
|
|
1247
1330
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -2127,10 +2210,34 @@ function stripCodeFences(text) {
|
|
|
2127
2210
|
var CheckResponseSchema = CheckResultSchema.omit({ usage: true });
|
|
2128
2211
|
var AskResponseSchema = AskResultSchema.omit({ usage: true });
|
|
2129
2212
|
var CompareResponseSchema = CompareResultSchema.omit({ usage: true });
|
|
2213
|
+
var STRAY_CONTROL_CHARS = /[\u0000-\u0008\u000b\u000c\u000e-\u001f\u007f]/g;
|
|
2214
|
+
function parseJson(text) {
|
|
2215
|
+
try {
|
|
2216
|
+
return JSON.parse(text);
|
|
2217
|
+
} catch (first) {
|
|
2218
|
+
try {
|
|
2219
|
+
return JSON.parse(text.replace(/[\u0000-\u001f]/g, " "));
|
|
2220
|
+
} catch {
|
|
2221
|
+
throw first;
|
|
2222
|
+
}
|
|
2223
|
+
}
|
|
2224
|
+
}
|
|
2225
|
+
function stripControlCharacters(value) {
|
|
2226
|
+
if (typeof value === "string") return value.replace(STRAY_CONTROL_CHARS, "");
|
|
2227
|
+
if (Array.isArray(value)) return value.map((item) => stripControlCharacters(item));
|
|
2228
|
+
if (value !== null && typeof value === "object") {
|
|
2229
|
+
const out = {};
|
|
2230
|
+
for (const [key, item] of Object.entries(value)) {
|
|
2231
|
+
out[key] = stripControlCharacters(item);
|
|
2232
|
+
}
|
|
2233
|
+
return out;
|
|
2234
|
+
}
|
|
2235
|
+
return value;
|
|
2236
|
+
}
|
|
2130
2237
|
function parseResponse(raw, schema) {
|
|
2131
2238
|
let parsed;
|
|
2132
2239
|
try {
|
|
2133
|
-
parsed =
|
|
2240
|
+
parsed = stripControlCharacters(parseJson(stripCodeFences(raw)));
|
|
2134
2241
|
} catch {
|
|
2135
2242
|
throw new VisualAIResponseParseError(
|
|
2136
2243
|
`Failed to parse AI response as JSON: ${raw.slice(0, 200)}`,
|
|
@@ -2225,7 +2332,8 @@ function visualAI(config = {}) {
|
|
|
2225
2332
|
model: resolvedConfig.model,
|
|
2226
2333
|
maxTokens: resolvedConfig.maxTokens,
|
|
2227
2334
|
reasoningEffort: resolvedConfig.reasoningEffort,
|
|
2228
|
-
imageDetail: resolvedConfig.imageDetail
|
|
2335
|
+
imageDetail: resolvedConfig.imageDetail,
|
|
2336
|
+
timeout: resolvedConfig.timeout
|
|
2229
2337
|
};
|
|
2230
2338
|
const driver = createDriver(resolvedConfig.provider, driverConfig);
|
|
2231
2339
|
const maxImageDimension = resolvedConfig.maxImageDimension;
|