visual-ai-assertions 0.13.1 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -14,6 +14,7 @@ declare const Provider: {
14
14
  readonly ANTHROPIC: "anthropic";
15
15
  readonly OPENAI: "openai";
16
16
  readonly GOOGLE: "google";
17
+ readonly OPENROUTER: "openrouter";
17
18
  };
18
19
  /** Known model names grouped by provider. */
19
20
  declare const Model: {
@@ -39,19 +40,34 @@ declare const Model: {
39
40
  readonly GPT_5_MINI: "gpt-5-mini";
40
41
  };
41
42
  readonly Google: {
43
+ readonly GEMINI_3_6_FLASH: "gemini-3.6-flash";
42
44
  readonly GEMINI_3_5_FLASH: "gemini-3.5-flash";
45
+ readonly GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite";
43
46
  readonly GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview";
44
47
  readonly GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite";
45
48
  readonly GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview";
46
49
  };
50
+ /**
51
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
52
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
53
+ * recognizes them. All listed models accept image input.
54
+ */
55
+ readonly OpenRouter: {
56
+ readonly GROK_4_5: "x-ai/grok-4.5";
57
+ readonly KIMI_K3: "moonshotai/kimi-k3";
58
+ readonly KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code";
59
+ readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
60
+ readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
61
+ };
47
62
  };
48
63
  /** Union of all built-in model name literals exposed by `Model`. */
49
- type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google];
64
+ type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google] | (typeof Model.OpenRouter)[keyof typeof Model.OpenRouter];
50
65
  /** Default model selection used when a caller omits `config.model`. */
51
66
  declare const DEFAULT_MODELS: {
52
67
  readonly anthropic: "claude-sonnet-4-6";
53
68
  readonly openai: "gpt-5.4-mini";
54
69
  readonly google: "gemini-3-flash-preview";
70
+ readonly openrouter: "qwen/qwen3.6-flash";
55
71
  };
56
72
  /** Built-in content checks available through `client.content()`. */
57
73
  declare const Content: {
@@ -493,12 +509,42 @@ type ImageInput = Buffer | Uint8Array | string;
493
509
  * an image or a video.
494
510
  */
495
511
  type MediaInput = ImageInput;
512
+ /**
513
+ * A single pre-sampled frame with an explicit timestamp. Use this shape when
514
+ * you want to control the timestamp the model sees; otherwise pass a bare
515
+ * `ImageInput` and the timestamp is derived from `fps` and the frame's position.
516
+ */
517
+ interface TimestampedFrameInput {
518
+ /** Image source for the frame (buffer, data URL, base64, file path, or URL). */
519
+ image: ImageInput;
520
+ /** Timestamp in seconds, from the start of the sequence. Derived from `fps` when omitted. */
521
+ timestampSeconds?: number;
522
+ }
523
+ /**
524
+ * Pre-sampled frames supplied directly instead of a video file. Useful when you
525
+ * already have an array of screenshots and cannot (or would rather not) pass a
526
+ * video — this path never loads ffmpeg. Passed to `check()` / `ask()` in place
527
+ * of a `MediaInput`; downstream handling is identical to a sampled video.
528
+ */
529
+ interface FramesInput {
530
+ /**
531
+ * Ordered frames, earliest first. Each entry is an image source or a
532
+ * `{ image, timestampSeconds }` pair. Must be non-empty and is capped at the
533
+ * same per-request frame limit as video sampling (60).
534
+ */
535
+ frames: ReadonlyArray<ImageInput | TimestampedFrameInput>;
536
+ /**
537
+ * Frame rate used to derive timestamps for frames that don't carry their own
538
+ * `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
539
+ */
540
+ fps?: number;
541
+ }
496
542
  /** Supported image MIME types accepted by all providers. */
497
543
  type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
498
544
  /** Supported video MIME types the client can accept and sample frames from. */
499
545
  type SupportedVideoMimeType = "video/mp4" | "video/webm" | "video/quicktime" | "video/x-matroska";
500
546
  /** Supported provider identifiers. */
501
- type ProviderName = "anthropic" | "openai" | "google";
547
+ type ProviderName = "anthropic" | "openai" | "google" | "openrouter";
502
548
  /**
503
549
  * Configuration for creating a visual AI client.
504
550
  *
@@ -633,9 +679,11 @@ interface VisualAIClient {
633
679
  * frames automatically; statements pass if they are true at any sampled
634
680
  * frame, and each statement result includes the timestamp where it
635
681
  * matched. The `frames` metadata on the result reports which timestamps
636
- * the model saw.
682
+ * the model saw. Pass a `FramesInput` (`{ frames, fps? }`) to supply
683
+ * pre-sampled frames directly — handled identically to a video timeline but
684
+ * without loading ffmpeg.
637
685
  *
638
- * @param input Image or video source as a buffer, URL, file path, or base64 string.
686
+ * @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
639
687
  * @param statements One or more statements to validate against the input.
640
688
  * @param options Optional additional instructions and video sampling overrides.
641
689
  * @returns A structured result describing pass/fail, issues, and statement reasoning.
@@ -658,15 +706,17 @@ interface VisualAIClient {
658
706
  * console.log(result.statements[0].timestampSeconds); // e.g. 3.5
659
707
  * ```
660
708
  */
661
- check(input: MediaInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
709
+ check(input: MediaInput | FramesInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
662
710
  /**
663
711
  * Asks an open-ended question about an image or video and returns a structured summary.
664
712
  *
665
713
  * Video inputs are sampled into frames and analyzed as a chronological
666
714
  * timeline. The result's `frameReferences` array surfaces which frames the
667
- * model relied on for its answer.
715
+ * model relied on for its answer. Pass a `FramesInput` (`{ frames, fps? }`)
716
+ * to supply pre-sampled frames directly — handled identically to a video
717
+ * timeline but without loading ffmpeg.
668
718
  *
669
- * @param input Image or video source as a buffer, URL, file path, or base64 string.
719
+ * @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
670
720
  * @param prompt Prompt describing what to inspect in the input.
671
721
  * @param options Optional additional instructions and video sampling overrides.
672
722
  * @returns A summary with any detected issues.
@@ -678,7 +728,7 @@ interface VisualAIClient {
678
728
  * const result = await client.ask(screenshot, "What looks visually broken on this page?");
679
729
  * ```
680
730
  */
681
- ask(input: MediaInput, prompt: string, options?: AskOptions): Promise<AskResult>;
731
+ ask(input: MediaInput | FramesInput, prompt: string, options?: AskOptions): Promise<AskResult>;
682
732
  /**
683
733
  * Compares two images and reports meaningful visual differences.
684
734
  *
@@ -1055,4 +1105,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
1055
1105
  */
1056
1106
  declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
1057
1107
 
1058
- export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
1108
+ export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
package/dist/index.d.ts CHANGED
@@ -14,6 +14,7 @@ declare const Provider: {
14
14
  readonly ANTHROPIC: "anthropic";
15
15
  readonly OPENAI: "openai";
16
16
  readonly GOOGLE: "google";
17
+ readonly OPENROUTER: "openrouter";
17
18
  };
18
19
  /** Known model names grouped by provider. */
19
20
  declare const Model: {
@@ -39,19 +40,34 @@ declare const Model: {
39
40
  readonly GPT_5_MINI: "gpt-5-mini";
40
41
  };
41
42
  readonly Google: {
43
+ readonly GEMINI_3_6_FLASH: "gemini-3.6-flash";
42
44
  readonly GEMINI_3_5_FLASH: "gemini-3.5-flash";
45
+ readonly GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite";
43
46
  readonly GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview";
44
47
  readonly GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite";
45
48
  readonly GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview";
46
49
  };
50
+ /**
51
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
52
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
53
+ * recognizes them. All listed models accept image input.
54
+ */
55
+ readonly OpenRouter: {
56
+ readonly GROK_4_5: "x-ai/grok-4.5";
57
+ readonly KIMI_K3: "moonshotai/kimi-k3";
58
+ readonly KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code";
59
+ readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
60
+ readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
61
+ };
47
62
  };
48
63
  /** Union of all built-in model name literals exposed by `Model`. */
49
- type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google];
64
+ type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google] | (typeof Model.OpenRouter)[keyof typeof Model.OpenRouter];
50
65
  /** Default model selection used when a caller omits `config.model`. */
51
66
  declare const DEFAULT_MODELS: {
52
67
  readonly anthropic: "claude-sonnet-4-6";
53
68
  readonly openai: "gpt-5.4-mini";
54
69
  readonly google: "gemini-3-flash-preview";
70
+ readonly openrouter: "qwen/qwen3.6-flash";
55
71
  };
56
72
  /** Built-in content checks available through `client.content()`. */
57
73
  declare const Content: {
@@ -493,12 +509,42 @@ type ImageInput = Buffer | Uint8Array | string;
493
509
  * an image or a video.
494
510
  */
495
511
  type MediaInput = ImageInput;
512
+ /**
513
+ * A single pre-sampled frame with an explicit timestamp. Use this shape when
514
+ * you want to control the timestamp the model sees; otherwise pass a bare
515
+ * `ImageInput` and the timestamp is derived from `fps` and the frame's position.
516
+ */
517
+ interface TimestampedFrameInput {
518
+ /** Image source for the frame (buffer, data URL, base64, file path, or URL). */
519
+ image: ImageInput;
520
+ /** Timestamp in seconds, from the start of the sequence. Derived from `fps` when omitted. */
521
+ timestampSeconds?: number;
522
+ }
523
+ /**
524
+ * Pre-sampled frames supplied directly instead of a video file. Useful when you
525
+ * already have an array of screenshots and cannot (or would rather not) pass a
526
+ * video — this path never loads ffmpeg. Passed to `check()` / `ask()` in place
527
+ * of a `MediaInput`; downstream handling is identical to a sampled video.
528
+ */
529
+ interface FramesInput {
530
+ /**
531
+ * Ordered frames, earliest first. Each entry is an image source or a
532
+ * `{ image, timestampSeconds }` pair. Must be non-empty and is capped at the
533
+ * same per-request frame limit as video sampling (60).
534
+ */
535
+ frames: ReadonlyArray<ImageInput | TimestampedFrameInput>;
536
+ /**
537
+ * Frame rate used to derive timestamps for frames that don't carry their own
538
+ * `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
539
+ */
540
+ fps?: number;
541
+ }
496
542
  /** Supported image MIME types accepted by all providers. */
497
543
  type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
498
544
  /** Supported video MIME types the client can accept and sample frames from. */
499
545
  type SupportedVideoMimeType = "video/mp4" | "video/webm" | "video/quicktime" | "video/x-matroska";
500
546
  /** Supported provider identifiers. */
501
- type ProviderName = "anthropic" | "openai" | "google";
547
+ type ProviderName = "anthropic" | "openai" | "google" | "openrouter";
502
548
  /**
503
549
  * Configuration for creating a visual AI client.
504
550
  *
@@ -633,9 +679,11 @@ interface VisualAIClient {
633
679
  * frames automatically; statements pass if they are true at any sampled
634
680
  * frame, and each statement result includes the timestamp where it
635
681
  * matched. The `frames` metadata on the result reports which timestamps
636
- * the model saw.
682
+ * the model saw. Pass a `FramesInput` (`{ frames, fps? }`) to supply
683
+ * pre-sampled frames directly — handled identically to a video timeline but
684
+ * without loading ffmpeg.
637
685
  *
638
- * @param input Image or video source as a buffer, URL, file path, or base64 string.
686
+ * @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
639
687
  * @param statements One or more statements to validate against the input.
640
688
  * @param options Optional additional instructions and video sampling overrides.
641
689
  * @returns A structured result describing pass/fail, issues, and statement reasoning.
@@ -658,15 +706,17 @@ interface VisualAIClient {
658
706
  * console.log(result.statements[0].timestampSeconds); // e.g. 3.5
659
707
  * ```
660
708
  */
661
- check(input: MediaInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
709
+ check(input: MediaInput | FramesInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
662
710
  /**
663
711
  * Asks an open-ended question about an image or video and returns a structured summary.
664
712
  *
665
713
  * Video inputs are sampled into frames and analyzed as a chronological
666
714
  * timeline. The result's `frameReferences` array surfaces which frames the
667
- * model relied on for its answer.
715
+ * model relied on for its answer. Pass a `FramesInput` (`{ frames, fps? }`)
716
+ * to supply pre-sampled frames directly — handled identically to a video
717
+ * timeline but without loading ffmpeg.
668
718
  *
669
- * @param input Image or video source as a buffer, URL, file path, or base64 string.
719
+ * @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
670
720
  * @param prompt Prompt describing what to inspect in the input.
671
721
  * @param options Optional additional instructions and video sampling overrides.
672
722
  * @returns A summary with any detected issues.
@@ -678,7 +728,7 @@ interface VisualAIClient {
678
728
  * const result = await client.ask(screenshot, "What looks visually broken on this page?");
679
729
  * ```
680
730
  */
681
- ask(input: MediaInput, prompt: string, options?: AskOptions): Promise<AskResult>;
731
+ ask(input: MediaInput | FramesInput, prompt: string, options?: AskOptions): Promise<AskResult>;
682
732
  /**
683
733
  * Compares two images and reports meaningful visual differences.
684
734
  *
@@ -1055,4 +1105,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
1055
1105
  */
1056
1106
  declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
1057
1107
 
1058
- export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
1108
+ export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
package/dist/index.js CHANGED
@@ -8,7 +8,8 @@ var ReasoningEffort = {
8
8
  var Provider = {
9
9
  ANTHROPIC: "anthropic",
10
10
  OPENAI: "openai",
11
- GOOGLE: "google"
11
+ GOOGLE: "google",
12
+ OPENROUTER: "openrouter"
12
13
  };
13
14
  var Model = {
14
15
  Anthropic: {
@@ -33,29 +34,47 @@ var Model = {
33
34
  GPT_5_MINI: "gpt-5-mini"
34
35
  },
35
36
  Google: {
37
+ GEMINI_3_6_FLASH: "gemini-3.6-flash",
36
38
  GEMINI_3_5_FLASH: "gemini-3.5-flash",
39
+ GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
37
40
  GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
38
41
  GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
39
42
  GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
43
+ },
44
+ /**
45
+ * Models routed through OpenRouter (https://openrouter.ai). Slugs always
46
+ * carry a vendor prefix (`vendor/model`), which is how provider inference
47
+ * recognizes them. All listed models accept image input.
48
+ */
49
+ OpenRouter: {
50
+ GROK_4_5: "x-ai/grok-4.5",
51
+ KIMI_K3: "moonshotai/kimi-k3",
52
+ KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
53
+ QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
54
+ QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
40
55
  }
41
56
  };
42
57
  var DEFAULT_MODELS = {
43
58
  [Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
44
59
  [Provider.OPENAI]: Model.OpenAI.GPT_5_4_MINI,
45
- [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW
60
+ [Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
61
+ [Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
46
62
  };
47
63
  var DEFAULT_MAX_TOKENS = 4096;
48
64
  var OPENAI_REASONING_MAX_TOKENS = 16384;
49
65
  var MODEL_TO_PROVIDER = new Map([
50
66
  ...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
51
67
  ...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
52
- ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE])
68
+ ...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
69
+ ...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
53
70
  ]);
54
71
  var VALID_PROVIDERS = Object.values(Provider);
55
72
  var PROVIDER_DEFAULT_REASONING = {
56
73
  openai: "medium",
57
74
  anthropic: "off",
58
- google: "off"
75
+ google: "off",
76
+ // Varies by upstream model; the driver sends no reasoning field unless configured.
77
+ openrouter: "off"
59
78
  };
60
79
  var Content = {
61
80
  /** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
@@ -521,6 +540,13 @@ function mapEffort(level, model) {
521
540
  if (level !== "xhigh") return level;
522
541
  return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
523
542
  }
543
+ var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
544
+ var EFFORT_TO_BUDGET_TOKENS = {
545
+ low: 1024,
546
+ medium: 4096,
547
+ high: 8192,
548
+ xhigh: 16384
549
+ };
524
550
  var AnthropicDriver = class {
525
551
  client;
526
552
  model;
@@ -576,10 +602,16 @@ var AnthropicDriver = class {
576
602
  ]
577
603
  };
578
604
  if (this.reasoningEffort) {
579
- requestParams.thinking = { type: "adaptive" };
580
- requestParams.output_config = {
581
- effort: mapEffort(this.reasoningEffort, this.model)
582
- };
605
+ if (BUDGET_THINKING_MODELS.has(this.model)) {
606
+ const budgetTokens = EFFORT_TO_BUDGET_TOKENS[this.reasoningEffort];
607
+ requestParams.thinking = { type: "enabled", budget_tokens: budgetTokens };
608
+ requestParams.max_tokens = Math.max(this.maxTokens, budgetTokens + DEFAULT_MAX_TOKENS);
609
+ } else {
610
+ requestParams.thinking = { type: "adaptive" };
611
+ requestParams.output_config = {
612
+ effort: mapEffort(this.reasoningEffort, this.model)
613
+ };
614
+ }
583
615
  }
584
616
  const message = await client.messages.create(requestParams);
585
617
  const textBlock = message.content.find((block) => block.type === "text");
@@ -820,6 +852,109 @@ var OpenAIDriver = class {
820
852
  }
821
853
  };
822
854
 
855
+ // src/providers/openrouter.ts
856
+ var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
857
+ var OPENROUTER_REASONING_EFFORT = {
858
+ low: "low",
859
+ medium: "medium",
860
+ high: "high",
861
+ xhigh: "high"
862
+ };
863
+ var OpenRouterDriver = class {
864
+ client;
865
+ model;
866
+ maxTokens;
867
+ apiKeyOrEnv;
868
+ reasoningEffort;
869
+ constructor(config) {
870
+ this.model = config.model;
871
+ this.maxTokens = config.maxTokens;
872
+ this.client = null;
873
+ this.apiKeyOrEnv = config.apiKey;
874
+ this.reasoningEffort = config.reasoningEffort;
875
+ }
876
+ async getClient() {
877
+ if (this.client) return this.client;
878
+ let OpenAI;
879
+ try {
880
+ const mod = await import("openai");
881
+ OpenAI = mod.default;
882
+ } catch {
883
+ throw new VisualAIConfigError(
884
+ "OpenAI SDK not installed (required for the OpenRouter provider). Run: npm install openai"
885
+ );
886
+ }
887
+ const apiKey = this.apiKeyOrEnv ?? process.env.OPENROUTER_API_KEY;
888
+ if (!apiKey) {
889
+ throw new VisualAIAuthError(
890
+ "OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
891
+ );
892
+ }
893
+ this.client = new OpenAI({ apiKey, baseURL: OPENROUTER_BASE_URL });
894
+ return this.client;
895
+ }
896
+ async sendMessage(images, prompt, options) {
897
+ const client = await this.getClient();
898
+ const imageParts = images.map((img) => ({
899
+ type: "image_url",
900
+ image_url: { url: `data:${img.mimeType};base64,${img.base64}` }
901
+ }));
902
+ try {
903
+ const responseFormat = options?.responseSchema ? {
904
+ type: "json_schema",
905
+ json_schema: {
906
+ name: "visual_ai_response",
907
+ strict: true,
908
+ schema: options.responseSchema
909
+ }
910
+ } : { type: "json_object" };
911
+ const requestParams = {
912
+ model: this.model,
913
+ max_tokens: this.maxTokens,
914
+ response_format: responseFormat,
915
+ messages: [
916
+ {
917
+ role: "user",
918
+ content: [...imageParts, { type: "text", text: prompt }]
919
+ }
920
+ ],
921
+ // OpenRouter-specific: include token accounting in the response.
922
+ usage: { include: true }
923
+ };
924
+ if (this.reasoningEffort) {
925
+ requestParams.reasoning = { effort: OPENROUTER_REASONING_EFFORT[this.reasoningEffort] };
926
+ }
927
+ const response = await client.chat.completions.create(requestParams);
928
+ const choice = response.choices?.[0];
929
+ if (!choice?.message) {
930
+ throw new VisualAIProviderError("OpenRouter returned an empty response (no choices).");
931
+ }
932
+ const text = choice.message.content ?? "";
933
+ if (choice.finish_reason === "length") {
934
+ throw new VisualAITruncationError(
935
+ `Response truncated: OpenRouter returned finish_reason "length". The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
936
+ text,
937
+ this.maxTokens
938
+ );
939
+ }
940
+ const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
941
+ return {
942
+ text,
943
+ usage: response.usage ? {
944
+ inputTokens: response.usage.prompt_tokens,
945
+ outputTokens: response.usage.completion_tokens,
946
+ ...reasoningTokens !== void 0 && { reasoningTokens }
947
+ } : void 0
948
+ };
949
+ } catch (err) {
950
+ if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) {
951
+ throw err;
952
+ }
953
+ throw mapProviderError(err);
954
+ }
955
+ }
956
+ };
957
+
823
958
  // src/core/config.ts
824
959
  var MODEL_PREFIX_TO_PROVIDER = [
825
960
  ["claude-", "anthropic"],
@@ -832,6 +967,7 @@ var MODEL_PREFIX_TO_PROVIDER = [
832
967
  function inferProviderFromModel(model) {
833
968
  const known = MODEL_TO_PROVIDER.get(model);
834
969
  if (known) return known;
970
+ if (model.includes("/")) return "openrouter";
835
971
  const prefixMatch = MODEL_PREFIX_TO_PROVIDER.find(([prefix]) => model.startsWith(prefix));
836
972
  return prefixMatch?.[1];
837
973
  }
@@ -844,12 +980,13 @@ function resolveProvider(config) {
844
980
  const apiKeyProviderMap = [
845
981
  ["ANTHROPIC_API_KEY", "anthropic"],
846
982
  ["OPENAI_API_KEY", "openai"],
847
- ["GOOGLE_API_KEY", "google"]
983
+ ["GOOGLE_API_KEY", "google"],
984
+ ["OPENROUTER_API_KEY", "openrouter"]
848
985
  ];
849
986
  const detected = apiKeyProviderMap.find(([key]) => process.env[key]);
850
987
  if (detected) return detected[1];
851
988
  throw new VisualAIConfigError(
852
- "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY)."
989
+ "Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, OPENROUTER_API_KEY)."
853
990
  );
854
991
  }
855
992
  function parseBooleanEnv(envName, value) {
@@ -877,11 +1014,11 @@ function resolveConfig(config) {
877
1014
  }
878
1015
  const userSetMaxTokens = config.maxTokens !== void 0;
879
1016
  let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
880
- if (!userSetMaxTokens && provider === "openai" && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
1017
+ if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
881
1018
  maxTokens = OPENAI_REASONING_MAX_TOKENS;
882
1019
  if (debug) {
883
1020
  process.stderr.write(
884
- `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for OpenAI with reasoningEffort "${config.reasoningEffort}".
1021
+ `[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for provider "${provider}" with reasoningEffort "${config.reasoningEffort}".
885
1022
  `
886
1023
  );
887
1024
  }
@@ -970,10 +1107,18 @@ var PRICING_TABLE = {
970
1107
  inputPricePerToken: 0.25 / PER_MILLION,
971
1108
  outputPricePerToken: 2 / PER_MILLION
972
1109
  },
1110
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
1111
+ inputPricePerToken: 1.5 / PER_MILLION,
1112
+ outputPricePerToken: 7.5 / PER_MILLION
1113
+ },
973
1114
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH}`]: {
974
1115
  inputPricePerToken: 1.5 / PER_MILLION,
975
1116
  outputPricePerToken: 9 / PER_MILLION
976
1117
  },
1118
+ [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH_LITE}`]: {
1119
+ inputPricePerToken: 0.3 / PER_MILLION,
1120
+ outputPricePerToken: 2.5 / PER_MILLION
1121
+ },
977
1122
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_1_PRO_PREVIEW}`]: {
978
1123
  inputPricePerToken: 2 / PER_MILLION,
979
1124
  outputPricePerToken: 12 / PER_MILLION
@@ -985,6 +1130,28 @@ var PRICING_TABLE = {
985
1130
  [`${Provider.GOOGLE}:${Model.Google.GEMINI_3_FLASH_PREVIEW}`]: {
986
1131
  inputPricePerToken: 0.5 / PER_MILLION,
987
1132
  outputPricePerToken: 3 / PER_MILLION
1133
+ },
1134
+ // OpenRouter passes through upstream per-model pricing (verified 2026-07-22
1135
+ // against https://openrouter.ai/api/v1/models).
1136
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
1137
+ inputPricePerToken: 2 / PER_MILLION,
1138
+ outputPricePerToken: 6 / PER_MILLION
1139
+ },
1140
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K3}`]: {
1141
+ inputPricePerToken: 3 / PER_MILLION,
1142
+ outputPricePerToken: 15 / PER_MILLION
1143
+ },
1144
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K2_7_CODE}`]: {
1145
+ inputPricePerToken: 0.82 / PER_MILLION,
1146
+ outputPricePerToken: 3.75 / PER_MILLION
1147
+ },
1148
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
1149
+ inputPricePerToken: 0.32 / PER_MILLION,
1150
+ outputPricePerToken: 1.28 / PER_MILLION
1151
+ },
1152
+ [`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
1153
+ inputPricePerToken: 0.1875 / PER_MILLION,
1154
+ outputPricePerToken: 1.125 / PER_MILLION
988
1155
  }
989
1156
  };
990
1157
  function calculateCost(provider, model, inputTokens, outputTokens) {
@@ -1062,7 +1229,8 @@ async function timedSendMessage(driver, images, prompt, options) {
1062
1229
  import sharp from "sharp";
1063
1230
  var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
1064
1231
  Model.Google.GEMINI_3_FLASH_PREVIEW,
1065
- Model.Google.GEMINI_3_5_FLASH
1232
+ Model.Google.GEMINI_3_5_FLASH,
1233
+ Model.Google.GEMINI_3_6_FLASH
1066
1234
  ]);
1067
1235
  async function generateAiDiff(imgA, imgB, model, driver) {
1068
1236
  if (!driver.generateImage) {
@@ -1705,7 +1873,57 @@ function isVideoInput(input) {
1705
1873
  }
1706
1874
  return false;
1707
1875
  }
1876
+ function isFramesInput(input) {
1877
+ return typeof input === "object" && input !== null && !Buffer.isBuffer(input) && !(input instanceof Uint8Array) && Array.isArray(input.frames);
1878
+ }
1879
+ function isTimestampedFrameInput(frame) {
1880
+ return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
1881
+ }
1882
+ async function normalizeFrames(input) {
1883
+ const rawFrames = input.frames;
1884
+ const fps = input.fps ?? DEFAULT_FPS;
1885
+ if (rawFrames.length === 0) {
1886
+ throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
1887
+ }
1888
+ if (rawFrames.length > MAX_FRAMES_HARD_CAP) {
1889
+ throw new VisualAIVideoError(
1890
+ `frames length ${rawFrames.length} exceeds the hard cap of ${MAX_FRAMES_HARD_CAP}. Pass fewer frames or open an issue if you need a larger limit.`
1891
+ );
1892
+ }
1893
+ if (!Number.isFinite(fps) || fps <= 0) {
1894
+ throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
1895
+ }
1896
+ const frames = await Promise.all(
1897
+ rawFrames.map(async (raw, index) => {
1898
+ const timestamped = isTimestampedFrameInput(raw);
1899
+ const imageInput = timestamped ? raw.image : raw;
1900
+ const explicit = timestamped ? raw.timestampSeconds : void 0;
1901
+ const timestampSeconds = explicit ?? index / fps;
1902
+ if (!Number.isFinite(timestampSeconds) || timestampSeconds < 0) {
1903
+ throw new VisualAIVideoError(
1904
+ `Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
1905
+ );
1906
+ }
1907
+ const image = await normalizeImage(imageInput);
1908
+ return {
1909
+ data: image.data,
1910
+ mimeType: image.mimeType,
1911
+ get base64() {
1912
+ return image.base64;
1913
+ },
1914
+ timestampSeconds,
1915
+ index
1916
+ };
1917
+ })
1918
+ );
1919
+ const durationSeconds = frames.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
1920
+ await saveDebugFrames(frames);
1921
+ return { kind: "video", frames, durationSeconds };
1922
+ }
1708
1923
  async function normalizeMedia(input, videoOptions) {
1924
+ if (isFramesInput(input)) {
1925
+ return normalizeFrames(input);
1926
+ }
1709
1927
  if (isVideoInput(input)) {
1710
1928
  const { path, cleanup } = await resolveVideoToPath(input);
1711
1929
  try {
@@ -1862,7 +2080,8 @@ function toSchemaOptions(schema) {
1862
2080
  var PROVIDER_REGISTRY = {
1863
2081
  anthropic: (config) => new AnthropicDriver(config),
1864
2082
  openai: (config) => new OpenAIDriver(config),
1865
- google: (config) => new GoogleDriver(config)
2083
+ google: (config) => new GoogleDriver(config),
2084
+ openrouter: (config) => new OpenRouterDriver(config)
1866
2085
  };
1867
2086
  function createDriver(provider, config) {
1868
2087
  return PROVIDER_REGISTRY[provider](config);