visual-ai-assertions 0.13.1 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +56 -11
- package/dist/index.cjs +233 -14
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +59 -9
- package/dist/index.d.ts +59 -9
- package/dist/index.js +233 -14
- package/dist/index.js.map +1 -1
- package/package.json +6 -2
package/dist/index.d.cts
CHANGED
|
@@ -14,6 +14,7 @@ declare const Provider: {
|
|
|
14
14
|
readonly ANTHROPIC: "anthropic";
|
|
15
15
|
readonly OPENAI: "openai";
|
|
16
16
|
readonly GOOGLE: "google";
|
|
17
|
+
readonly OPENROUTER: "openrouter";
|
|
17
18
|
};
|
|
18
19
|
/** Known model names grouped by provider. */
|
|
19
20
|
declare const Model: {
|
|
@@ -39,19 +40,34 @@ declare const Model: {
|
|
|
39
40
|
readonly GPT_5_MINI: "gpt-5-mini";
|
|
40
41
|
};
|
|
41
42
|
readonly Google: {
|
|
43
|
+
readonly GEMINI_3_6_FLASH: "gemini-3.6-flash";
|
|
42
44
|
readonly GEMINI_3_5_FLASH: "gemini-3.5-flash";
|
|
45
|
+
readonly GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite";
|
|
43
46
|
readonly GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview";
|
|
44
47
|
readonly GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite";
|
|
45
48
|
readonly GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview";
|
|
46
49
|
};
|
|
50
|
+
/**
|
|
51
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
52
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
53
|
+
* recognizes them. All listed models accept image input.
|
|
54
|
+
*/
|
|
55
|
+
readonly OpenRouter: {
|
|
56
|
+
readonly GROK_4_5: "x-ai/grok-4.5";
|
|
57
|
+
readonly KIMI_K3: "moonshotai/kimi-k3";
|
|
58
|
+
readonly KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code";
|
|
59
|
+
readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
|
|
60
|
+
readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
|
|
61
|
+
};
|
|
47
62
|
};
|
|
48
63
|
/** Union of all built-in model name literals exposed by `Model`. */
|
|
49
|
-
type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google];
|
|
64
|
+
type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google] | (typeof Model.OpenRouter)[keyof typeof Model.OpenRouter];
|
|
50
65
|
/** Default model selection used when a caller omits `config.model`. */
|
|
51
66
|
declare const DEFAULT_MODELS: {
|
|
52
67
|
readonly anthropic: "claude-sonnet-4-6";
|
|
53
68
|
readonly openai: "gpt-5.4-mini";
|
|
54
69
|
readonly google: "gemini-3-flash-preview";
|
|
70
|
+
readonly openrouter: "qwen/qwen3.6-flash";
|
|
55
71
|
};
|
|
56
72
|
/** Built-in content checks available through `client.content()`. */
|
|
57
73
|
declare const Content: {
|
|
@@ -493,12 +509,42 @@ type ImageInput = Buffer | Uint8Array | string;
|
|
|
493
509
|
* an image or a video.
|
|
494
510
|
*/
|
|
495
511
|
type MediaInput = ImageInput;
|
|
512
|
+
/**
|
|
513
|
+
* A single pre-sampled frame with an explicit timestamp. Use this shape when
|
|
514
|
+
* you want to control the timestamp the model sees; otherwise pass a bare
|
|
515
|
+
* `ImageInput` and the timestamp is derived from `fps` and the frame's position.
|
|
516
|
+
*/
|
|
517
|
+
interface TimestampedFrameInput {
|
|
518
|
+
/** Image source for the frame (buffer, data URL, base64, file path, or URL). */
|
|
519
|
+
image: ImageInput;
|
|
520
|
+
/** Timestamp in seconds, from the start of the sequence. Derived from `fps` when omitted. */
|
|
521
|
+
timestampSeconds?: number;
|
|
522
|
+
}
|
|
523
|
+
/**
|
|
524
|
+
* Pre-sampled frames supplied directly instead of a video file. Useful when you
|
|
525
|
+
* already have an array of screenshots and cannot (or would rather not) pass a
|
|
526
|
+
* video — this path never loads ffmpeg. Passed to `check()` / `ask()` in place
|
|
527
|
+
* of a `MediaInput`; downstream handling is identical to a sampled video.
|
|
528
|
+
*/
|
|
529
|
+
interface FramesInput {
|
|
530
|
+
/**
|
|
531
|
+
* Ordered frames, earliest first. Each entry is an image source or a
|
|
532
|
+
* `{ image, timestampSeconds }` pair. Must be non-empty and is capped at the
|
|
533
|
+
* same per-request frame limit as video sampling (60).
|
|
534
|
+
*/
|
|
535
|
+
frames: ReadonlyArray<ImageInput | TimestampedFrameInput>;
|
|
536
|
+
/**
|
|
537
|
+
* Frame rate used to derive timestamps for frames that don't carry their own
|
|
538
|
+
* `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
|
|
539
|
+
*/
|
|
540
|
+
fps?: number;
|
|
541
|
+
}
|
|
496
542
|
/** Supported image MIME types accepted by all providers. */
|
|
497
543
|
type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
|
|
498
544
|
/** Supported video MIME types the client can accept and sample frames from. */
|
|
499
545
|
type SupportedVideoMimeType = "video/mp4" | "video/webm" | "video/quicktime" | "video/x-matroska";
|
|
500
546
|
/** Supported provider identifiers. */
|
|
501
|
-
type ProviderName = "anthropic" | "openai" | "google";
|
|
547
|
+
type ProviderName = "anthropic" | "openai" | "google" | "openrouter";
|
|
502
548
|
/**
|
|
503
549
|
* Configuration for creating a visual AI client.
|
|
504
550
|
*
|
|
@@ -633,9 +679,11 @@ interface VisualAIClient {
|
|
|
633
679
|
* frames automatically; statements pass if they are true at any sampled
|
|
634
680
|
* frame, and each statement result includes the timestamp where it
|
|
635
681
|
* matched. The `frames` metadata on the result reports which timestamps
|
|
636
|
-
* the model saw.
|
|
682
|
+
* the model saw. Pass a `FramesInput` (`{ frames, fps? }`) to supply
|
|
683
|
+
* pre-sampled frames directly — handled identically to a video timeline but
|
|
684
|
+
* without loading ffmpeg.
|
|
637
685
|
*
|
|
638
|
-
* @param input Image or video source as a buffer, URL, file path, or base64 string.
|
|
686
|
+
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
639
687
|
* @param statements One or more statements to validate against the input.
|
|
640
688
|
* @param options Optional additional instructions and video sampling overrides.
|
|
641
689
|
* @returns A structured result describing pass/fail, issues, and statement reasoning.
|
|
@@ -658,15 +706,17 @@ interface VisualAIClient {
|
|
|
658
706
|
* console.log(result.statements[0].timestampSeconds); // e.g. 3.5
|
|
659
707
|
* ```
|
|
660
708
|
*/
|
|
661
|
-
check(input: MediaInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
|
|
709
|
+
check(input: MediaInput | FramesInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
|
|
662
710
|
/**
|
|
663
711
|
* Asks an open-ended question about an image or video and returns a structured summary.
|
|
664
712
|
*
|
|
665
713
|
* Video inputs are sampled into frames and analyzed as a chronological
|
|
666
714
|
* timeline. The result's `frameReferences` array surfaces which frames the
|
|
667
|
-
* model relied on for its answer.
|
|
715
|
+
* model relied on for its answer. Pass a `FramesInput` (`{ frames, fps? }`)
|
|
716
|
+
* to supply pre-sampled frames directly — handled identically to a video
|
|
717
|
+
* timeline but without loading ffmpeg.
|
|
668
718
|
*
|
|
669
|
-
* @param input Image or video source as a buffer, URL, file path, or base64 string.
|
|
719
|
+
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
670
720
|
* @param prompt Prompt describing what to inspect in the input.
|
|
671
721
|
* @param options Optional additional instructions and video sampling overrides.
|
|
672
722
|
* @returns A summary with any detected issues.
|
|
@@ -678,7 +728,7 @@ interface VisualAIClient {
|
|
|
678
728
|
* const result = await client.ask(screenshot, "What looks visually broken on this page?");
|
|
679
729
|
* ```
|
|
680
730
|
*/
|
|
681
|
-
ask(input: MediaInput, prompt: string, options?: AskOptions): Promise<AskResult>;
|
|
731
|
+
ask(input: MediaInput | FramesInput, prompt: string, options?: AskOptions): Promise<AskResult>;
|
|
682
732
|
/**
|
|
683
733
|
* Compares two images and reports meaningful visual differences.
|
|
684
734
|
*
|
|
@@ -1055,4 +1105,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
|
|
|
1055
1105
|
*/
|
|
1056
1106
|
declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
|
|
1057
1107
|
|
|
1058
|
-
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
|
1108
|
+
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
package/dist/index.d.ts
CHANGED
|
@@ -14,6 +14,7 @@ declare const Provider: {
|
|
|
14
14
|
readonly ANTHROPIC: "anthropic";
|
|
15
15
|
readonly OPENAI: "openai";
|
|
16
16
|
readonly GOOGLE: "google";
|
|
17
|
+
readonly OPENROUTER: "openrouter";
|
|
17
18
|
};
|
|
18
19
|
/** Known model names grouped by provider. */
|
|
19
20
|
declare const Model: {
|
|
@@ -39,19 +40,34 @@ declare const Model: {
|
|
|
39
40
|
readonly GPT_5_MINI: "gpt-5-mini";
|
|
40
41
|
};
|
|
41
42
|
readonly Google: {
|
|
43
|
+
readonly GEMINI_3_6_FLASH: "gemini-3.6-flash";
|
|
42
44
|
readonly GEMINI_3_5_FLASH: "gemini-3.5-flash";
|
|
45
|
+
readonly GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite";
|
|
43
46
|
readonly GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview";
|
|
44
47
|
readonly GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite";
|
|
45
48
|
readonly GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview";
|
|
46
49
|
};
|
|
50
|
+
/**
|
|
51
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
52
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
53
|
+
* recognizes them. All listed models accept image input.
|
|
54
|
+
*/
|
|
55
|
+
readonly OpenRouter: {
|
|
56
|
+
readonly GROK_4_5: "x-ai/grok-4.5";
|
|
57
|
+
readonly KIMI_K3: "moonshotai/kimi-k3";
|
|
58
|
+
readonly KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code";
|
|
59
|
+
readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
|
|
60
|
+
readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
|
|
61
|
+
};
|
|
47
62
|
};
|
|
48
63
|
/** Union of all built-in model name literals exposed by `Model`. */
|
|
49
|
-
type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google];
|
|
64
|
+
type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google] | (typeof Model.OpenRouter)[keyof typeof Model.OpenRouter];
|
|
50
65
|
/** Default model selection used when a caller omits `config.model`. */
|
|
51
66
|
declare const DEFAULT_MODELS: {
|
|
52
67
|
readonly anthropic: "claude-sonnet-4-6";
|
|
53
68
|
readonly openai: "gpt-5.4-mini";
|
|
54
69
|
readonly google: "gemini-3-flash-preview";
|
|
70
|
+
readonly openrouter: "qwen/qwen3.6-flash";
|
|
55
71
|
};
|
|
56
72
|
/** Built-in content checks available through `client.content()`. */
|
|
57
73
|
declare const Content: {
|
|
@@ -493,12 +509,42 @@ type ImageInput = Buffer | Uint8Array | string;
|
|
|
493
509
|
* an image or a video.
|
|
494
510
|
*/
|
|
495
511
|
type MediaInput = ImageInput;
|
|
512
|
+
/**
|
|
513
|
+
* A single pre-sampled frame with an explicit timestamp. Use this shape when
|
|
514
|
+
* you want to control the timestamp the model sees; otherwise pass a bare
|
|
515
|
+
* `ImageInput` and the timestamp is derived from `fps` and the frame's position.
|
|
516
|
+
*/
|
|
517
|
+
interface TimestampedFrameInput {
|
|
518
|
+
/** Image source for the frame (buffer, data URL, base64, file path, or URL). */
|
|
519
|
+
image: ImageInput;
|
|
520
|
+
/** Timestamp in seconds, from the start of the sequence. Derived from `fps` when omitted. */
|
|
521
|
+
timestampSeconds?: number;
|
|
522
|
+
}
|
|
523
|
+
/**
|
|
524
|
+
* Pre-sampled frames supplied directly instead of a video file. Useful when you
|
|
525
|
+
* already have an array of screenshots and cannot (or would rather not) pass a
|
|
526
|
+
* video — this path never loads ffmpeg. Passed to `check()` / `ask()` in place
|
|
527
|
+
* of a `MediaInput`; downstream handling is identical to a sampled video.
|
|
528
|
+
*/
|
|
529
|
+
interface FramesInput {
|
|
530
|
+
/**
|
|
531
|
+
* Ordered frames, earliest first. Each entry is an image source or a
|
|
532
|
+
* `{ image, timestampSeconds }` pair. Must be non-empty and is capped at the
|
|
533
|
+
* same per-request frame limit as video sampling (60).
|
|
534
|
+
*/
|
|
535
|
+
frames: ReadonlyArray<ImageInput | TimestampedFrameInput>;
|
|
536
|
+
/**
|
|
537
|
+
* Frame rate used to derive timestamps for frames that don't carry their own
|
|
538
|
+
* `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
|
|
539
|
+
*/
|
|
540
|
+
fps?: number;
|
|
541
|
+
}
|
|
496
542
|
/** Supported image MIME types accepted by all providers. */
|
|
497
543
|
type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
|
|
498
544
|
/** Supported video MIME types the client can accept and sample frames from. */
|
|
499
545
|
type SupportedVideoMimeType = "video/mp4" | "video/webm" | "video/quicktime" | "video/x-matroska";
|
|
500
546
|
/** Supported provider identifiers. */
|
|
501
|
-
type ProviderName = "anthropic" | "openai" | "google";
|
|
547
|
+
type ProviderName = "anthropic" | "openai" | "google" | "openrouter";
|
|
502
548
|
/**
|
|
503
549
|
* Configuration for creating a visual AI client.
|
|
504
550
|
*
|
|
@@ -633,9 +679,11 @@ interface VisualAIClient {
|
|
|
633
679
|
* frames automatically; statements pass if they are true at any sampled
|
|
634
680
|
* frame, and each statement result includes the timestamp where it
|
|
635
681
|
* matched. The `frames` metadata on the result reports which timestamps
|
|
636
|
-
* the model saw.
|
|
682
|
+
* the model saw. Pass a `FramesInput` (`{ frames, fps? }`) to supply
|
|
683
|
+
* pre-sampled frames directly — handled identically to a video timeline but
|
|
684
|
+
* without loading ffmpeg.
|
|
637
685
|
*
|
|
638
|
-
* @param input Image or video source as a buffer, URL, file path, or base64 string.
|
|
686
|
+
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
639
687
|
* @param statements One or more statements to validate against the input.
|
|
640
688
|
* @param options Optional additional instructions and video sampling overrides.
|
|
641
689
|
* @returns A structured result describing pass/fail, issues, and statement reasoning.
|
|
@@ -658,15 +706,17 @@ interface VisualAIClient {
|
|
|
658
706
|
* console.log(result.statements[0].timestampSeconds); // e.g. 3.5
|
|
659
707
|
* ```
|
|
660
708
|
*/
|
|
661
|
-
check(input: MediaInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
|
|
709
|
+
check(input: MediaInput | FramesInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
|
|
662
710
|
/**
|
|
663
711
|
* Asks an open-ended question about an image or video and returns a structured summary.
|
|
664
712
|
*
|
|
665
713
|
* Video inputs are sampled into frames and analyzed as a chronological
|
|
666
714
|
* timeline. The result's `frameReferences` array surfaces which frames the
|
|
667
|
-
* model relied on for its answer.
|
|
715
|
+
* model relied on for its answer. Pass a `FramesInput` (`{ frames, fps? }`)
|
|
716
|
+
* to supply pre-sampled frames directly — handled identically to a video
|
|
717
|
+
* timeline but without loading ffmpeg.
|
|
668
718
|
*
|
|
669
|
-
* @param input Image or video source as a buffer, URL, file path, or base64 string.
|
|
719
|
+
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
670
720
|
* @param prompt Prompt describing what to inspect in the input.
|
|
671
721
|
* @param options Optional additional instructions and video sampling overrides.
|
|
672
722
|
* @returns A summary with any detected issues.
|
|
@@ -678,7 +728,7 @@ interface VisualAIClient {
|
|
|
678
728
|
* const result = await client.ask(screenshot, "What looks visually broken on this page?");
|
|
679
729
|
* ```
|
|
680
730
|
*/
|
|
681
|
-
ask(input: MediaInput, prompt: string, options?: AskOptions): Promise<AskResult>;
|
|
731
|
+
ask(input: MediaInput | FramesInput, prompt: string, options?: AskOptions): Promise<AskResult>;
|
|
682
732
|
/**
|
|
683
733
|
* Compares two images and reports meaningful visual differences.
|
|
684
734
|
*
|
|
@@ -1055,4 +1105,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
|
|
|
1055
1105
|
*/
|
|
1056
1106
|
declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
|
|
1057
1107
|
|
|
1058
|
-
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
|
1108
|
+
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
package/dist/index.js
CHANGED
|
@@ -8,7 +8,8 @@ var ReasoningEffort = {
|
|
|
8
8
|
var Provider = {
|
|
9
9
|
ANTHROPIC: "anthropic",
|
|
10
10
|
OPENAI: "openai",
|
|
11
|
-
GOOGLE: "google"
|
|
11
|
+
GOOGLE: "google",
|
|
12
|
+
OPENROUTER: "openrouter"
|
|
12
13
|
};
|
|
13
14
|
var Model = {
|
|
14
15
|
Anthropic: {
|
|
@@ -33,29 +34,47 @@ var Model = {
|
|
|
33
34
|
GPT_5_MINI: "gpt-5-mini"
|
|
34
35
|
},
|
|
35
36
|
Google: {
|
|
37
|
+
GEMINI_3_6_FLASH: "gemini-3.6-flash",
|
|
36
38
|
GEMINI_3_5_FLASH: "gemini-3.5-flash",
|
|
39
|
+
GEMINI_3_5_FLASH_LITE: "gemini-3.5-flash-lite",
|
|
37
40
|
GEMINI_3_1_PRO_PREVIEW: "gemini-3.1-pro-preview",
|
|
38
41
|
GEMINI_3_1_FLASH_LITE: "gemini-3.1-flash-lite",
|
|
39
42
|
GEMINI_3_FLASH_PREVIEW: "gemini-3-flash-preview"
|
|
43
|
+
},
|
|
44
|
+
/**
|
|
45
|
+
* Models routed through OpenRouter (https://openrouter.ai). Slugs always
|
|
46
|
+
* carry a vendor prefix (`vendor/model`), which is how provider inference
|
|
47
|
+
* recognizes them. All listed models accept image input.
|
|
48
|
+
*/
|
|
49
|
+
OpenRouter: {
|
|
50
|
+
GROK_4_5: "x-ai/grok-4.5",
|
|
51
|
+
KIMI_K3: "moonshotai/kimi-k3",
|
|
52
|
+
KIMI_K2_7_CODE: "moonshotai/kimi-k2.7-code",
|
|
53
|
+
QWEN_3_7_PLUS: "qwen/qwen3.7-plus",
|
|
54
|
+
QWEN_3_6_FLASH: "qwen/qwen3.6-flash"
|
|
40
55
|
}
|
|
41
56
|
};
|
|
42
57
|
var DEFAULT_MODELS = {
|
|
43
58
|
[Provider.ANTHROPIC]: Model.Anthropic.SONNET_4_6,
|
|
44
59
|
[Provider.OPENAI]: Model.OpenAI.GPT_5_4_MINI,
|
|
45
|
-
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW
|
|
60
|
+
[Provider.GOOGLE]: Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
61
|
+
[Provider.OPENROUTER]: Model.OpenRouter.QWEN_3_6_FLASH
|
|
46
62
|
};
|
|
47
63
|
var DEFAULT_MAX_TOKENS = 4096;
|
|
48
64
|
var OPENAI_REASONING_MAX_TOKENS = 16384;
|
|
49
65
|
var MODEL_TO_PROVIDER = new Map([
|
|
50
66
|
...Object.values(Model.Anthropic).map((m) => [m, Provider.ANTHROPIC]),
|
|
51
67
|
...Object.values(Model.OpenAI).map((m) => [m, Provider.OPENAI]),
|
|
52
|
-
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE])
|
|
68
|
+
...Object.values(Model.Google).map((m) => [m, Provider.GOOGLE]),
|
|
69
|
+
...Object.values(Model.OpenRouter).map((m) => [m, Provider.OPENROUTER])
|
|
53
70
|
]);
|
|
54
71
|
var VALID_PROVIDERS = Object.values(Provider);
|
|
55
72
|
var PROVIDER_DEFAULT_REASONING = {
|
|
56
73
|
openai: "medium",
|
|
57
74
|
anthropic: "off",
|
|
58
|
-
google: "off"
|
|
75
|
+
google: "off",
|
|
76
|
+
// Varies by upstream model; the driver sends no reasoning field unless configured.
|
|
77
|
+
openrouter: "off"
|
|
59
78
|
};
|
|
60
79
|
var Content = {
|
|
61
80
|
/** Detects Lorem ipsum, TODO, TBD, and similar placeholder text */
|
|
@@ -521,6 +540,13 @@ function mapEffort(level, model) {
|
|
|
521
540
|
if (level !== "xhigh") return level;
|
|
522
541
|
return XHIGH_CAPABLE_MODELS.has(model) ? "xhigh" : "max";
|
|
523
542
|
}
|
|
543
|
+
var BUDGET_THINKING_MODELS = /* @__PURE__ */ new Set([Model.Anthropic.HAIKU_4_5]);
|
|
544
|
+
var EFFORT_TO_BUDGET_TOKENS = {
|
|
545
|
+
low: 1024,
|
|
546
|
+
medium: 4096,
|
|
547
|
+
high: 8192,
|
|
548
|
+
xhigh: 16384
|
|
549
|
+
};
|
|
524
550
|
var AnthropicDriver = class {
|
|
525
551
|
client;
|
|
526
552
|
model;
|
|
@@ -576,10 +602,16 @@ var AnthropicDriver = class {
|
|
|
576
602
|
]
|
|
577
603
|
};
|
|
578
604
|
if (this.reasoningEffort) {
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
605
|
+
if (BUDGET_THINKING_MODELS.has(this.model)) {
|
|
606
|
+
const budgetTokens = EFFORT_TO_BUDGET_TOKENS[this.reasoningEffort];
|
|
607
|
+
requestParams.thinking = { type: "enabled", budget_tokens: budgetTokens };
|
|
608
|
+
requestParams.max_tokens = Math.max(this.maxTokens, budgetTokens + DEFAULT_MAX_TOKENS);
|
|
609
|
+
} else {
|
|
610
|
+
requestParams.thinking = { type: "adaptive" };
|
|
611
|
+
requestParams.output_config = {
|
|
612
|
+
effort: mapEffort(this.reasoningEffort, this.model)
|
|
613
|
+
};
|
|
614
|
+
}
|
|
583
615
|
}
|
|
584
616
|
const message = await client.messages.create(requestParams);
|
|
585
617
|
const textBlock = message.content.find((block) => block.type === "text");
|
|
@@ -820,6 +852,109 @@ var OpenAIDriver = class {
|
|
|
820
852
|
}
|
|
821
853
|
};
|
|
822
854
|
|
|
855
|
+
// src/providers/openrouter.ts
|
|
856
|
+
var OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
|
|
857
|
+
var OPENROUTER_REASONING_EFFORT = {
|
|
858
|
+
low: "low",
|
|
859
|
+
medium: "medium",
|
|
860
|
+
high: "high",
|
|
861
|
+
xhigh: "high"
|
|
862
|
+
};
|
|
863
|
+
var OpenRouterDriver = class {
|
|
864
|
+
client;
|
|
865
|
+
model;
|
|
866
|
+
maxTokens;
|
|
867
|
+
apiKeyOrEnv;
|
|
868
|
+
reasoningEffort;
|
|
869
|
+
constructor(config) {
|
|
870
|
+
this.model = config.model;
|
|
871
|
+
this.maxTokens = config.maxTokens;
|
|
872
|
+
this.client = null;
|
|
873
|
+
this.apiKeyOrEnv = config.apiKey;
|
|
874
|
+
this.reasoningEffort = config.reasoningEffort;
|
|
875
|
+
}
|
|
876
|
+
async getClient() {
|
|
877
|
+
if (this.client) return this.client;
|
|
878
|
+
let OpenAI;
|
|
879
|
+
try {
|
|
880
|
+
const mod = await import("openai");
|
|
881
|
+
OpenAI = mod.default;
|
|
882
|
+
} catch {
|
|
883
|
+
throw new VisualAIConfigError(
|
|
884
|
+
"OpenAI SDK not installed (required for the OpenRouter provider). Run: npm install openai"
|
|
885
|
+
);
|
|
886
|
+
}
|
|
887
|
+
const apiKey = this.apiKeyOrEnv ?? process.env.OPENROUTER_API_KEY;
|
|
888
|
+
if (!apiKey) {
|
|
889
|
+
throw new VisualAIAuthError(
|
|
890
|
+
"OpenRouter API key not found. Set OPENROUTER_API_KEY or pass apiKey in config."
|
|
891
|
+
);
|
|
892
|
+
}
|
|
893
|
+
this.client = new OpenAI({ apiKey, baseURL: OPENROUTER_BASE_URL });
|
|
894
|
+
return this.client;
|
|
895
|
+
}
|
|
896
|
+
async sendMessage(images, prompt, options) {
|
|
897
|
+
const client = await this.getClient();
|
|
898
|
+
const imageParts = images.map((img) => ({
|
|
899
|
+
type: "image_url",
|
|
900
|
+
image_url: { url: `data:${img.mimeType};base64,${img.base64}` }
|
|
901
|
+
}));
|
|
902
|
+
try {
|
|
903
|
+
const responseFormat = options?.responseSchema ? {
|
|
904
|
+
type: "json_schema",
|
|
905
|
+
json_schema: {
|
|
906
|
+
name: "visual_ai_response",
|
|
907
|
+
strict: true,
|
|
908
|
+
schema: options.responseSchema
|
|
909
|
+
}
|
|
910
|
+
} : { type: "json_object" };
|
|
911
|
+
const requestParams = {
|
|
912
|
+
model: this.model,
|
|
913
|
+
max_tokens: this.maxTokens,
|
|
914
|
+
response_format: responseFormat,
|
|
915
|
+
messages: [
|
|
916
|
+
{
|
|
917
|
+
role: "user",
|
|
918
|
+
content: [...imageParts, { type: "text", text: prompt }]
|
|
919
|
+
}
|
|
920
|
+
],
|
|
921
|
+
// OpenRouter-specific: include token accounting in the response.
|
|
922
|
+
usage: { include: true }
|
|
923
|
+
};
|
|
924
|
+
if (this.reasoningEffort) {
|
|
925
|
+
requestParams.reasoning = { effort: OPENROUTER_REASONING_EFFORT[this.reasoningEffort] };
|
|
926
|
+
}
|
|
927
|
+
const response = await client.chat.completions.create(requestParams);
|
|
928
|
+
const choice = response.choices?.[0];
|
|
929
|
+
if (!choice?.message) {
|
|
930
|
+
throw new VisualAIProviderError("OpenRouter returned an empty response (no choices).");
|
|
931
|
+
}
|
|
932
|
+
const text = choice.message.content ?? "";
|
|
933
|
+
if (choice.finish_reason === "length") {
|
|
934
|
+
throw new VisualAITruncationError(
|
|
935
|
+
`Response truncated: OpenRouter returned finish_reason "length". The model exhausted the output token budget (${this.maxTokens} tokens). This commonly happens with higher reasoning effort levels. Increase maxTokens in your config (e.g., maxTokens: 16384) or lower reasoningEffort.`,
|
|
936
|
+
text,
|
|
937
|
+
this.maxTokens
|
|
938
|
+
);
|
|
939
|
+
}
|
|
940
|
+
const reasoningTokens = response.usage?.completion_tokens_details?.reasoning_tokens;
|
|
941
|
+
return {
|
|
942
|
+
text,
|
|
943
|
+
usage: response.usage ? {
|
|
944
|
+
inputTokens: response.usage.prompt_tokens,
|
|
945
|
+
outputTokens: response.usage.completion_tokens,
|
|
946
|
+
...reasoningTokens !== void 0 && { reasoningTokens }
|
|
947
|
+
} : void 0
|
|
948
|
+
};
|
|
949
|
+
} catch (err) {
|
|
950
|
+
if (err instanceof VisualAITruncationError || err instanceof VisualAIProviderError) {
|
|
951
|
+
throw err;
|
|
952
|
+
}
|
|
953
|
+
throw mapProviderError(err);
|
|
954
|
+
}
|
|
955
|
+
}
|
|
956
|
+
};
|
|
957
|
+
|
|
823
958
|
// src/core/config.ts
|
|
824
959
|
var MODEL_PREFIX_TO_PROVIDER = [
|
|
825
960
|
["claude-", "anthropic"],
|
|
@@ -832,6 +967,7 @@ var MODEL_PREFIX_TO_PROVIDER = [
|
|
|
832
967
|
function inferProviderFromModel(model) {
|
|
833
968
|
const known = MODEL_TO_PROVIDER.get(model);
|
|
834
969
|
if (known) return known;
|
|
970
|
+
if (model.includes("/")) return "openrouter";
|
|
835
971
|
const prefixMatch = MODEL_PREFIX_TO_PROVIDER.find(([prefix]) => model.startsWith(prefix));
|
|
836
972
|
return prefixMatch?.[1];
|
|
837
973
|
}
|
|
@@ -844,12 +980,13 @@ function resolveProvider(config) {
|
|
|
844
980
|
const apiKeyProviderMap = [
|
|
845
981
|
["ANTHROPIC_API_KEY", "anthropic"],
|
|
846
982
|
["OPENAI_API_KEY", "openai"],
|
|
847
|
-
["GOOGLE_API_KEY", "google"]
|
|
983
|
+
["GOOGLE_API_KEY", "google"],
|
|
984
|
+
["OPENROUTER_API_KEY", "openrouter"]
|
|
848
985
|
];
|
|
849
986
|
const detected = apiKeyProviderMap.find(([key]) => process.env[key]);
|
|
850
987
|
if (detected) return detected[1];
|
|
851
988
|
throw new VisualAIConfigError(
|
|
852
|
-
"Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY)."
|
|
989
|
+
"Cannot determine provider. Set a model name (config or VISUAL_AI_MODEL) or an API key env variable (ANTHROPIC_API_KEY, OPENAI_API_KEY, GOOGLE_API_KEY, OPENROUTER_API_KEY)."
|
|
853
990
|
);
|
|
854
991
|
}
|
|
855
992
|
function parseBooleanEnv(envName, value) {
|
|
@@ -877,11 +1014,11 @@ function resolveConfig(config) {
|
|
|
877
1014
|
}
|
|
878
1015
|
const userSetMaxTokens = config.maxTokens !== void 0;
|
|
879
1016
|
let maxTokens = config.maxTokens ?? DEFAULT_MAX_TOKENS;
|
|
880
|
-
if (!userSetMaxTokens && provider === "openai" && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
|
|
1017
|
+
if (!userSetMaxTokens && (provider === "openai" || provider === "openrouter") && (config.reasoningEffort === "high" || config.reasoningEffort === "xhigh")) {
|
|
881
1018
|
maxTokens = OPENAI_REASONING_MAX_TOKENS;
|
|
882
1019
|
if (debug) {
|
|
883
1020
|
process.stderr.write(
|
|
884
|
-
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for
|
|
1021
|
+
`[visual-ai-assertions] Auto-increased maxTokens from ${DEFAULT_MAX_TOKENS} to ${OPENAI_REASONING_MAX_TOKENS} for provider "${provider}" with reasoningEffort "${config.reasoningEffort}".
|
|
885
1022
|
`
|
|
886
1023
|
);
|
|
887
1024
|
}
|
|
@@ -970,10 +1107,18 @@ var PRICING_TABLE = {
|
|
|
970
1107
|
inputPricePerToken: 0.25 / PER_MILLION,
|
|
971
1108
|
outputPricePerToken: 2 / PER_MILLION
|
|
972
1109
|
},
|
|
1110
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_6_FLASH}`]: {
|
|
1111
|
+
inputPricePerToken: 1.5 / PER_MILLION,
|
|
1112
|
+
outputPricePerToken: 7.5 / PER_MILLION
|
|
1113
|
+
},
|
|
973
1114
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH}`]: {
|
|
974
1115
|
inputPricePerToken: 1.5 / PER_MILLION,
|
|
975
1116
|
outputPricePerToken: 9 / PER_MILLION
|
|
976
1117
|
},
|
|
1118
|
+
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_5_FLASH_LITE}`]: {
|
|
1119
|
+
inputPricePerToken: 0.3 / PER_MILLION,
|
|
1120
|
+
outputPricePerToken: 2.5 / PER_MILLION
|
|
1121
|
+
},
|
|
977
1122
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_1_PRO_PREVIEW}`]: {
|
|
978
1123
|
inputPricePerToken: 2 / PER_MILLION,
|
|
979
1124
|
outputPricePerToken: 12 / PER_MILLION
|
|
@@ -985,6 +1130,28 @@ var PRICING_TABLE = {
|
|
|
985
1130
|
[`${Provider.GOOGLE}:${Model.Google.GEMINI_3_FLASH_PREVIEW}`]: {
|
|
986
1131
|
inputPricePerToken: 0.5 / PER_MILLION,
|
|
987
1132
|
outputPricePerToken: 3 / PER_MILLION
|
|
1133
|
+
},
|
|
1134
|
+
// OpenRouter passes through upstream per-model pricing (verified 2026-07-22
|
|
1135
|
+
// against https://openrouter.ai/api/v1/models).
|
|
1136
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.GROK_4_5}`]: {
|
|
1137
|
+
inputPricePerToken: 2 / PER_MILLION,
|
|
1138
|
+
outputPricePerToken: 6 / PER_MILLION
|
|
1139
|
+
},
|
|
1140
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K3}`]: {
|
|
1141
|
+
inputPricePerToken: 3 / PER_MILLION,
|
|
1142
|
+
outputPricePerToken: 15 / PER_MILLION
|
|
1143
|
+
},
|
|
1144
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.KIMI_K2_7_CODE}`]: {
|
|
1145
|
+
inputPricePerToken: 0.82 / PER_MILLION,
|
|
1146
|
+
outputPricePerToken: 3.75 / PER_MILLION
|
|
1147
|
+
},
|
|
1148
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_7_PLUS}`]: {
|
|
1149
|
+
inputPricePerToken: 0.32 / PER_MILLION,
|
|
1150
|
+
outputPricePerToken: 1.28 / PER_MILLION
|
|
1151
|
+
},
|
|
1152
|
+
[`${Provider.OPENROUTER}:${Model.OpenRouter.QWEN_3_6_FLASH}`]: {
|
|
1153
|
+
inputPricePerToken: 0.1875 / PER_MILLION,
|
|
1154
|
+
outputPricePerToken: 1.125 / PER_MILLION
|
|
988
1155
|
}
|
|
989
1156
|
};
|
|
990
1157
|
function calculateCost(provider, model, inputTokens, outputTokens) {
|
|
@@ -1062,7 +1229,8 @@ async function timedSendMessage(driver, images, prompt, options) {
|
|
|
1062
1229
|
import sharp from "sharp";
|
|
1063
1230
|
var DIFF_ALLOWED_MODELS = /* @__PURE__ */ new Set([
|
|
1064
1231
|
Model.Google.GEMINI_3_FLASH_PREVIEW,
|
|
1065
|
-
Model.Google.GEMINI_3_5_FLASH
|
|
1232
|
+
Model.Google.GEMINI_3_5_FLASH,
|
|
1233
|
+
Model.Google.GEMINI_3_6_FLASH
|
|
1066
1234
|
]);
|
|
1067
1235
|
async function generateAiDiff(imgA, imgB, model, driver) {
|
|
1068
1236
|
if (!driver.generateImage) {
|
|
@@ -1705,7 +1873,57 @@ function isVideoInput(input) {
|
|
|
1705
1873
|
}
|
|
1706
1874
|
return false;
|
|
1707
1875
|
}
|
|
1876
|
+
function isFramesInput(input) {
|
|
1877
|
+
return typeof input === "object" && input !== null && !Buffer.isBuffer(input) && !(input instanceof Uint8Array) && Array.isArray(input.frames);
|
|
1878
|
+
}
|
|
1879
|
+
function isTimestampedFrameInput(frame) {
|
|
1880
|
+
return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
|
|
1881
|
+
}
|
|
1882
|
+
async function normalizeFrames(input) {
|
|
1883
|
+
const rawFrames = input.frames;
|
|
1884
|
+
const fps = input.fps ?? DEFAULT_FPS;
|
|
1885
|
+
if (rawFrames.length === 0) {
|
|
1886
|
+
throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
|
|
1887
|
+
}
|
|
1888
|
+
if (rawFrames.length > MAX_FRAMES_HARD_CAP) {
|
|
1889
|
+
throw new VisualAIVideoError(
|
|
1890
|
+
`frames length ${rawFrames.length} exceeds the hard cap of ${MAX_FRAMES_HARD_CAP}. Pass fewer frames or open an issue if you need a larger limit.`
|
|
1891
|
+
);
|
|
1892
|
+
}
|
|
1893
|
+
if (!Number.isFinite(fps) || fps <= 0) {
|
|
1894
|
+
throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
|
|
1895
|
+
}
|
|
1896
|
+
const frames = await Promise.all(
|
|
1897
|
+
rawFrames.map(async (raw, index) => {
|
|
1898
|
+
const timestamped = isTimestampedFrameInput(raw);
|
|
1899
|
+
const imageInput = timestamped ? raw.image : raw;
|
|
1900
|
+
const explicit = timestamped ? raw.timestampSeconds : void 0;
|
|
1901
|
+
const timestampSeconds = explicit ?? index / fps;
|
|
1902
|
+
if (!Number.isFinite(timestampSeconds) || timestampSeconds < 0) {
|
|
1903
|
+
throw new VisualAIVideoError(
|
|
1904
|
+
`Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
|
|
1905
|
+
);
|
|
1906
|
+
}
|
|
1907
|
+
const image = await normalizeImage(imageInput);
|
|
1908
|
+
return {
|
|
1909
|
+
data: image.data,
|
|
1910
|
+
mimeType: image.mimeType,
|
|
1911
|
+
get base64() {
|
|
1912
|
+
return image.base64;
|
|
1913
|
+
},
|
|
1914
|
+
timestampSeconds,
|
|
1915
|
+
index
|
|
1916
|
+
};
|
|
1917
|
+
})
|
|
1918
|
+
);
|
|
1919
|
+
const durationSeconds = frames.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
|
|
1920
|
+
await saveDebugFrames(frames);
|
|
1921
|
+
return { kind: "video", frames, durationSeconds };
|
|
1922
|
+
}
|
|
1708
1923
|
async function normalizeMedia(input, videoOptions) {
|
|
1924
|
+
if (isFramesInput(input)) {
|
|
1925
|
+
return normalizeFrames(input);
|
|
1926
|
+
}
|
|
1709
1927
|
if (isVideoInput(input)) {
|
|
1710
1928
|
const { path, cleanup } = await resolveVideoToPath(input);
|
|
1711
1929
|
try {
|
|
@@ -1862,7 +2080,8 @@ function toSchemaOptions(schema) {
|
|
|
1862
2080
|
var PROVIDER_REGISTRY = {
|
|
1863
2081
|
anthropic: (config) => new AnthropicDriver(config),
|
|
1864
2082
|
openai: (config) => new OpenAIDriver(config),
|
|
1865
|
-
google: (config) => new GoogleDriver(config)
|
|
2083
|
+
google: (config) => new GoogleDriver(config),
|
|
2084
|
+
openrouter: (config) => new OpenRouterDriver(config)
|
|
1866
2085
|
};
|
|
1867
2086
|
function createDriver(provider, config) {
|
|
1868
2087
|
return PROVIDER_REGISTRY[provider](config);
|