visual-ai-assertions 0.13.0 → 0.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -0
- package/dist/index.cjs +60 -2
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +50 -11
- package/dist/index.d.ts +50 -11
- package/dist/index.js +60 -2
- package/dist/index.js.map +1 -1
- package/package.json +1 -1
package/dist/index.d.cts
CHANGED
|
@@ -420,8 +420,12 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
420
420
|
/**
|
|
421
421
|
* For video inputs, the indices of frames the model relied on to answer.
|
|
422
422
|
* Indices are 0-based and refer to entries in `frames.timestampsSeconds`.
|
|
423
|
+
*
|
|
424
|
+
* Nullable because providers with strict structured-output schemas (e.g. OpenAI)
|
|
425
|
+
* must mark every field required and represent "no value" as `null` rather than
|
|
426
|
+
* omitting the key, even for image inputs that were never asked to populate it.
|
|
423
427
|
*/
|
|
424
|
-
frameReferences: z.ZodOptional<z.ZodArray<z.ZodNumber, "many"
|
|
428
|
+
frameReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
425
429
|
usage: z.ZodOptional<z.ZodObject<{
|
|
426
430
|
inputTokens: z.ZodNumber;
|
|
427
431
|
outputTokens: z.ZodNumber;
|
|
@@ -457,7 +461,7 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
457
461
|
estimatedCost?: number | undefined;
|
|
458
462
|
durationSeconds?: number | undefined;
|
|
459
463
|
} | undefined;
|
|
460
|
-
frameReferences?: number[] | undefined;
|
|
464
|
+
frameReferences?: number[] | null | undefined;
|
|
461
465
|
}, {
|
|
462
466
|
issues: {
|
|
463
467
|
priority: "critical" | "major" | "minor";
|
|
@@ -473,11 +477,12 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
473
477
|
estimatedCost?: number | undefined;
|
|
474
478
|
durationSeconds?: number | undefined;
|
|
475
479
|
} | undefined;
|
|
476
|
-
frameReferences?: number[] | undefined;
|
|
480
|
+
frameReferences?: number[] | null | undefined;
|
|
477
481
|
}>;
|
|
478
482
|
/** Result returned by `ask()`. */
|
|
479
|
-
type AskResult = z.infer<typeof AskResultSchema> & {
|
|
483
|
+
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences"> & {
|
|
480
484
|
/** Present only when the input was a video. Describes which frames the model saw. */
|
|
485
|
+
frameReferences?: number[];
|
|
481
486
|
frames?: VideoFramesMetadata;
|
|
482
487
|
};
|
|
483
488
|
/** Supported input shapes for image arguments accepted by the client. */
|
|
@@ -488,6 +493,36 @@ type ImageInput = Buffer | Uint8Array | string;
|
|
|
488
493
|
* an image or a video.
|
|
489
494
|
*/
|
|
490
495
|
type MediaInput = ImageInput;
|
|
496
|
+
/**
|
|
497
|
+
* A single pre-sampled frame with an explicit timestamp. Use this shape when
|
|
498
|
+
* you want to control the timestamp the model sees; otherwise pass a bare
|
|
499
|
+
* `ImageInput` and the timestamp is derived from `fps` and the frame's position.
|
|
500
|
+
*/
|
|
501
|
+
interface TimestampedFrameInput {
|
|
502
|
+
/** Image source for the frame (buffer, data URL, base64, file path, or URL). */
|
|
503
|
+
image: ImageInput;
|
|
504
|
+
/** Timestamp in seconds, from the start of the sequence. Derived from `fps` when omitted. */
|
|
505
|
+
timestampSeconds?: number;
|
|
506
|
+
}
|
|
507
|
+
/**
|
|
508
|
+
* Pre-sampled frames supplied directly instead of a video file. Useful when you
|
|
509
|
+
* already have an array of screenshots and cannot (or would rather not) pass a
|
|
510
|
+
* video — this path never loads ffmpeg. Passed to `check()` / `ask()` in place
|
|
511
|
+
* of a `MediaInput`; downstream handling is identical to a sampled video.
|
|
512
|
+
*/
|
|
513
|
+
interface FramesInput {
|
|
514
|
+
/**
|
|
515
|
+
* Ordered frames, earliest first. Each entry is an image source or a
|
|
516
|
+
* `{ image, timestampSeconds }` pair. Must be non-empty and is capped at the
|
|
517
|
+
* same per-request frame limit as video sampling (60).
|
|
518
|
+
*/
|
|
519
|
+
frames: ReadonlyArray<ImageInput | TimestampedFrameInput>;
|
|
520
|
+
/**
|
|
521
|
+
* Frame rate used to derive timestamps for frames that don't carry their own
|
|
522
|
+
* `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
|
|
523
|
+
*/
|
|
524
|
+
fps?: number;
|
|
525
|
+
}
|
|
491
526
|
/** Supported image MIME types accepted by all providers. */
|
|
492
527
|
type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
|
|
493
528
|
/** Supported video MIME types the client can accept and sample frames from. */
|
|
@@ -628,9 +663,11 @@ interface VisualAIClient {
|
|
|
628
663
|
* frames automatically; statements pass if they are true at any sampled
|
|
629
664
|
* frame, and each statement result includes the timestamp where it
|
|
630
665
|
* matched. The `frames` metadata on the result reports which timestamps
|
|
631
|
-
* the model saw.
|
|
666
|
+
* the model saw. Pass a `FramesInput` (`{ frames, fps? }`) to supply
|
|
667
|
+
* pre-sampled frames directly — handled identically to a video timeline but
|
|
668
|
+
* without loading ffmpeg.
|
|
632
669
|
*
|
|
633
|
-
* @param input Image or video source as a buffer, URL, file path, or base64 string.
|
|
670
|
+
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
634
671
|
* @param statements One or more statements to validate against the input.
|
|
635
672
|
* @param options Optional additional instructions and video sampling overrides.
|
|
636
673
|
* @returns A structured result describing pass/fail, issues, and statement reasoning.
|
|
@@ -653,15 +690,17 @@ interface VisualAIClient {
|
|
|
653
690
|
* console.log(result.statements[0].timestampSeconds); // e.g. 3.5
|
|
654
691
|
* ```
|
|
655
692
|
*/
|
|
656
|
-
check(input: MediaInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
|
|
693
|
+
check(input: MediaInput | FramesInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
|
|
657
694
|
/**
|
|
658
695
|
* Asks an open-ended question about an image or video and returns a structured summary.
|
|
659
696
|
*
|
|
660
697
|
* Video inputs are sampled into frames and analyzed as a chronological
|
|
661
698
|
* timeline. The result's `frameReferences` array surfaces which frames the
|
|
662
|
-
* model relied on for its answer.
|
|
699
|
+
* model relied on for its answer. Pass a `FramesInput` (`{ frames, fps? }`)
|
|
700
|
+
* to supply pre-sampled frames directly — handled identically to a video
|
|
701
|
+
* timeline but without loading ffmpeg.
|
|
663
702
|
*
|
|
664
|
-
* @param input Image or video source as a buffer, URL, file path, or base64 string.
|
|
703
|
+
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
665
704
|
* @param prompt Prompt describing what to inspect in the input.
|
|
666
705
|
* @param options Optional additional instructions and video sampling overrides.
|
|
667
706
|
* @returns A summary with any detected issues.
|
|
@@ -673,7 +712,7 @@ interface VisualAIClient {
|
|
|
673
712
|
* const result = await client.ask(screenshot, "What looks visually broken on this page?");
|
|
674
713
|
* ```
|
|
675
714
|
*/
|
|
676
|
-
ask(input: MediaInput, prompt: string, options?: AskOptions): Promise<AskResult>;
|
|
715
|
+
ask(input: MediaInput | FramesInput, prompt: string, options?: AskOptions): Promise<AskResult>;
|
|
677
716
|
/**
|
|
678
717
|
* Compares two images and reports meaningful visual differences.
|
|
679
718
|
*
|
|
@@ -1050,4 +1089,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
|
|
|
1050
1089
|
*/
|
|
1051
1090
|
declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
|
|
1052
1091
|
|
|
1053
|
-
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
|
1092
|
+
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
package/dist/index.d.ts
CHANGED
|
@@ -420,8 +420,12 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
420
420
|
/**
|
|
421
421
|
* For video inputs, the indices of frames the model relied on to answer.
|
|
422
422
|
* Indices are 0-based and refer to entries in `frames.timestampsSeconds`.
|
|
423
|
+
*
|
|
424
|
+
* Nullable because providers with strict structured-output schemas (e.g. OpenAI)
|
|
425
|
+
* must mark every field required and represent "no value" as `null` rather than
|
|
426
|
+
* omitting the key, even for image inputs that were never asked to populate it.
|
|
423
427
|
*/
|
|
424
|
-
frameReferences: z.ZodOptional<z.ZodArray<z.ZodNumber, "many"
|
|
428
|
+
frameReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
425
429
|
usage: z.ZodOptional<z.ZodObject<{
|
|
426
430
|
inputTokens: z.ZodNumber;
|
|
427
431
|
outputTokens: z.ZodNumber;
|
|
@@ -457,7 +461,7 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
457
461
|
estimatedCost?: number | undefined;
|
|
458
462
|
durationSeconds?: number | undefined;
|
|
459
463
|
} | undefined;
|
|
460
|
-
frameReferences?: number[] | undefined;
|
|
464
|
+
frameReferences?: number[] | null | undefined;
|
|
461
465
|
}, {
|
|
462
466
|
issues: {
|
|
463
467
|
priority: "critical" | "major" | "minor";
|
|
@@ -473,11 +477,12 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
473
477
|
estimatedCost?: number | undefined;
|
|
474
478
|
durationSeconds?: number | undefined;
|
|
475
479
|
} | undefined;
|
|
476
|
-
frameReferences?: number[] | undefined;
|
|
480
|
+
frameReferences?: number[] | null | undefined;
|
|
477
481
|
}>;
|
|
478
482
|
/** Result returned by `ask()`. */
|
|
479
|
-
type AskResult = z.infer<typeof AskResultSchema> & {
|
|
483
|
+
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences"> & {
|
|
480
484
|
/** Present only when the input was a video. Describes which frames the model saw. */
|
|
485
|
+
frameReferences?: number[];
|
|
481
486
|
frames?: VideoFramesMetadata;
|
|
482
487
|
};
|
|
483
488
|
/** Supported input shapes for image arguments accepted by the client. */
|
|
@@ -488,6 +493,36 @@ type ImageInput = Buffer | Uint8Array | string;
|
|
|
488
493
|
* an image or a video.
|
|
489
494
|
*/
|
|
490
495
|
type MediaInput = ImageInput;
|
|
496
|
+
/**
|
|
497
|
+
* A single pre-sampled frame with an explicit timestamp. Use this shape when
|
|
498
|
+
* you want to control the timestamp the model sees; otherwise pass a bare
|
|
499
|
+
* `ImageInput` and the timestamp is derived from `fps` and the frame's position.
|
|
500
|
+
*/
|
|
501
|
+
interface TimestampedFrameInput {
|
|
502
|
+
/** Image source for the frame (buffer, data URL, base64, file path, or URL). */
|
|
503
|
+
image: ImageInput;
|
|
504
|
+
/** Timestamp in seconds, from the start of the sequence. Derived from `fps` when omitted. */
|
|
505
|
+
timestampSeconds?: number;
|
|
506
|
+
}
|
|
507
|
+
/**
|
|
508
|
+
* Pre-sampled frames supplied directly instead of a video file. Useful when you
|
|
509
|
+
* already have an array of screenshots and cannot (or would rather not) pass a
|
|
510
|
+
* video — this path never loads ffmpeg. Passed to `check()` / `ask()` in place
|
|
511
|
+
* of a `MediaInput`; downstream handling is identical to a sampled video.
|
|
512
|
+
*/
|
|
513
|
+
interface FramesInput {
|
|
514
|
+
/**
|
|
515
|
+
* Ordered frames, earliest first. Each entry is an image source or a
|
|
516
|
+
* `{ image, timestampSeconds }` pair. Must be non-empty and is capped at the
|
|
517
|
+
* same per-request frame limit as video sampling (60).
|
|
518
|
+
*/
|
|
519
|
+
frames: ReadonlyArray<ImageInput | TimestampedFrameInput>;
|
|
520
|
+
/**
|
|
521
|
+
* Frame rate used to derive timestamps for frames that don't carry their own
|
|
522
|
+
* `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
|
|
523
|
+
*/
|
|
524
|
+
fps?: number;
|
|
525
|
+
}
|
|
491
526
|
/** Supported image MIME types accepted by all providers. */
|
|
492
527
|
type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
|
|
493
528
|
/** Supported video MIME types the client can accept and sample frames from. */
|
|
@@ -628,9 +663,11 @@ interface VisualAIClient {
|
|
|
628
663
|
* frames automatically; statements pass if they are true at any sampled
|
|
629
664
|
* frame, and each statement result includes the timestamp where it
|
|
630
665
|
* matched. The `frames` metadata on the result reports which timestamps
|
|
631
|
-
* the model saw.
|
|
666
|
+
* the model saw. Pass a `FramesInput` (`{ frames, fps? }`) to supply
|
|
667
|
+
* pre-sampled frames directly — handled identically to a video timeline but
|
|
668
|
+
* without loading ffmpeg.
|
|
632
669
|
*
|
|
633
|
-
* @param input Image or video source as a buffer, URL, file path, or base64 string.
|
|
670
|
+
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
634
671
|
* @param statements One or more statements to validate against the input.
|
|
635
672
|
* @param options Optional additional instructions and video sampling overrides.
|
|
636
673
|
* @returns A structured result describing pass/fail, issues, and statement reasoning.
|
|
@@ -653,15 +690,17 @@ interface VisualAIClient {
|
|
|
653
690
|
* console.log(result.statements[0].timestampSeconds); // e.g. 3.5
|
|
654
691
|
* ```
|
|
655
692
|
*/
|
|
656
|
-
check(input: MediaInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
|
|
693
|
+
check(input: MediaInput | FramesInput, statements: string | string[], options?: CheckOptions): Promise<CheckResult>;
|
|
657
694
|
/**
|
|
658
695
|
* Asks an open-ended question about an image or video and returns a structured summary.
|
|
659
696
|
*
|
|
660
697
|
* Video inputs are sampled into frames and analyzed as a chronological
|
|
661
698
|
* timeline. The result's `frameReferences` array surfaces which frames the
|
|
662
|
-
* model relied on for its answer.
|
|
699
|
+
* model relied on for its answer. Pass a `FramesInput` (`{ frames, fps? }`)
|
|
700
|
+
* to supply pre-sampled frames directly — handled identically to a video
|
|
701
|
+
* timeline but without loading ffmpeg.
|
|
663
702
|
*
|
|
664
|
-
* @param input Image or video source as a buffer, URL, file path, or base64 string.
|
|
703
|
+
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
665
704
|
* @param prompt Prompt describing what to inspect in the input.
|
|
666
705
|
* @param options Optional additional instructions and video sampling overrides.
|
|
667
706
|
* @returns A summary with any detected issues.
|
|
@@ -673,7 +712,7 @@ interface VisualAIClient {
|
|
|
673
712
|
* const result = await client.ask(screenshot, "What looks visually broken on this page?");
|
|
674
713
|
* ```
|
|
675
714
|
*/
|
|
676
|
-
ask(input: MediaInput, prompt: string, options?: AskOptions): Promise<AskResult>;
|
|
715
|
+
ask(input: MediaInput | FramesInput, prompt: string, options?: AskOptions): Promise<AskResult>;
|
|
677
716
|
/**
|
|
678
717
|
* Compares two images and reports meaningful visual differences.
|
|
679
718
|
*
|
|
@@ -1050,4 +1089,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
|
|
|
1050
1089
|
*/
|
|
1051
1090
|
declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
|
|
1052
1091
|
|
|
1053
|
-
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
|
1092
|
+
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
package/dist/index.js
CHANGED
|
@@ -1705,7 +1705,57 @@ function isVideoInput(input) {
|
|
|
1705
1705
|
}
|
|
1706
1706
|
return false;
|
|
1707
1707
|
}
|
|
1708
|
+
function isFramesInput(input) {
|
|
1709
|
+
return typeof input === "object" && input !== null && !Buffer.isBuffer(input) && !(input instanceof Uint8Array) && Array.isArray(input.frames);
|
|
1710
|
+
}
|
|
1711
|
+
function isTimestampedFrameInput(frame) {
|
|
1712
|
+
return typeof frame === "object" && !Buffer.isBuffer(frame) && !(frame instanceof Uint8Array) && "image" in frame;
|
|
1713
|
+
}
|
|
1714
|
+
async function normalizeFrames(input) {
|
|
1715
|
+
const rawFrames = input.frames;
|
|
1716
|
+
const fps = input.fps ?? DEFAULT_FPS;
|
|
1717
|
+
if (rawFrames.length === 0) {
|
|
1718
|
+
throw new VisualAIVideoError("frames must be a non-empty array of image inputs");
|
|
1719
|
+
}
|
|
1720
|
+
if (rawFrames.length > MAX_FRAMES_HARD_CAP) {
|
|
1721
|
+
throw new VisualAIVideoError(
|
|
1722
|
+
`frames length ${rawFrames.length} exceeds the hard cap of ${MAX_FRAMES_HARD_CAP}. Pass fewer frames or open an issue if you need a larger limit.`
|
|
1723
|
+
);
|
|
1724
|
+
}
|
|
1725
|
+
if (!Number.isFinite(fps) || fps <= 0) {
|
|
1726
|
+
throw new VisualAIVideoError(`Invalid fps: ${fps}. Must be a finite number > 0.`);
|
|
1727
|
+
}
|
|
1728
|
+
const frames = await Promise.all(
|
|
1729
|
+
rawFrames.map(async (raw, index) => {
|
|
1730
|
+
const timestamped = isTimestampedFrameInput(raw);
|
|
1731
|
+
const imageInput = timestamped ? raw.image : raw;
|
|
1732
|
+
const explicit = timestamped ? raw.timestampSeconds : void 0;
|
|
1733
|
+
const timestampSeconds = explicit ?? index / fps;
|
|
1734
|
+
if (!Number.isFinite(timestampSeconds) || timestampSeconds < 0) {
|
|
1735
|
+
throw new VisualAIVideoError(
|
|
1736
|
+
`Invalid timestampSeconds for frame ${index}: ${String(timestampSeconds)}. Must be a finite number >= 0.`
|
|
1737
|
+
);
|
|
1738
|
+
}
|
|
1739
|
+
const image = await normalizeImage(imageInput);
|
|
1740
|
+
return {
|
|
1741
|
+
data: image.data,
|
|
1742
|
+
mimeType: image.mimeType,
|
|
1743
|
+
get base64() {
|
|
1744
|
+
return image.base64;
|
|
1745
|
+
},
|
|
1746
|
+
timestampSeconds,
|
|
1747
|
+
index
|
|
1748
|
+
};
|
|
1749
|
+
})
|
|
1750
|
+
);
|
|
1751
|
+
const durationSeconds = frames.reduce((max, f) => Math.max(max, f.timestampSeconds), 0);
|
|
1752
|
+
await saveDebugFrames(frames);
|
|
1753
|
+
return { kind: "video", frames, durationSeconds };
|
|
1754
|
+
}
|
|
1708
1755
|
async function normalizeMedia(input, videoOptions) {
|
|
1756
|
+
if (isFramesInput(input)) {
|
|
1757
|
+
return normalizeFrames(input);
|
|
1758
|
+
}
|
|
1709
1759
|
if (isVideoInput(input)) {
|
|
1710
1760
|
const { path, cleanup } = await resolveVideoToPath(input);
|
|
1711
1761
|
try {
|
|
@@ -1785,8 +1835,12 @@ var AskResultSchema = z.object({
|
|
|
1785
1835
|
/**
|
|
1786
1836
|
* For video inputs, the indices of frames the model relied on to answer.
|
|
1787
1837
|
* Indices are 0-based and refer to entries in `frames.timestampsSeconds`.
|
|
1838
|
+
*
|
|
1839
|
+
* Nullable because providers with strict structured-output schemas (e.g. OpenAI)
|
|
1840
|
+
* must mark every field required and represent "no value" as `null` rather than
|
|
1841
|
+
* omitting the key, even for image inputs that were never asked to populate it.
|
|
1788
1842
|
*/
|
|
1789
|
-
frameReferences: z.array(z.number().int().nonnegative()).optional(),
|
|
1843
|
+
frameReferences: z.array(z.number().int().nonnegative()).nullable().optional(),
|
|
1790
1844
|
usage: UsageInfoSchema.optional()
|
|
1791
1845
|
});
|
|
1792
1846
|
|
|
@@ -1838,7 +1892,11 @@ function parseCheckResponse(raw) {
|
|
|
1838
1892
|
return reconcileCheckResult(result);
|
|
1839
1893
|
}
|
|
1840
1894
|
function parseAskResponse(raw) {
|
|
1841
|
-
|
|
1895
|
+
const result = parseResponse(raw, AskResponseSchema);
|
|
1896
|
+
return {
|
|
1897
|
+
...result,
|
|
1898
|
+
frameReferences: result.frameReferences ?? void 0
|
|
1899
|
+
};
|
|
1842
1900
|
}
|
|
1843
1901
|
function parseCompareResponse(raw) {
|
|
1844
1902
|
return parseResponse(raw, CompareResponseSchema);
|