visual-ai-assertions 0.23.0 → 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +44 -7
- package/dist/index.cjs +385 -55
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +105 -17
- package/dist/index.d.ts +105 -17
- package/dist/index.js +385 -55
- package/dist/index.js.map +1 -1
- package/package.json +8 -8
package/dist/index.d.cts
CHANGED
|
@@ -368,17 +368,39 @@ declare const CheckResultSchema: z.ZodObject<{
|
|
|
368
368
|
* Populated client-side; not part of the model's response.
|
|
369
369
|
*/
|
|
370
370
|
interface VideoFramesMetadata {
|
|
371
|
-
/**
|
|
371
|
+
/** Number of frames actually sent to the model (after unchanged frames were dropped). */
|
|
372
372
|
count: number;
|
|
373
|
-
/** Timestamp (seconds, from the start of the clip) of each
|
|
373
|
+
/** Timestamp (seconds, from the start of the clip) of each sent frame, in order. */
|
|
374
374
|
timestampsSeconds: number[];
|
|
375
375
|
/** Total duration of the source video in seconds. */
|
|
376
376
|
durationSeconds: number;
|
|
377
|
+
/**
|
|
378
|
+
* Number of sampled frames dropped because they did not visibly change from
|
|
379
|
+
* the preceding kept frame. `0` when dedupe is disabled or nothing was dropped.
|
|
380
|
+
* See `FrameDedupeOptions`.
|
|
381
|
+
*/
|
|
382
|
+
droppedUnchanged: number;
|
|
383
|
+
}
|
|
384
|
+
/**
|
|
385
|
+
* Metadata describing a video that was delivered to the model natively (as the
|
|
386
|
+
* video itself rather than sampled frames). Populated client-side.
|
|
387
|
+
*/
|
|
388
|
+
interface NativeVideoMetadata {
|
|
389
|
+
/** Total duration of the source video in seconds. */
|
|
390
|
+
durationSeconds: number;
|
|
391
|
+
/** Sampling rate requested from the provider, in frames per second. */
|
|
392
|
+
fps: number;
|
|
393
|
+
/** MIME type the video was sent as. */
|
|
394
|
+
mimeType: SupportedVideoMimeType;
|
|
395
|
+
/** Whether the bytes went inline in the request or through the provider's file upload API. */
|
|
396
|
+
delivery: "inline" | "file";
|
|
377
397
|
}
|
|
378
398
|
/** Result returned by `check()` and the template convenience methods. */
|
|
379
399
|
type CheckResult = z.infer<typeof CheckResultSchema> & {
|
|
380
|
-
/** Present only when the input was a video. Describes which frames the model saw. */
|
|
400
|
+
/** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
|
|
381
401
|
frames?: VideoFramesMetadata;
|
|
402
|
+
/** Present only when the input was a video delivered natively to the provider. */
|
|
403
|
+
video?: NativeVideoMetadata;
|
|
382
404
|
};
|
|
383
405
|
/** Zod schema for an individual visual change reported by `compare()`. */
|
|
384
406
|
declare const ChangeEntrySchema: z.ZodObject<{
|
|
@@ -514,6 +536,12 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
514
536
|
* omitting the key, even for image inputs that were never asked to populate it.
|
|
515
537
|
*/
|
|
516
538
|
frameReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
539
|
+
/**
|
|
540
|
+
* For natively delivered video, the timestamps (seconds from the start of
|
|
541
|
+
* the clip) the model relied on to answer. The native counterpart of
|
|
542
|
+
* `frameReferences`. Nullable for the same strict-schema reason.
|
|
543
|
+
*/
|
|
544
|
+
timestampReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
517
545
|
usage: z.ZodOptional<z.ZodObject<{
|
|
518
546
|
inputTokens: z.ZodNumber;
|
|
519
547
|
outputTokens: z.ZodNumber;
|
|
@@ -566,6 +594,7 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
566
594
|
durationSeconds?: number | undefined;
|
|
567
595
|
} | undefined;
|
|
568
596
|
frameReferences?: number[] | null | undefined;
|
|
597
|
+
timestampReferences?: number[] | null | undefined;
|
|
569
598
|
}, {
|
|
570
599
|
issues: {
|
|
571
600
|
priority: "critical" | "major" | "minor";
|
|
@@ -584,12 +613,18 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
584
613
|
durationSeconds?: number | undefined;
|
|
585
614
|
} | undefined;
|
|
586
615
|
frameReferences?: number[] | null | undefined;
|
|
616
|
+
timestampReferences?: number[] | null | undefined;
|
|
587
617
|
}>;
|
|
588
618
|
/** Result returned by `ask()`. */
|
|
589
|
-
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences"> & {
|
|
590
|
-
/** Present only when the input was a video
|
|
619
|
+
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences" | "timestampReferences"> & {
|
|
620
|
+
/** Present only when the input was a video sampled into frames. Indices into `frames.timestampsSeconds`. */
|
|
591
621
|
frameReferences?: number[];
|
|
622
|
+
/** Present only when the input was a video delivered natively. Seconds from the start of the clip. */
|
|
623
|
+
timestampReferences?: number[];
|
|
624
|
+
/** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
|
|
592
625
|
frames?: VideoFramesMetadata;
|
|
626
|
+
/** Present only when the input was a video delivered natively to the provider. */
|
|
627
|
+
video?: NativeVideoMetadata;
|
|
593
628
|
};
|
|
594
629
|
/** Supported input shapes for image arguments accepted by the client. */
|
|
595
630
|
type ImageInput = Buffer | Uint8Array | string;
|
|
@@ -628,6 +663,11 @@ interface FramesInput {
|
|
|
628
663
|
* `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
|
|
629
664
|
*/
|
|
630
665
|
fps?: number;
|
|
666
|
+
/**
|
|
667
|
+
* Drop frames that did not visibly change from the preceding kept frame
|
|
668
|
+
* before sending to the provider. Default `true`. See `FrameDedupeOptions`.
|
|
669
|
+
*/
|
|
670
|
+
dedupe?: FrameDedupeOptions;
|
|
631
671
|
}
|
|
632
672
|
/** Supported image MIME types accepted by all providers. */
|
|
633
673
|
type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
|
|
@@ -794,7 +834,51 @@ interface VideoSamplingOptions {
|
|
|
794
834
|
* Default `10`.
|
|
795
835
|
*/
|
|
796
836
|
maxDurationSeconds?: number;
|
|
837
|
+
/**
|
|
838
|
+
* Drop sampled frames that did not visibly change from the preceding kept
|
|
839
|
+
* frame before sending to the provider. Default `true`. See `FrameDedupeOptions`.
|
|
840
|
+
* Only applies when frames are sampled; ignored for native delivery.
|
|
841
|
+
*/
|
|
842
|
+
dedupe?: FrameDedupeOptions;
|
|
843
|
+
/**
|
|
844
|
+
* How the video reaches the model. Default `"auto"`. See `VideoDeliveryMode`.
|
|
845
|
+
*/
|
|
846
|
+
mode?: VideoDeliveryMode;
|
|
797
847
|
}
|
|
848
|
+
/**
|
|
849
|
+
* How a video input is delivered to the model.
|
|
850
|
+
*
|
|
851
|
+
* - `"auto"` (default): send the video itself when the provider accepts video
|
|
852
|
+
* natively (Google models), otherwise sample frames with ffmpeg.
|
|
853
|
+
* - `"native"`: always send the video itself. Throws `VisualAIConfigError`
|
|
854
|
+
* when the provider has no native video support.
|
|
855
|
+
* - `"frames"`: always sample frames with ffmpeg, whatever the provider.
|
|
856
|
+
*
|
|
857
|
+
* Native delivery still probes the duration and enforces `maxDurationSeconds`
|
|
858
|
+
* before any provider call, and passes `fps` on as the provider's sampling
|
|
859
|
+
* rate. `maxFrames` and `dedupe` apply to frame sampling only. Pre-sampled
|
|
860
|
+
* `FramesInput` is always sent as frames.
|
|
861
|
+
*/
|
|
862
|
+
type VideoDeliveryMode = "auto" | "native" | "frames";
|
|
863
|
+
/**
|
|
864
|
+
* Controls dropping of frames that did not visibly change from the previous
|
|
865
|
+
* kept frame, so a static screen does not cost input tokens for every sample.
|
|
866
|
+
*
|
|
867
|
+
* - `true` (default): drop unchanged frames using the default threshold.
|
|
868
|
+
* - `false`: send every sampled frame.
|
|
869
|
+
* - `{ threshold }`: the fraction `(0, 1]` of a frame's pixels that must differ
|
|
870
|
+
* from the last kept frame for it to count as changed. Default `0.001` (0.1%,
|
|
871
|
+
* roughly a 37x37 px region on a 1568x880 frame). Lower it to keep smaller
|
|
872
|
+
* changes; raise it to ignore more.
|
|
873
|
+
*
|
|
874
|
+
* Each frame is compared against the most recently *kept* frame, so gradual
|
|
875
|
+
* drift accumulates and is eventually kept. The first frame is always kept.
|
|
876
|
+
* Pixel-level compression noise and tiny flickers such as a blinking text
|
|
877
|
+
* caret fall below the default threshold.
|
|
878
|
+
*/
|
|
879
|
+
type FrameDedupeOptions = boolean | {
|
|
880
|
+
threshold?: number;
|
|
881
|
+
};
|
|
798
882
|
/**
|
|
799
883
|
* A single frame extracted from a video input. Identical in shape to
|
|
800
884
|
* `NormalizedImage` so it can be passed transparently to provider drivers.
|
|
@@ -820,12 +904,14 @@ interface VisualAIClient {
|
|
|
820
904
|
* Verifies one or more statements against a single image or video.
|
|
821
905
|
*
|
|
822
906
|
* Pass an image (PNG/JPEG/WebP/GIF) for a single-frame check. Pass a video
|
|
823
|
-
* (MP4/WebM/MOV/MKV file path,
|
|
824
|
-
*
|
|
825
|
-
*
|
|
826
|
-
*
|
|
827
|
-
* the
|
|
828
|
-
*
|
|
907
|
+
* (MP4/WebM/MOV/MKV file path, base64, Buffer) and statements pass if they
|
|
908
|
+
* are true at any point, with each statement result carrying the timestamp
|
|
909
|
+
* where it matched. On providers that accept video natively (Google models)
|
|
910
|
+
* the video itself is sent and the result's `video` metadata describes the
|
|
911
|
+
* delivery; elsewhere the client samples frames with ffmpeg and the `frames`
|
|
912
|
+
* metadata reports which timestamps the model saw. Control this with
|
|
913
|
+
* `video.mode`. Pass a `FramesInput` (`{ frames, fps? }`) to supply
|
|
914
|
+
* pre-sampled frames directly — handled identically to a sampled timeline but
|
|
829
915
|
* without loading ffmpeg.
|
|
830
916
|
*
|
|
831
917
|
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
@@ -855,11 +941,13 @@ interface VisualAIClient {
|
|
|
855
941
|
/**
|
|
856
942
|
* Asks an open-ended question about an image or video and returns a structured summary.
|
|
857
943
|
*
|
|
858
|
-
* Video inputs are
|
|
859
|
-
*
|
|
860
|
-
*
|
|
861
|
-
*
|
|
862
|
-
*
|
|
944
|
+
* Video inputs are analyzed as a chronological timeline. On providers that
|
|
945
|
+
* accept video natively (Google models) the video itself is sent and the
|
|
946
|
+
* result's `timestampReferences` array surfaces the moments the model relied
|
|
947
|
+
* on; elsewhere frames are sampled with ffmpeg and `frameReferences` indexes
|
|
948
|
+
* into `frames.timestampsSeconds`. Control this with `video.mode`. Pass a
|
|
949
|
+
* `FramesInput` (`{ frames, fps? }`) to supply pre-sampled frames directly —
|
|
950
|
+
* handled identically to a sampled timeline but without loading ffmpeg.
|
|
863
951
|
*
|
|
864
952
|
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
865
953
|
* @param prompt Prompt describing what to inspect in the input.
|
|
@@ -1250,4 +1338,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
|
|
|
1250
1338
|
*/
|
|
1251
1339
|
declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
|
|
1252
1340
|
|
|
1253
|
-
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
|
1341
|
+
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FrameDedupeOptions, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type NativeVideoMetadata, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoDeliveryMode, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
package/dist/index.d.ts
CHANGED
|
@@ -368,17 +368,39 @@ declare const CheckResultSchema: z.ZodObject<{
|
|
|
368
368
|
* Populated client-side; not part of the model's response.
|
|
369
369
|
*/
|
|
370
370
|
interface VideoFramesMetadata {
|
|
371
|
-
/**
|
|
371
|
+
/** Number of frames actually sent to the model (after unchanged frames were dropped). */
|
|
372
372
|
count: number;
|
|
373
|
-
/** Timestamp (seconds, from the start of the clip) of each
|
|
373
|
+
/** Timestamp (seconds, from the start of the clip) of each sent frame, in order. */
|
|
374
374
|
timestampsSeconds: number[];
|
|
375
375
|
/** Total duration of the source video in seconds. */
|
|
376
376
|
durationSeconds: number;
|
|
377
|
+
/**
|
|
378
|
+
* Number of sampled frames dropped because they did not visibly change from
|
|
379
|
+
* the preceding kept frame. `0` when dedupe is disabled or nothing was dropped.
|
|
380
|
+
* See `FrameDedupeOptions`.
|
|
381
|
+
*/
|
|
382
|
+
droppedUnchanged: number;
|
|
383
|
+
}
|
|
384
|
+
/**
|
|
385
|
+
* Metadata describing a video that was delivered to the model natively (as the
|
|
386
|
+
* video itself rather than sampled frames). Populated client-side.
|
|
387
|
+
*/
|
|
388
|
+
interface NativeVideoMetadata {
|
|
389
|
+
/** Total duration of the source video in seconds. */
|
|
390
|
+
durationSeconds: number;
|
|
391
|
+
/** Sampling rate requested from the provider, in frames per second. */
|
|
392
|
+
fps: number;
|
|
393
|
+
/** MIME type the video was sent as. */
|
|
394
|
+
mimeType: SupportedVideoMimeType;
|
|
395
|
+
/** Whether the bytes went inline in the request or through the provider's file upload API. */
|
|
396
|
+
delivery: "inline" | "file";
|
|
377
397
|
}
|
|
378
398
|
/** Result returned by `check()` and the template convenience methods. */
|
|
379
399
|
type CheckResult = z.infer<typeof CheckResultSchema> & {
|
|
380
|
-
/** Present only when the input was a video. Describes which frames the model saw. */
|
|
400
|
+
/** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
|
|
381
401
|
frames?: VideoFramesMetadata;
|
|
402
|
+
/** Present only when the input was a video delivered natively to the provider. */
|
|
403
|
+
video?: NativeVideoMetadata;
|
|
382
404
|
};
|
|
383
405
|
/** Zod schema for an individual visual change reported by `compare()`. */
|
|
384
406
|
declare const ChangeEntrySchema: z.ZodObject<{
|
|
@@ -514,6 +536,12 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
514
536
|
* omitting the key, even for image inputs that were never asked to populate it.
|
|
515
537
|
*/
|
|
516
538
|
frameReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
539
|
+
/**
|
|
540
|
+
* For natively delivered video, the timestamps (seconds from the start of
|
|
541
|
+
* the clip) the model relied on to answer. The native counterpart of
|
|
542
|
+
* `frameReferences`. Nullable for the same strict-schema reason.
|
|
543
|
+
*/
|
|
544
|
+
timestampReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
517
545
|
usage: z.ZodOptional<z.ZodObject<{
|
|
518
546
|
inputTokens: z.ZodNumber;
|
|
519
547
|
outputTokens: z.ZodNumber;
|
|
@@ -566,6 +594,7 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
566
594
|
durationSeconds?: number | undefined;
|
|
567
595
|
} | undefined;
|
|
568
596
|
frameReferences?: number[] | null | undefined;
|
|
597
|
+
timestampReferences?: number[] | null | undefined;
|
|
569
598
|
}, {
|
|
570
599
|
issues: {
|
|
571
600
|
priority: "critical" | "major" | "minor";
|
|
@@ -584,12 +613,18 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
584
613
|
durationSeconds?: number | undefined;
|
|
585
614
|
} | undefined;
|
|
586
615
|
frameReferences?: number[] | null | undefined;
|
|
616
|
+
timestampReferences?: number[] | null | undefined;
|
|
587
617
|
}>;
|
|
588
618
|
/** Result returned by `ask()`. */
|
|
589
|
-
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences"> & {
|
|
590
|
-
/** Present only when the input was a video
|
|
619
|
+
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences" | "timestampReferences"> & {
|
|
620
|
+
/** Present only when the input was a video sampled into frames. Indices into `frames.timestampsSeconds`. */
|
|
591
621
|
frameReferences?: number[];
|
|
622
|
+
/** Present only when the input was a video delivered natively. Seconds from the start of the clip. */
|
|
623
|
+
timestampReferences?: number[];
|
|
624
|
+
/** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
|
|
592
625
|
frames?: VideoFramesMetadata;
|
|
626
|
+
/** Present only when the input was a video delivered natively to the provider. */
|
|
627
|
+
video?: NativeVideoMetadata;
|
|
593
628
|
};
|
|
594
629
|
/** Supported input shapes for image arguments accepted by the client. */
|
|
595
630
|
type ImageInput = Buffer | Uint8Array | string;
|
|
@@ -628,6 +663,11 @@ interface FramesInput {
|
|
|
628
663
|
* `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
|
|
629
664
|
*/
|
|
630
665
|
fps?: number;
|
|
666
|
+
/**
|
|
667
|
+
* Drop frames that did not visibly change from the preceding kept frame
|
|
668
|
+
* before sending to the provider. Default `true`. See `FrameDedupeOptions`.
|
|
669
|
+
*/
|
|
670
|
+
dedupe?: FrameDedupeOptions;
|
|
631
671
|
}
|
|
632
672
|
/** Supported image MIME types accepted by all providers. */
|
|
633
673
|
type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
|
|
@@ -794,7 +834,51 @@ interface VideoSamplingOptions {
|
|
|
794
834
|
* Default `10`.
|
|
795
835
|
*/
|
|
796
836
|
maxDurationSeconds?: number;
|
|
837
|
+
/**
|
|
838
|
+
* Drop sampled frames that did not visibly change from the preceding kept
|
|
839
|
+
* frame before sending to the provider. Default `true`. See `FrameDedupeOptions`.
|
|
840
|
+
* Only applies when frames are sampled; ignored for native delivery.
|
|
841
|
+
*/
|
|
842
|
+
dedupe?: FrameDedupeOptions;
|
|
843
|
+
/**
|
|
844
|
+
* How the video reaches the model. Default `"auto"`. See `VideoDeliveryMode`.
|
|
845
|
+
*/
|
|
846
|
+
mode?: VideoDeliveryMode;
|
|
797
847
|
}
|
|
848
|
+
/**
|
|
849
|
+
* How a video input is delivered to the model.
|
|
850
|
+
*
|
|
851
|
+
* - `"auto"` (default): send the video itself when the provider accepts video
|
|
852
|
+
* natively (Google models), otherwise sample frames with ffmpeg.
|
|
853
|
+
* - `"native"`: always send the video itself. Throws `VisualAIConfigError`
|
|
854
|
+
* when the provider has no native video support.
|
|
855
|
+
* - `"frames"`: always sample frames with ffmpeg, whatever the provider.
|
|
856
|
+
*
|
|
857
|
+
* Native delivery still probes the duration and enforces `maxDurationSeconds`
|
|
858
|
+
* before any provider call, and passes `fps` on as the provider's sampling
|
|
859
|
+
* rate. `maxFrames` and `dedupe` apply to frame sampling only. Pre-sampled
|
|
860
|
+
* `FramesInput` is always sent as frames.
|
|
861
|
+
*/
|
|
862
|
+
type VideoDeliveryMode = "auto" | "native" | "frames";
|
|
863
|
+
/**
|
|
864
|
+
* Controls dropping of frames that did not visibly change from the previous
|
|
865
|
+
* kept frame, so a static screen does not cost input tokens for every sample.
|
|
866
|
+
*
|
|
867
|
+
* - `true` (default): drop unchanged frames using the default threshold.
|
|
868
|
+
* - `false`: send every sampled frame.
|
|
869
|
+
* - `{ threshold }`: the fraction `(0, 1]` of a frame's pixels that must differ
|
|
870
|
+
* from the last kept frame for it to count as changed. Default `0.001` (0.1%,
|
|
871
|
+
* roughly a 37x37 px region on a 1568x880 frame). Lower it to keep smaller
|
|
872
|
+
* changes; raise it to ignore more.
|
|
873
|
+
*
|
|
874
|
+
* Each frame is compared against the most recently *kept* frame, so gradual
|
|
875
|
+
* drift accumulates and is eventually kept. The first frame is always kept.
|
|
876
|
+
* Pixel-level compression noise and tiny flickers such as a blinking text
|
|
877
|
+
* caret fall below the default threshold.
|
|
878
|
+
*/
|
|
879
|
+
type FrameDedupeOptions = boolean | {
|
|
880
|
+
threshold?: number;
|
|
881
|
+
};
|
|
798
882
|
/**
|
|
799
883
|
* A single frame extracted from a video input. Identical in shape to
|
|
800
884
|
* `NormalizedImage` so it can be passed transparently to provider drivers.
|
|
@@ -820,12 +904,14 @@ interface VisualAIClient {
|
|
|
820
904
|
* Verifies one or more statements against a single image or video.
|
|
821
905
|
*
|
|
822
906
|
* Pass an image (PNG/JPEG/WebP/GIF) for a single-frame check. Pass a video
|
|
823
|
-
* (MP4/WebM/MOV/MKV file path,
|
|
824
|
-
*
|
|
825
|
-
*
|
|
826
|
-
*
|
|
827
|
-
* the
|
|
828
|
-
*
|
|
907
|
+
* (MP4/WebM/MOV/MKV file path, base64, Buffer) and statements pass if they
|
|
908
|
+
* are true at any point, with each statement result carrying the timestamp
|
|
909
|
+
* where it matched. On providers that accept video natively (Google models)
|
|
910
|
+
* the video itself is sent and the result's `video` metadata describes the
|
|
911
|
+
* delivery; elsewhere the client samples frames with ffmpeg and the `frames`
|
|
912
|
+
* metadata reports which timestamps the model saw. Control this with
|
|
913
|
+
* `video.mode`. Pass a `FramesInput` (`{ frames, fps? }`) to supply
|
|
914
|
+
* pre-sampled frames directly — handled identically to a sampled timeline but
|
|
829
915
|
* without loading ffmpeg.
|
|
830
916
|
*
|
|
831
917
|
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
@@ -855,11 +941,13 @@ interface VisualAIClient {
|
|
|
855
941
|
/**
|
|
856
942
|
* Asks an open-ended question about an image or video and returns a structured summary.
|
|
857
943
|
*
|
|
858
|
-
* Video inputs are
|
|
859
|
-
*
|
|
860
|
-
*
|
|
861
|
-
*
|
|
862
|
-
*
|
|
944
|
+
* Video inputs are analyzed as a chronological timeline. On providers that
|
|
945
|
+
* accept video natively (Google models) the video itself is sent and the
|
|
946
|
+
* result's `timestampReferences` array surfaces the moments the model relied
|
|
947
|
+
* on; elsewhere frames are sampled with ffmpeg and `frameReferences` indexes
|
|
948
|
+
* into `frames.timestampsSeconds`. Control this with `video.mode`. Pass a
|
|
949
|
+
* `FramesInput` (`{ frames, fps? }`) to supply pre-sampled frames directly —
|
|
950
|
+
* handled identically to a sampled timeline but without loading ffmpeg.
|
|
863
951
|
*
|
|
864
952
|
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
865
953
|
* @param prompt Prompt describing what to inspect in the input.
|
|
@@ -1250,4 +1338,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
|
|
|
1250
1338
|
*/
|
|
1251
1339
|
declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
|
|
1252
1340
|
|
|
1253
|
-
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
|
1341
|
+
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FrameDedupeOptions, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type NativeVideoMetadata, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoDeliveryMode, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|