visual-ai-assertions 0.22.0 → 0.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +49 -9
- package/dist/index.cjs +393 -56
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +106 -17
- package/dist/index.d.ts +106 -17
- package/dist/index.js +393 -56
- package/dist/index.js.map +1 -1
- package/package.json +8 -8
package/dist/index.d.cts
CHANGED
|
@@ -79,6 +79,7 @@ declare const Model: {
|
|
|
79
79
|
readonly QWEN_3_8_MAX: "qwen/qwen3.8-max";
|
|
80
80
|
readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
|
|
81
81
|
readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
|
|
82
|
+
readonly GLM_5_3_FLASH: "z-ai/glm-5.3-flash";
|
|
82
83
|
};
|
|
83
84
|
};
|
|
84
85
|
/** Union of all built-in model name literals exposed by `Model`. */
|
|
@@ -367,17 +368,39 @@ declare const CheckResultSchema: z.ZodObject<{
|
|
|
367
368
|
* Populated client-side; not part of the model's response.
|
|
368
369
|
*/
|
|
369
370
|
interface VideoFramesMetadata {
|
|
370
|
-
/**
|
|
371
|
+
/** Number of frames actually sent to the model (after unchanged frames were dropped). */
|
|
371
372
|
count: number;
|
|
372
|
-
/** Timestamp (seconds, from the start of the clip) of each
|
|
373
|
+
/** Timestamp (seconds, from the start of the clip) of each sent frame, in order. */
|
|
373
374
|
timestampsSeconds: number[];
|
|
374
375
|
/** Total duration of the source video in seconds. */
|
|
375
376
|
durationSeconds: number;
|
|
377
|
+
/**
|
|
378
|
+
* Number of sampled frames dropped because they did not visibly change from
|
|
379
|
+
* the preceding kept frame. `0` when dedupe is disabled or nothing was dropped.
|
|
380
|
+
* See `FrameDedupeOptions`.
|
|
381
|
+
*/
|
|
382
|
+
droppedUnchanged: number;
|
|
383
|
+
}
|
|
384
|
+
/**
|
|
385
|
+
* Metadata describing a video that was delivered to the model natively (as the
|
|
386
|
+
* video itself rather than sampled frames). Populated client-side.
|
|
387
|
+
*/
|
|
388
|
+
interface NativeVideoMetadata {
|
|
389
|
+
/** Total duration of the source video in seconds. */
|
|
390
|
+
durationSeconds: number;
|
|
391
|
+
/** Sampling rate requested from the provider, in frames per second. */
|
|
392
|
+
fps: number;
|
|
393
|
+
/** MIME type the video was sent as. */
|
|
394
|
+
mimeType: SupportedVideoMimeType;
|
|
395
|
+
/** Whether the bytes went inline in the request or through the provider's file upload API. */
|
|
396
|
+
delivery: "inline" | "file";
|
|
376
397
|
}
|
|
377
398
|
/** Result returned by `check()` and the template convenience methods. */
|
|
378
399
|
type CheckResult = z.infer<typeof CheckResultSchema> & {
|
|
379
|
-
/** Present only when the input was a video. Describes which frames the model saw. */
|
|
400
|
+
/** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
|
|
380
401
|
frames?: VideoFramesMetadata;
|
|
402
|
+
/** Present only when the input was a video delivered natively to the provider. */
|
|
403
|
+
video?: NativeVideoMetadata;
|
|
381
404
|
};
|
|
382
405
|
/** Zod schema for an individual visual change reported by `compare()`. */
|
|
383
406
|
declare const ChangeEntrySchema: z.ZodObject<{
|
|
@@ -513,6 +536,12 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
513
536
|
* omitting the key, even for image inputs that were never asked to populate it.
|
|
514
537
|
*/
|
|
515
538
|
frameReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
539
|
+
/**
|
|
540
|
+
* For natively delivered video, the timestamps (seconds from the start of
|
|
541
|
+
* the clip) the model relied on to answer. The native counterpart of
|
|
542
|
+
* `frameReferences`. Nullable for the same strict-schema reason.
|
|
543
|
+
*/
|
|
544
|
+
timestampReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
516
545
|
usage: z.ZodOptional<z.ZodObject<{
|
|
517
546
|
inputTokens: z.ZodNumber;
|
|
518
547
|
outputTokens: z.ZodNumber;
|
|
@@ -565,6 +594,7 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
565
594
|
durationSeconds?: number | undefined;
|
|
566
595
|
} | undefined;
|
|
567
596
|
frameReferences?: number[] | null | undefined;
|
|
597
|
+
timestampReferences?: number[] | null | undefined;
|
|
568
598
|
}, {
|
|
569
599
|
issues: {
|
|
570
600
|
priority: "critical" | "major" | "minor";
|
|
@@ -583,12 +613,18 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
583
613
|
durationSeconds?: number | undefined;
|
|
584
614
|
} | undefined;
|
|
585
615
|
frameReferences?: number[] | null | undefined;
|
|
616
|
+
timestampReferences?: number[] | null | undefined;
|
|
586
617
|
}>;
|
|
587
618
|
/** Result returned by `ask()`. */
|
|
588
|
-
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences"> & {
|
|
589
|
-
/** Present only when the input was a video
|
|
619
|
+
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences" | "timestampReferences"> & {
|
|
620
|
+
/** Present only when the input was a video sampled into frames. Indices into `frames.timestampsSeconds`. */
|
|
590
621
|
frameReferences?: number[];
|
|
622
|
+
/** Present only when the input was a video delivered natively. Seconds from the start of the clip. */
|
|
623
|
+
timestampReferences?: number[];
|
|
624
|
+
/** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
|
|
591
625
|
frames?: VideoFramesMetadata;
|
|
626
|
+
/** Present only when the input was a video delivered natively to the provider. */
|
|
627
|
+
video?: NativeVideoMetadata;
|
|
592
628
|
};
|
|
593
629
|
/** Supported input shapes for image arguments accepted by the client. */
|
|
594
630
|
type ImageInput = Buffer | Uint8Array | string;
|
|
@@ -627,6 +663,11 @@ interface FramesInput {
|
|
|
627
663
|
* `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
|
|
628
664
|
*/
|
|
629
665
|
fps?: number;
|
|
666
|
+
/**
|
|
667
|
+
* Drop frames that did not visibly change from the preceding kept frame
|
|
668
|
+
* before sending to the provider. Default `true`. See `FrameDedupeOptions`.
|
|
669
|
+
*/
|
|
670
|
+
dedupe?: FrameDedupeOptions;
|
|
630
671
|
}
|
|
631
672
|
/** Supported image MIME types accepted by all providers. */
|
|
632
673
|
type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
|
|
@@ -793,7 +834,51 @@ interface VideoSamplingOptions {
|
|
|
793
834
|
* Default `10`.
|
|
794
835
|
*/
|
|
795
836
|
maxDurationSeconds?: number;
|
|
837
|
+
/**
|
|
838
|
+
* Drop sampled frames that did not visibly change from the preceding kept
|
|
839
|
+
* frame before sending to the provider. Default `true`. See `FrameDedupeOptions`.
|
|
840
|
+
* Only applies when frames are sampled; ignored for native delivery.
|
|
841
|
+
*/
|
|
842
|
+
dedupe?: FrameDedupeOptions;
|
|
843
|
+
/**
|
|
844
|
+
* How the video reaches the model. Default `"auto"`. See `VideoDeliveryMode`.
|
|
845
|
+
*/
|
|
846
|
+
mode?: VideoDeliveryMode;
|
|
796
847
|
}
|
|
848
|
+
/**
|
|
849
|
+
* How a video input is delivered to the model.
|
|
850
|
+
*
|
|
851
|
+
* - `"auto"` (default): send the video itself when the provider accepts video
|
|
852
|
+
* natively (Google models), otherwise sample frames with ffmpeg.
|
|
853
|
+
* - `"native"`: always send the video itself. Throws `VisualAIConfigError`
|
|
854
|
+
* when the provider has no native video support.
|
|
855
|
+
* - `"frames"`: always sample frames with ffmpeg, whatever the provider.
|
|
856
|
+
*
|
|
857
|
+
* Native delivery still probes the duration and enforces `maxDurationSeconds`
|
|
858
|
+
* before any provider call, and passes `fps` on as the provider's sampling
|
|
859
|
+
* rate. `maxFrames` and `dedupe` apply to frame sampling only. Pre-sampled
|
|
860
|
+
* `FramesInput` is always sent as frames.
|
|
861
|
+
*/
|
|
862
|
+
type VideoDeliveryMode = "auto" | "native" | "frames";
|
|
863
|
+
/**
|
|
864
|
+
* Controls dropping of frames that did not visibly change from the previous
|
|
865
|
+
* kept frame, so a static screen does not cost input tokens for every sample.
|
|
866
|
+
*
|
|
867
|
+
* - `true` (default): drop unchanged frames using the default threshold.
|
|
868
|
+
* - `false`: send every sampled frame.
|
|
869
|
+
* - `{ threshold }`: the fraction `(0, 1]` of a frame's pixels that must differ
|
|
870
|
+
* from the last kept frame for it to count as changed. Default `0.001` (0.1%,
|
|
871
|
+
* roughly a 37x37 px region on a 1568x880 frame). Lower it to keep smaller
|
|
872
|
+
* changes; raise it to ignore more.
|
|
873
|
+
*
|
|
874
|
+
* Each frame is compared against the most recently *kept* frame, so gradual
|
|
875
|
+
* drift accumulates and is eventually kept. The first frame is always kept.
|
|
876
|
+
* Pixel-level compression noise and tiny flickers such as a blinking text
|
|
877
|
+
* caret fall below the default threshold.
|
|
878
|
+
*/
|
|
879
|
+
type FrameDedupeOptions = boolean | {
|
|
880
|
+
threshold?: number;
|
|
881
|
+
};
|
|
797
882
|
/**
|
|
798
883
|
* A single frame extracted from a video input. Identical in shape to
|
|
799
884
|
* `NormalizedImage` so it can be passed transparently to provider drivers.
|
|
@@ -819,12 +904,14 @@ interface VisualAIClient {
|
|
|
819
904
|
* Verifies one or more statements against a single image or video.
|
|
820
905
|
*
|
|
821
906
|
* Pass an image (PNG/JPEG/WebP/GIF) for a single-frame check. Pass a video
|
|
822
|
-
* (MP4/WebM/MOV/MKV file path,
|
|
823
|
-
*
|
|
824
|
-
*
|
|
825
|
-
*
|
|
826
|
-
* the
|
|
827
|
-
*
|
|
907
|
+
* (MP4/WebM/MOV/MKV file path, base64, Buffer) and statements pass if they
|
|
908
|
+
* are true at any point, with each statement result carrying the timestamp
|
|
909
|
+
* where it matched. On providers that accept video natively (Google models)
|
|
910
|
+
* the video itself is sent and the result's `video` metadata describes the
|
|
911
|
+
* delivery; elsewhere the client samples frames with ffmpeg and the `frames`
|
|
912
|
+
* metadata reports which timestamps the model saw. Control this with
|
|
913
|
+
* `video.mode`. Pass a `FramesInput` (`{ frames, fps? }`) to supply
|
|
914
|
+
* pre-sampled frames directly — handled identically to a sampled timeline but
|
|
828
915
|
* without loading ffmpeg.
|
|
829
916
|
*
|
|
830
917
|
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
@@ -854,11 +941,13 @@ interface VisualAIClient {
|
|
|
854
941
|
/**
|
|
855
942
|
* Asks an open-ended question about an image or video and returns a structured summary.
|
|
856
943
|
*
|
|
857
|
-
* Video inputs are
|
|
858
|
-
*
|
|
859
|
-
*
|
|
860
|
-
*
|
|
861
|
-
*
|
|
944
|
+
* Video inputs are analyzed as a chronological timeline. On providers that
|
|
945
|
+
* accept video natively (Google models) the video itself is sent and the
|
|
946
|
+
* result's `timestampReferences` array surfaces the moments the model relied
|
|
947
|
+
* on; elsewhere frames are sampled with ffmpeg and `frameReferences` indexes
|
|
948
|
+
* into `frames.timestampsSeconds`. Control this with `video.mode`. Pass a
|
|
949
|
+
* `FramesInput` (`{ frames, fps? }`) to supply pre-sampled frames directly —
|
|
950
|
+
* handled identically to a sampled timeline but without loading ffmpeg.
|
|
862
951
|
*
|
|
863
952
|
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
864
953
|
* @param prompt Prompt describing what to inspect in the input.
|
|
@@ -1249,4 +1338,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
|
|
|
1249
1338
|
*/
|
|
1250
1339
|
declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
|
|
1251
1340
|
|
|
1252
|
-
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
|
1341
|
+
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FrameDedupeOptions, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type NativeVideoMetadata, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoDeliveryMode, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
package/dist/index.d.ts
CHANGED
|
@@ -79,6 +79,7 @@ declare const Model: {
|
|
|
79
79
|
readonly QWEN_3_8_MAX: "qwen/qwen3.8-max";
|
|
80
80
|
readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
|
|
81
81
|
readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
|
|
82
|
+
readonly GLM_5_3_FLASH: "z-ai/glm-5.3-flash";
|
|
82
83
|
};
|
|
83
84
|
};
|
|
84
85
|
/** Union of all built-in model name literals exposed by `Model`. */
|
|
@@ -367,17 +368,39 @@ declare const CheckResultSchema: z.ZodObject<{
|
|
|
367
368
|
* Populated client-side; not part of the model's response.
|
|
368
369
|
*/
|
|
369
370
|
interface VideoFramesMetadata {
|
|
370
|
-
/**
|
|
371
|
+
/** Number of frames actually sent to the model (after unchanged frames were dropped). */
|
|
371
372
|
count: number;
|
|
372
|
-
/** Timestamp (seconds, from the start of the clip) of each
|
|
373
|
+
/** Timestamp (seconds, from the start of the clip) of each sent frame, in order. */
|
|
373
374
|
timestampsSeconds: number[];
|
|
374
375
|
/** Total duration of the source video in seconds. */
|
|
375
376
|
durationSeconds: number;
|
|
377
|
+
/**
|
|
378
|
+
* Number of sampled frames dropped because they did not visibly change from
|
|
379
|
+
* the preceding kept frame. `0` when dedupe is disabled or nothing was dropped.
|
|
380
|
+
* See `FrameDedupeOptions`.
|
|
381
|
+
*/
|
|
382
|
+
droppedUnchanged: number;
|
|
383
|
+
}
|
|
384
|
+
/**
|
|
385
|
+
* Metadata describing a video that was delivered to the model natively (as the
|
|
386
|
+
* video itself rather than sampled frames). Populated client-side.
|
|
387
|
+
*/
|
|
388
|
+
interface NativeVideoMetadata {
|
|
389
|
+
/** Total duration of the source video in seconds. */
|
|
390
|
+
durationSeconds: number;
|
|
391
|
+
/** Sampling rate requested from the provider, in frames per second. */
|
|
392
|
+
fps: number;
|
|
393
|
+
/** MIME type the video was sent as. */
|
|
394
|
+
mimeType: SupportedVideoMimeType;
|
|
395
|
+
/** Whether the bytes went inline in the request or through the provider's file upload API. */
|
|
396
|
+
delivery: "inline" | "file";
|
|
376
397
|
}
|
|
377
398
|
/** Result returned by `check()` and the template convenience methods. */
|
|
378
399
|
type CheckResult = z.infer<typeof CheckResultSchema> & {
|
|
379
|
-
/** Present only when the input was a video. Describes which frames the model saw. */
|
|
400
|
+
/** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
|
|
380
401
|
frames?: VideoFramesMetadata;
|
|
402
|
+
/** Present only when the input was a video delivered natively to the provider. */
|
|
403
|
+
video?: NativeVideoMetadata;
|
|
381
404
|
};
|
|
382
405
|
/** Zod schema for an individual visual change reported by `compare()`. */
|
|
383
406
|
declare const ChangeEntrySchema: z.ZodObject<{
|
|
@@ -513,6 +536,12 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
513
536
|
* omitting the key, even for image inputs that were never asked to populate it.
|
|
514
537
|
*/
|
|
515
538
|
frameReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
539
|
+
/**
|
|
540
|
+
* For natively delivered video, the timestamps (seconds from the start of
|
|
541
|
+
* the clip) the model relied on to answer. The native counterpart of
|
|
542
|
+
* `frameReferences`. Nullable for the same strict-schema reason.
|
|
543
|
+
*/
|
|
544
|
+
timestampReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
|
|
516
545
|
usage: z.ZodOptional<z.ZodObject<{
|
|
517
546
|
inputTokens: z.ZodNumber;
|
|
518
547
|
outputTokens: z.ZodNumber;
|
|
@@ -565,6 +594,7 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
565
594
|
durationSeconds?: number | undefined;
|
|
566
595
|
} | undefined;
|
|
567
596
|
frameReferences?: number[] | null | undefined;
|
|
597
|
+
timestampReferences?: number[] | null | undefined;
|
|
568
598
|
}, {
|
|
569
599
|
issues: {
|
|
570
600
|
priority: "critical" | "major" | "minor";
|
|
@@ -583,12 +613,18 @@ declare const AskResultSchema: z.ZodObject<{
|
|
|
583
613
|
durationSeconds?: number | undefined;
|
|
584
614
|
} | undefined;
|
|
585
615
|
frameReferences?: number[] | null | undefined;
|
|
616
|
+
timestampReferences?: number[] | null | undefined;
|
|
586
617
|
}>;
|
|
587
618
|
/** Result returned by `ask()`. */
|
|
588
|
-
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences"> & {
|
|
589
|
-
/** Present only when the input was a video
|
|
619
|
+
type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences" | "timestampReferences"> & {
|
|
620
|
+
/** Present only when the input was a video sampled into frames. Indices into `frames.timestampsSeconds`. */
|
|
590
621
|
frameReferences?: number[];
|
|
622
|
+
/** Present only when the input was a video delivered natively. Seconds from the start of the clip. */
|
|
623
|
+
timestampReferences?: number[];
|
|
624
|
+
/** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
|
|
591
625
|
frames?: VideoFramesMetadata;
|
|
626
|
+
/** Present only when the input was a video delivered natively to the provider. */
|
|
627
|
+
video?: NativeVideoMetadata;
|
|
592
628
|
};
|
|
593
629
|
/** Supported input shapes for image arguments accepted by the client. */
|
|
594
630
|
type ImageInput = Buffer | Uint8Array | string;
|
|
@@ -627,6 +663,11 @@ interface FramesInput {
|
|
|
627
663
|
* `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
|
|
628
664
|
*/
|
|
629
665
|
fps?: number;
|
|
666
|
+
/**
|
|
667
|
+
* Drop frames that did not visibly change from the preceding kept frame
|
|
668
|
+
* before sending to the provider. Default `true`. See `FrameDedupeOptions`.
|
|
669
|
+
*/
|
|
670
|
+
dedupe?: FrameDedupeOptions;
|
|
630
671
|
}
|
|
631
672
|
/** Supported image MIME types accepted by all providers. */
|
|
632
673
|
type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
|
|
@@ -793,7 +834,51 @@ interface VideoSamplingOptions {
|
|
|
793
834
|
* Default `10`.
|
|
794
835
|
*/
|
|
795
836
|
maxDurationSeconds?: number;
|
|
837
|
+
/**
|
|
838
|
+
* Drop sampled frames that did not visibly change from the preceding kept
|
|
839
|
+
* frame before sending to the provider. Default `true`. See `FrameDedupeOptions`.
|
|
840
|
+
* Only applies when frames are sampled; ignored for native delivery.
|
|
841
|
+
*/
|
|
842
|
+
dedupe?: FrameDedupeOptions;
|
|
843
|
+
/**
|
|
844
|
+
* How the video reaches the model. Default `"auto"`. See `VideoDeliveryMode`.
|
|
845
|
+
*/
|
|
846
|
+
mode?: VideoDeliveryMode;
|
|
796
847
|
}
|
|
848
|
+
/**
|
|
849
|
+
* How a video input is delivered to the model.
|
|
850
|
+
*
|
|
851
|
+
* - `"auto"` (default): send the video itself when the provider accepts video
|
|
852
|
+
* natively (Google models), otherwise sample frames with ffmpeg.
|
|
853
|
+
* - `"native"`: always send the video itself. Throws `VisualAIConfigError`
|
|
854
|
+
* when the provider has no native video support.
|
|
855
|
+
* - `"frames"`: always sample frames with ffmpeg, whatever the provider.
|
|
856
|
+
*
|
|
857
|
+
* Native delivery still probes the duration and enforces `maxDurationSeconds`
|
|
858
|
+
* before any provider call, and passes `fps` on as the provider's sampling
|
|
859
|
+
* rate. `maxFrames` and `dedupe` apply to frame sampling only. Pre-sampled
|
|
860
|
+
* `FramesInput` is always sent as frames.
|
|
861
|
+
*/
|
|
862
|
+
type VideoDeliveryMode = "auto" | "native" | "frames";
|
|
863
|
+
/**
|
|
864
|
+
* Controls dropping of frames that did not visibly change from the previous
|
|
865
|
+
* kept frame, so a static screen does not cost input tokens for every sample.
|
|
866
|
+
*
|
|
867
|
+
* - `true` (default): drop unchanged frames using the default threshold.
|
|
868
|
+
* - `false`: send every sampled frame.
|
|
869
|
+
* - `{ threshold }`: the fraction `(0, 1]` of a frame's pixels that must differ
|
|
870
|
+
* from the last kept frame for it to count as changed. Default `0.001` (0.1%,
|
|
871
|
+
* roughly a 37x37 px region on a 1568x880 frame). Lower it to keep smaller
|
|
872
|
+
* changes; raise it to ignore more.
|
|
873
|
+
*
|
|
874
|
+
* Each frame is compared against the most recently *kept* frame, so gradual
|
|
875
|
+
* drift accumulates and is eventually kept. The first frame is always kept.
|
|
876
|
+
* Pixel-level compression noise and tiny flickers such as a blinking text
|
|
877
|
+
* caret fall below the default threshold.
|
|
878
|
+
*/
|
|
879
|
+
type FrameDedupeOptions = boolean | {
|
|
880
|
+
threshold?: number;
|
|
881
|
+
};
|
|
797
882
|
/**
|
|
798
883
|
* A single frame extracted from a video input. Identical in shape to
|
|
799
884
|
* `NormalizedImage` so it can be passed transparently to provider drivers.
|
|
@@ -819,12 +904,14 @@ interface VisualAIClient {
|
|
|
819
904
|
* Verifies one or more statements against a single image or video.
|
|
820
905
|
*
|
|
821
906
|
* Pass an image (PNG/JPEG/WebP/GIF) for a single-frame check. Pass a video
|
|
822
|
-
* (MP4/WebM/MOV/MKV file path,
|
|
823
|
-
*
|
|
824
|
-
*
|
|
825
|
-
*
|
|
826
|
-
* the
|
|
827
|
-
*
|
|
907
|
+
* (MP4/WebM/MOV/MKV file path, base64, Buffer) and statements pass if they
|
|
908
|
+
* are true at any point, with each statement result carrying the timestamp
|
|
909
|
+
* where it matched. On providers that accept video natively (Google models)
|
|
910
|
+
* the video itself is sent and the result's `video` metadata describes the
|
|
911
|
+
* delivery; elsewhere the client samples frames with ffmpeg and the `frames`
|
|
912
|
+
* metadata reports which timestamps the model saw. Control this with
|
|
913
|
+
* `video.mode`. Pass a `FramesInput` (`{ frames, fps? }`) to supply
|
|
914
|
+
* pre-sampled frames directly — handled identically to a sampled timeline but
|
|
828
915
|
* without loading ffmpeg.
|
|
829
916
|
*
|
|
830
917
|
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
@@ -854,11 +941,13 @@ interface VisualAIClient {
|
|
|
854
941
|
/**
|
|
855
942
|
* Asks an open-ended question about an image or video and returns a structured summary.
|
|
856
943
|
*
|
|
857
|
-
* Video inputs are
|
|
858
|
-
*
|
|
859
|
-
*
|
|
860
|
-
*
|
|
861
|
-
*
|
|
944
|
+
* Video inputs are analyzed as a chronological timeline. On providers that
|
|
945
|
+
* accept video natively (Google models) the video itself is sent and the
|
|
946
|
+
* result's `timestampReferences` array surfaces the moments the model relied
|
|
947
|
+
* on; elsewhere frames are sampled with ffmpeg and `frameReferences` indexes
|
|
948
|
+
* into `frames.timestampsSeconds`. Control this with `video.mode`. Pass a
|
|
949
|
+
* `FramesInput` (`{ frames, fps? }`) to supply pre-sampled frames directly —
|
|
950
|
+
* handled identically to a sampled timeline but without loading ffmpeg.
|
|
862
951
|
*
|
|
863
952
|
* @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
|
|
864
953
|
* @param prompt Prompt describing what to inspect in the input.
|
|
@@ -1249,4 +1338,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
|
|
|
1249
1338
|
*/
|
|
1250
1339
|
declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
|
|
1251
1340
|
|
|
1252
|
-
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|
|
1341
|
+
export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FrameDedupeOptions, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type NativeVideoMetadata, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoDeliveryMode, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
|