visual-ai-assertions 0.23.0 → 0.26.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -1,7 +1,21 @@
1
1
  import { z } from 'zod';
2
2
 
3
3
  /** Supported reasoning effort levels. */
4
+ /**
5
+ * Abstract reasoning-effort hint. Each provider driver maps it to its native
6
+ * mechanism; levels a provider lacks clamp to its nearest tier.
7
+ *
8
+ * `minimal` is **not universally accepted**. OpenAI rejects it per-model with
9
+ * HTTP 400 ("Unsupported value: 'minimal' is not supported with the
10
+ * '<model>' model"), observed on `gpt-6-astra` and `gpt-6.1-sol`, whose
11
+ * supported set is low/medium/high/xhigh/max, and on `gpt-6-sol` and
12
+ * `gpt-6-luna`, which accept none/low/medium/high/xhigh/max. Gemini defines
13
+ * the tier but some models (e.g. Gemini 3.1 Pro) reject it, so Google and
14
+ * OpenRouter clamp it to `low` rather than send it. Use `minimal` only
15
+ * against an OpenAI model known to accept it; `low` is the portable floor.
16
+ */
4
17
  declare const ReasoningEffort: {
18
+ readonly MINIMAL: "minimal";
5
19
  readonly LOW: "low";
6
20
  readonly MEDIUM: "medium";
7
21
  readonly HIGH: "high";
@@ -34,16 +48,21 @@ declare const Model: {
34
48
  readonly Anthropic: {
35
49
  readonly FABLE_5_1: "claude-fable-5-1";
36
50
  readonly FABLE_5: "claude-fable-5";
51
+ readonly OPUS_5_5: "claude-opus-5-5";
37
52
  readonly OPUS_5: "claude-opus-5";
38
53
  readonly OPUS_4_8: "claude-opus-4-8";
39
54
  readonly OPUS_4_7: "claude-opus-4-7";
40
55
  readonly OPUS_4_6: "claude-opus-4-6";
56
+ readonly SONNET_5_5: "claude-sonnet-5-5";
41
57
  readonly SONNET_5: "claude-sonnet-5";
42
58
  readonly SONNET_4_6: "claude-sonnet-4-6";
43
59
  readonly HAIKU_4_5: "claude-haiku-4-5";
44
60
  };
45
61
  readonly OpenAI: {
46
62
  readonly GPT_6_ASTRA: "gpt-6-astra";
63
+ readonly GPT_6_1_SOL: "gpt-6.1-sol";
64
+ readonly GPT_6_SOL: "gpt-6-sol";
65
+ readonly GPT_6_LUNA: "gpt-6-luna";
47
66
  readonly GPT_5_6_SOL: "gpt-5.6-sol";
48
67
  readonly GPT_5_6_TERRA: "gpt-5.6-terra";
49
68
  readonly GPT_5_6_LUNA: "gpt-5.6-luna";
@@ -72,6 +91,7 @@ declare const Model: {
72
91
  */
73
92
  readonly OpenRouter: {
74
93
  readonly MUSE_SPARK_1_3: "meta/muse-spark-1.3";
94
+ readonly GROK_4_7: "x-ai/grok-4.7";
75
95
  readonly GROK_4_6: "x-ai/grok-4.6";
76
96
  readonly GROK_4_5: "x-ai/grok-4.5";
77
97
  readonly KIMI_K3: "moonshotai/kimi-k3";
@@ -80,16 +100,28 @@ declare const Model: {
80
100
  readonly QWEN_3_7_PLUS: "qwen/qwen3.7-plus";
81
101
  readonly QWEN_3_6_FLASH: "qwen/qwen3.6-flash";
82
102
  readonly GLM_5_3_FLASH: "z-ai/glm-5.3-flash";
103
+ readonly MIMO_V2_6_PRO: "xiaomi/mimo-v2.6-pro";
83
104
  };
84
105
  };
85
106
  /** Union of all built-in model name literals exposed by `Model`. */
86
107
  type KnownModelName = (typeof Model.Anthropic)[keyof typeof Model.Anthropic] | (typeof Model.OpenAI)[keyof typeof Model.OpenAI] | (typeof Model.Google)[keyof typeof Model.Google] | (typeof Model.OpenRouter)[keyof typeof Model.OpenRouter];
87
- /** Default model selection used when a caller omits `config.model`. */
108
+ /**
109
+ * Default model selection used when a caller omits `config.model`.
110
+ *
111
+ * Two of these carry constraints the older defaults did not:
112
+ * - `gpt-6.1-sol` rejects `reasoningEffort: "minimal"` (and `none`) with HTTP
113
+ * 400; `low` is the floor. See the `ReasoningEffort` note above.
114
+ * - `meta/muse-spark-1.3` is age-gated on OpenRouter and returns HTTP 403 until
115
+ * the account completes the 18+ confirmation at
116
+ * openrouter.ai/settings/preferences. It also reasons by default (~370-814
117
+ * reasoning tokens per call) even with no effort configured, so cost per call
118
+ * runs above what its headline rate suggests.
119
+ */
88
120
  declare const DEFAULT_MODELS: {
89
- readonly anthropic: "claude-sonnet-4-6";
90
- readonly openai: "gpt-5.6-luna";
91
- readonly google: "gemini-3-flash-preview";
92
- readonly openrouter: "qwen/qwen3.6-flash";
121
+ readonly anthropic: "claude-sonnet-5-5";
122
+ readonly openai: "gpt-6.1-sol";
123
+ readonly google: "gemini-3.8-flash";
124
+ readonly openrouter: "meta/muse-spark-1.3";
93
125
  };
94
126
  /** Built-in content checks available through `client.content()`. */
95
127
  declare const Content: {
@@ -178,13 +210,13 @@ declare const StatementResultSchema: z.ZodObject<{
178
210
  statement: string;
179
211
  pass: boolean;
180
212
  reasoning: string;
181
- confidence?: "high" | "medium" | "low" | undefined;
213
+ confidence?: "low" | "medium" | "high" | undefined;
182
214
  timestampSeconds?: number | null | undefined;
183
215
  }, {
184
216
  statement: string;
185
217
  pass: boolean;
186
218
  reasoning: string;
187
- confidence?: "high" | "medium" | "low" | undefined;
219
+ confidence?: "low" | "medium" | "high" | undefined;
188
220
  timestampSeconds?: number | null | undefined;
189
221
  }>;
190
222
  /** Outcome of a single statement evaluated by `check()`. */
@@ -303,13 +335,13 @@ declare const CheckResultSchema: z.ZodObject<{
303
335
  statement: string;
304
336
  pass: boolean;
305
337
  reasoning: string;
306
- confidence?: "high" | "medium" | "low" | undefined;
338
+ confidence?: "low" | "medium" | "high" | undefined;
307
339
  timestampSeconds?: number | null | undefined;
308
340
  }, {
309
341
  statement: string;
310
342
  pass: boolean;
311
343
  reasoning: string;
312
- confidence?: "high" | "medium" | "low" | undefined;
344
+ confidence?: "low" | "medium" | "high" | undefined;
313
345
  timestampSeconds?: number | null | undefined;
314
346
  }>, "many">;
315
347
  }, "strip", z.ZodTypeAny, {
@@ -325,7 +357,7 @@ declare const CheckResultSchema: z.ZodObject<{
325
357
  statement: string;
326
358
  pass: boolean;
327
359
  reasoning: string;
328
- confidence?: "high" | "medium" | "low" | undefined;
360
+ confidence?: "low" | "medium" | "high" | undefined;
329
361
  timestampSeconds?: number | null | undefined;
330
362
  }[];
331
363
  usage?: {
@@ -350,7 +382,7 @@ declare const CheckResultSchema: z.ZodObject<{
350
382
  statement: string;
351
383
  pass: boolean;
352
384
  reasoning: string;
353
- confidence?: "high" | "medium" | "low" | undefined;
385
+ confidence?: "low" | "medium" | "high" | undefined;
354
386
  timestampSeconds?: number | null | undefined;
355
387
  }[];
356
388
  usage?: {
@@ -368,17 +400,39 @@ declare const CheckResultSchema: z.ZodObject<{
368
400
  * Populated client-side; not part of the model's response.
369
401
  */
370
402
  interface VideoFramesMetadata {
371
- /** Total number of frames sampled from the video. */
403
+ /** Number of frames actually sent to the model (after unchanged frames were dropped). */
372
404
  count: number;
373
- /** Timestamp (seconds, from the start of the clip) of each sampled frame, in order. */
405
+ /** Timestamp (seconds, from the start of the clip) of each sent frame, in order. */
374
406
  timestampsSeconds: number[];
375
407
  /** Total duration of the source video in seconds. */
376
408
  durationSeconds: number;
409
+ /**
410
+ * Number of sampled frames dropped because they did not visibly change from
411
+ * the preceding kept frame. `0` when dedupe is disabled or nothing was dropped.
412
+ * See `FrameDedupeOptions`.
413
+ */
414
+ droppedUnchanged: number;
415
+ }
416
+ /**
417
+ * Metadata describing a video that was delivered to the model natively (as the
418
+ * video itself rather than sampled frames). Populated client-side.
419
+ */
420
+ interface NativeVideoMetadata {
421
+ /** Total duration of the source video in seconds. */
422
+ durationSeconds: number;
423
+ /** Sampling rate requested from the provider, in frames per second. */
424
+ fps: number;
425
+ /** MIME type the video was sent as. */
426
+ mimeType: SupportedVideoMimeType;
427
+ /** Whether the bytes went inline in the request or through the provider's file upload API. */
428
+ delivery: "inline" | "file";
377
429
  }
378
430
  /** Result returned by `check()` and the template convenience methods. */
379
431
  type CheckResult = z.infer<typeof CheckResultSchema> & {
380
- /** Present only when the input was a video. Describes which frames the model saw. */
432
+ /** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
381
433
  frames?: VideoFramesMetadata;
434
+ /** Present only when the input was a video delivered natively to the provider. */
435
+ video?: NativeVideoMetadata;
382
436
  };
383
437
  /** Zod schema for an individual visual change reported by `compare()`. */
384
438
  declare const ChangeEntrySchema: z.ZodObject<{
@@ -514,6 +568,12 @@ declare const AskResultSchema: z.ZodObject<{
514
568
  * omitting the key, even for image inputs that were never asked to populate it.
515
569
  */
516
570
  frameReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
571
+ /**
572
+ * For natively delivered video, the timestamps (seconds from the start of
573
+ * the clip) the model relied on to answer. The native counterpart of
574
+ * `frameReferences`. Nullable for the same strict-schema reason.
575
+ */
576
+ timestampReferences: z.ZodOptional<z.ZodNullable<z.ZodArray<z.ZodNumber, "many">>>;
517
577
  usage: z.ZodOptional<z.ZodObject<{
518
578
  inputTokens: z.ZodNumber;
519
579
  outputTokens: z.ZodNumber;
@@ -566,6 +626,7 @@ declare const AskResultSchema: z.ZodObject<{
566
626
  durationSeconds?: number | undefined;
567
627
  } | undefined;
568
628
  frameReferences?: number[] | null | undefined;
629
+ timestampReferences?: number[] | null | undefined;
569
630
  }, {
570
631
  issues: {
571
632
  priority: "critical" | "major" | "minor";
@@ -584,12 +645,18 @@ declare const AskResultSchema: z.ZodObject<{
584
645
  durationSeconds?: number | undefined;
585
646
  } | undefined;
586
647
  frameReferences?: number[] | null | undefined;
648
+ timestampReferences?: number[] | null | undefined;
587
649
  }>;
588
650
  /** Result returned by `ask()`. */
589
- type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences"> & {
590
- /** Present only when the input was a video. Describes which frames the model saw. */
651
+ type AskResult = Omit<z.infer<typeof AskResultSchema>, "frameReferences" | "timestampReferences"> & {
652
+ /** Present only when the input was a video sampled into frames. Indices into `frames.timestampsSeconds`. */
591
653
  frameReferences?: number[];
654
+ /** Present only when the input was a video delivered natively. Seconds from the start of the clip. */
655
+ timestampReferences?: number[];
656
+ /** Present only when the input was a video sampled into frames. Describes which frames the model saw. */
592
657
  frames?: VideoFramesMetadata;
658
+ /** Present only when the input was a video delivered natively to the provider. */
659
+ video?: NativeVideoMetadata;
593
660
  };
594
661
  /** Supported input shapes for image arguments accepted by the client. */
595
662
  type ImageInput = Buffer | Uint8Array | string;
@@ -628,6 +695,11 @@ interface FramesInput {
628
695
  * `timestampSeconds` (frame `i` maps to `i / fps` seconds). Default `1`.
629
696
  */
630
697
  fps?: number;
698
+ /**
699
+ * Drop frames that did not visibly change from the preceding kept frame
700
+ * before sending to the provider. Default `true`. See `FrameDedupeOptions`.
701
+ */
702
+ dedupe?: FrameDedupeOptions;
631
703
  }
632
704
  /** Supported image MIME types accepted by all providers. */
633
705
  type SupportedMimeType = "image/jpeg" | "image/png" | "image/webp" | "image/gif";
@@ -794,7 +866,51 @@ interface VideoSamplingOptions {
794
866
  * Default `10`.
795
867
  */
796
868
  maxDurationSeconds?: number;
869
+ /**
870
+ * Drop sampled frames that did not visibly change from the preceding kept
871
+ * frame before sending to the provider. Default `true`. See `FrameDedupeOptions`.
872
+ * Only applies when frames are sampled; ignored for native delivery.
873
+ */
874
+ dedupe?: FrameDedupeOptions;
875
+ /**
876
+ * How the video reaches the model. Default `"auto"`. See `VideoDeliveryMode`.
877
+ */
878
+ mode?: VideoDeliveryMode;
797
879
  }
880
+ /**
881
+ * How a video input is delivered to the model.
882
+ *
883
+ * - `"auto"` (default): send the video itself when the provider accepts video
884
+ * natively (Google models), otherwise sample frames with ffmpeg.
885
+ * - `"native"`: always send the video itself. Throws `VisualAIConfigError`
886
+ * when the provider has no native video support.
887
+ * - `"frames"`: always sample frames with ffmpeg, whatever the provider.
888
+ *
889
+ * Native delivery still probes the duration and enforces `maxDurationSeconds`
890
+ * before any provider call, and passes `fps` on as the provider's sampling
891
+ * rate. `maxFrames` and `dedupe` apply to frame sampling only. Pre-sampled
892
+ * `FramesInput` is always sent as frames.
893
+ */
894
+ type VideoDeliveryMode = "auto" | "native" | "frames";
895
+ /**
896
+ * Controls dropping of frames that did not visibly change from the previous
897
+ * kept frame, so a static screen does not cost input tokens for every sample.
898
+ *
899
+ * - `true` (default): drop unchanged frames using the default threshold.
900
+ * - `false`: send every sampled frame.
901
+ * - `{ threshold }`: the fraction `(0, 1]` of a frame's pixels that must differ
902
+ * from the last kept frame for it to count as changed. Default `0.001` (0.1%,
903
+ * roughly a 37x37 px region on a 1568x880 frame). Lower it to keep smaller
904
+ * changes; raise it to ignore more.
905
+ *
906
+ * Each frame is compared against the most recently *kept* frame, so gradual
907
+ * drift accumulates and is eventually kept. The first frame is always kept.
908
+ * Pixel-level compression noise and tiny flickers such as a blinking text
909
+ * caret fall below the default threshold.
910
+ */
911
+ type FrameDedupeOptions = boolean | {
912
+ threshold?: number;
913
+ };
798
914
  /**
799
915
  * A single frame extracted from a video input. Identical in shape to
800
916
  * `NormalizedImage` so it can be passed transparently to provider drivers.
@@ -820,12 +936,14 @@ interface VisualAIClient {
820
936
  * Verifies one or more statements against a single image or video.
821
937
  *
822
938
  * Pass an image (PNG/JPEG/WebP/GIF) for a single-frame check. Pass a video
823
- * (MP4/WebM/MOV/MKV file path, URL, base64, Buffer) and the client samples
824
- * frames automatically; statements pass if they are true at any sampled
825
- * frame, and each statement result includes the timestamp where it
826
- * matched. The `frames` metadata on the result reports which timestamps
827
- * the model saw. Pass a `FramesInput` (`{ frames, fps? }`) to supply
828
- * pre-sampled frames directly — handled identically to a video timeline but
939
+ * (MP4/WebM/MOV/MKV file path, base64, Buffer) and statements pass if they
940
+ * are true at any point, with each statement result carrying the timestamp
941
+ * where it matched. On providers that accept video natively (Google models)
942
+ * the video itself is sent and the result's `video` metadata describes the
943
+ * delivery; elsewhere the client samples frames with ffmpeg and the `frames`
944
+ * metadata reports which timestamps the model saw. Control this with
945
+ * `video.mode`. Pass a `FramesInput` (`{ frames, fps? }`) to supply
946
+ * pre-sampled frames directly — handled identically to a sampled timeline but
829
947
  * without loading ffmpeg.
830
948
  *
831
949
  * @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
@@ -855,11 +973,13 @@ interface VisualAIClient {
855
973
  /**
856
974
  * Asks an open-ended question about an image or video and returns a structured summary.
857
975
  *
858
- * Video inputs are sampled into frames and analyzed as a chronological
859
- * timeline. The result's `frameReferences` array surfaces which frames the
860
- * model relied on for its answer. Pass a `FramesInput` (`{ frames, fps? }`)
861
- * to supply pre-sampled frames directly — handled identically to a video
862
- * timeline but without loading ffmpeg.
976
+ * Video inputs are analyzed as a chronological timeline. On providers that
977
+ * accept video natively (Google models) the video itself is sent and the
978
+ * result's `timestampReferences` array surfaces the moments the model relied
979
+ * on; elsewhere frames are sampled with ffmpeg and `frameReferences` indexes
980
+ * into `frames.timestampsSeconds`. Control this with `video.mode`. Pass a
981
+ * `FramesInput` (`{ frames, fps? }`) to supply pre-sampled frames directly —
982
+ * handled identically to a sampled timeline but without loading ffmpeg.
863
983
  *
864
984
  * @param input Image or video source as a buffer, URL, file path, or base64 string, or a `FramesInput` of pre-sampled frames.
865
985
  * @param prompt Prompt describing what to inspect in the input.
@@ -880,8 +1000,9 @@ interface VisualAIClient {
880
1000
  * @param imageA Baseline image source.
881
1001
  * @param imageB Candidate image source.
882
1002
  * @param options Optional comparison prompt, instructions, and diff-image settings.
883
- * `gemini-3-flash-preview` generates an annotated diff image by default;
884
- * pass `{ diffImage: false }` to opt out.
1003
+ * Gemini flash models (`DIFF_ALLOWED_MODELS`) generate an annotated diff image
1004
+ * by default; pass `{ diffImage: false }` to opt out. Flash-Lite and Pro tiers
1005
+ * produce none even when asked.
885
1006
  * @returns A structured comparison result with optional diff image metadata.
886
1007
  * @throws {VisualAIImageError} When either image cannot be loaded or decoded.
887
1008
  * @throws {VisualAIError} When the provider rejects the request or returns invalid output.
@@ -1250,4 +1371,4 @@ declare function assertVisualResult(result: CheckResult, label?: string): void;
1250
1371
  */
1251
1372
  declare function assertVisualCompareResult(result: CompareResult, label?: string): void;
1252
1373
 
1253
- export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };
1374
+ export { Accessibility, type AccessibilityCheckName, type AccessibilityOptions, type AskOptions, type AskResult, AskResultSchema, type ChangeEntry, ChangeEntrySchema, type CheckOptions, type CheckResult, CheckResultSchema, type CompareOptions, type CompareResult, CompareResultSchema, type Confidence, ConfidenceSchema, Content, type ContentCheckName, type ContentOptions, DEFAULT_MODELS, type DiffImageResult, type ElementsVisibilityOptions, type Frame, type FrameDedupeOptions, type FramesInput, ImageDetail, type ImageDetailLevel, type ImageInput, type Issue, type IssueCategory, IssueCategorySchema, type IssuePriority, IssuePrioritySchema, IssueSchema, type KnownModelName, Layout, type LayoutCheckName, type LayoutOptions, type MediaInput, Model, type NativeVideoMetadata, type PageLoadOptions, Provider, type ProviderName, ReasoningEffort, type ReasoningEffortLevel, type StatementResult, StatementResultSchema, type SupportedMimeType, type SupportedVideoMimeType, type TimestampedFrameInput, type UsageInfo, UsageInfoSchema, type VideoDeliveryMode, type VideoFramesMetadata, type VideoSamplingOptions, VisualAIAssertionError, VisualAIAuthError, type VisualAIClient, type VisualAIConfig, VisualAIConfigError, VisualAIError, type VisualAIErrorCode, VisualAIImageError, type VisualAIKnownError, VisualAIProviderError, VisualAIRateLimitError, VisualAIResponseParseError, VisualAITruncationError, VisualAIVideoError, assertVisualCompareResult, assertVisualResult, formatCheckResult, formatCompareResult, isVisualAIKnownError, visualAI };