@framefields/vision 0.0.0-stage → 2.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1198 @@
1
+ import { n as __reExport, t as __exportAll } from "./chunk-CFhWmker.mjs";
2
+ import { CanonicalBoneDef as SkeletonBone, PoseSkeletonOptions } from "@framefields/tensor-webgpu";
3
+ import * as _framefields_core0 from "@framefields/core";
4
+ import { DetectedObject, FrameContext, Landmark3D, ObjectBoundingBox, ProgrammaticSignal, TensorData, TrackedObject, VisionConfig, VisionNodeSpec, VisionTask, VisionVariant } from "@framefields/core";
5
+
6
+ //#region src/model/registry.d.ts
7
+
8
+ declare const VISION_TASKS: readonly VisionTask[];
9
+ declare const VISION_VARIANTS: readonly VisionVariant[];
10
+ type VisionModelKey = "rtmdet-ins-t" | "rtmdet-ins-s" | "rtmdet-ins-m" | "rtmo-t" | "rtmo-s" | "rtmo-m" | "selfie-square" | "selfie-landscape";
11
+ /** Which decoder a model's outputs feed (see runner/vision-runner.ts). */
12
+ type VisionModelFamily = "rtmdet-ins" | "rtmo" | "selfie";
13
+ interface VisionModelDescriptor {
14
+ readonly key: VisionModelKey;
15
+ readonly family: VisionModelFamily;
16
+ /** Cache filename (also the path appended to a custom `baseUrl`). */
17
+ readonly filename: string;
18
+ /** Pinned download URL (immutable revision). */
19
+ readonly url: string;
20
+ /** Exact byte size of the pinned file. */
21
+ readonly bytes: number;
22
+ /** Hex SHA-256 of the pinned file. */
23
+ readonly sha256: string;
24
+ /** Model input size, `[width, height]` — must match the ONNX graph's input shape. */
25
+ readonly input: readonly [number, number];
26
+ /** Upstream project and license, surfaced in docs and errors. */
27
+ readonly source: string;
28
+ }
29
+ declare const VISION_MODELS: Readonly<Record<VisionModelKey, VisionModelDescriptor>>;
30
+ /**
31
+ * Resolves the model for a task. `detect` and `segment` share RTMDet-Ins (one forward pass
32
+ * yields both); `matte` picks the landscape Selfie Segmenter for wide frames.
33
+ */
34
+ declare function modelKeyFor(task: VisionTask, variant?: VisionVariant, aspect?: number): VisionModelKey;
35
+ /** COCO 80 class names — index order matches the RTMDet-Ins `labels` output. */
36
+ declare const COCO_CLASSES: readonly ["person", "bicycle", "car", "motorcycle", "airplane", "bus", "train", "truck", "boat", "traffic light", "fire hydrant", "stop sign", "parking meter", "bench", "bird", "cat", "dog", "horse", "sheep", "cow", "elephant", "bear", "zebra", "giraffe", "backpack", "umbrella", "handbag", "tie", "suitcase", "frisbee", "skis", "snowboard", "sports ball", "kite", "baseball bat", "baseball glove", "skateboard", "surfboard", "tennis racket", "bottle", "wine glass", "cup", "fork", "knife", "spoon", "bowl", "banana", "apple", "sandwich", "orange", "broccoli", "carrot", "hot dog", "pizza", "donut", "cake", "chair", "couch", "potted plant", "bed", "dining table", "toilet", "tv", "laptop", "mouse", "remote", "keyboard", "cell phone", "microwave", "oven", "toaster", "sink", "refrigerator", "book", "clock", "vase", "scissors", "teddy bear", "hair drier", "toothbrush"];
37
+ type CocoClass = (typeof COCO_CLASSES)[number];
38
+ //#endregion
39
+ //#region src/model/model-store.d.ts
40
+ type ModelDownloadStatus = "pending" | "downloading" | "ready" | "error";
41
+ interface VisionModelStoreOptions {
42
+ readonly modelsDir?: string;
43
+ /**
44
+ * Mirror origin: models are fetched from `<baseUrl>/<filename>` instead of their pinned
45
+ * Hugging Face URLs. Defaults to `$FRAMEFIELDS_MODELS_BASE_URL` when set.
46
+ */
47
+ readonly baseUrl?: string;
48
+ readonly timeoutMs?: number;
49
+ /** Extra attempts after a network error, HTTP 429 or 5xx. Default 2. */
50
+ readonly retries?: number;
51
+ /** Base backoff between attempts (doubles each retry). Default 1000 ms. */
52
+ readonly retryDelayMs?: number;
53
+ /**
54
+ * Verify byte size + SHA-256 against the registry (default true). Only offline tests
55
+ * that serve fake bytes should turn this off.
56
+ */
57
+ readonly verify?: boolean;
58
+ /** Injectable fetch — offline unit tests inject a mock. */
59
+ readonly fetchImpl?: typeof fetch;
60
+ readonly onProgress?: (key: VisionModelKey, bytesLoaded: number, totalBytes: number) => void;
61
+ }
62
+ /**
63
+ * Resolves the default model cache directory:
64
+ * 1. `FRAMEFIELDS_MODELS_DIR`
65
+ * 2. `~/.cache/framefields/models`
66
+ */
67
+ declare function getDefaultModelsDir(): string;
68
+ /**
69
+ * Lazy model store. The constructor performs ZERO I/O; the only entry that can hit the
70
+ * network is `ensure(key)`, called by an inference function the first time it runs.
71
+ * Concurrent callers of the same key share a single in-flight download (promise dedupe),
72
+ * downloads are written atomically (temp + rename), and verified files on disk are reused
73
+ * across processes.
74
+ */
75
+ declare class VisionModelStore {
76
+ private readonly _modelsDir;
77
+ private readonly _baseUrl?;
78
+ private readonly _timeoutMs?;
79
+ private readonly _retries;
80
+ private readonly _retryDelayMs;
81
+ private readonly _verify;
82
+ private readonly _fetch;
83
+ private readonly _onProgress?;
84
+ private _inflight;
85
+ private _buffers;
86
+ private _status;
87
+ constructor(options?: VisionModelStoreOptions);
88
+ get modelsDir(): string;
89
+ pathFor(key: VisionModelKey): string;
90
+ descriptor(key: VisionModelKey): VisionModelDescriptor;
91
+ /** The URL `ensure(key)` downloads from (mirror when `baseUrl` is set). */
92
+ urlFor(key: VisionModelKey): string;
93
+ /** True when a complete cached file exists (exact registry size when verifying). */
94
+ has(key: VisionModelKey): boolean;
95
+ /** Drops a cached model (file + memory) so the next `ensure` re-downloads it. */
96
+ evict(key: VisionModelKey): Promise<void>;
97
+ status(key: VisionModelKey): ModelDownloadStatus;
98
+ /** Snapshot of every tracked model's status (empty until anything is fetched). */
99
+ statuses(): ReadonlyMap<VisionModelKey, ModelDownloadStatus>;
100
+ /** True when the model is resident in memory (loaded or downloaded this process). */
101
+ isLoaded(key: VisionModelKey): boolean;
102
+ /**
103
+ * THE only download entry point. Returns the model bytes from memory (cached),
104
+ * loading from disk or fetching from the registry URL as needed.
105
+ */
106
+ ensure(key: VisionModelKey): Promise<Uint8Array>;
107
+ /** Explicit warm-up — fetch multiple models ahead of use (agents, renderers). */
108
+ preload(keys: readonly VisionModelKey[]): Promise<void>;
109
+ private load;
110
+ /** Fetches with retries on transient failures (network errors, HTTP 429/5xx). */
111
+ private download;
112
+ private verifyBytes;
113
+ private readBuffer;
114
+ /** Drops in-memory buffers (model stays on disk for the next run). */
115
+ clearMemoryCache(): void;
116
+ }
117
+ //#endregion
118
+ //#region src/runtime/session-provider.d.ts
119
+ /**
120
+ * Minimal ONNX Runtime surface — isolates the ORT API to exactly two files so the decode
121
+ * and signals layers never touch it. Node path uses onnxruntime-node (CPU EP); the web
122
+ * entry can swap in a WebGPU-backed provider with the same interface.
123
+ */
124
+ /** Model output tensor. Integer outputs (e.g. RTMDet-Ins `labels`) arrive as BigInt64Array. */
125
+ interface VisionTensor {
126
+ readonly data: Float32Array | BigInt64Array | Int32Array;
127
+ readonly dims: readonly number[];
128
+ }
129
+ interface VisionTensorInput {
130
+ readonly name: string;
131
+ readonly data: Float32Array;
132
+ readonly dims: readonly number[];
133
+ }
134
+ type VisionTensorOutputs = Readonly<Record<string, VisionTensor>>;
135
+ interface VisionSession {
136
+ run(input: VisionTensorInput): Promise<VisionTensorOutputs>;
137
+ release(): void;
138
+ }
139
+ interface SessionProvider {
140
+ readonly kind: "node" | "webgpu";
141
+ createSession(modelBytes: Uint8Array): Promise<VisionSession>;
142
+ }
143
+ /**
144
+ * Sets the provider runners use when none is passed, for hosts that create
145
+ * runners indirectly (e.g. a browser player rendering vision nodes). `undefined`
146
+ * restores the default, onnxruntime-node.
147
+ */
148
+ declare function setDefaultSessionProvider(factory: (() => SessionProvider) | undefined): void;
149
+ /**
150
+ * Prefers the GPU (onnxruntime-web's WebGPU EP on the process's WebGPU device —
151
+ * the Dawn device the renderer already creates, or one this ensures) and falls
152
+ * back to the CPU provider (onnxruntime-node) when a WebGPU session cannot be
153
+ * created: no device, no Dawn, or the EP unavailable. The choice is made on the
154
+ * first session and reused for the runner's lifetime.
155
+ */
156
+ declare class AutoSessionProvider implements SessionProvider {
157
+ private _resolved?;
158
+ get kind(): "node" | "webgpu";
159
+ createSession(modelBytes: Uint8Array): Promise<VisionSession>;
160
+ }
161
+ /**
162
+ * The provider for a runner created without one: GPU if possible, otherwise CPU.
163
+ */
164
+ declare function createDefaultSessionProvider(): SessionProvider;
165
+ /** onnxruntime-node provider. The native module is imported lazily on first session. */
166
+ declare class NodeSessionProvider implements SessionProvider {
167
+ readonly kind: "node";
168
+ private ortPromise?;
169
+ private ort;
170
+ createSession(modelBytes: Uint8Array): Promise<VisionSession>;
171
+ }
172
+ declare namespace types_d_exports {
173
+ export { InstanceMask, LandmarkCoordinateSignals, ObjectAnchorName, PersonMatte, PinToLandmarkOptions, PinToObjectOptions, PoseKeypoint, PosePerson, PoseResult, SegmentationResult, VisionAnalysisClassSummary, VisionAnalysisReport, VisionAnalysisTrackSummary, VisionImageInput, VisionInputSource, VisionSummary };
174
+ }
175
+ import * as import__framefields_core from "@framefields/core";
176
+ type VisionInputSource = string | {
177
+ inputHandleId?: string;
178
+ id?: string;
179
+ [key: string]: unknown;
180
+ };
181
+ /** RGBA pixel buffer accepted by the runner. */
182
+ interface VisionImageInput {
183
+ readonly data: Uint8ClampedArray | Uint8Array;
184
+ readonly width: number;
185
+ readonly height: number;
186
+ }
187
+ /** Instance segmentation mask aligned to the ORIGINAL source frame. */
188
+ interface InstanceMask {
189
+ readonly category: string;
190
+ readonly mask: Uint8Array;
191
+ readonly width: number;
192
+ readonly height: number;
193
+ readonly area: number;
194
+ readonly coverage: number;
195
+ /** Index of the detection this mask belongs to (see segmentation result detections). */
196
+ readonly detectionIndex: number;
197
+ /** Assigned by the caller (renderer) after temporal tracking — matches TrackedObject.trackId. */
198
+ readonly trackId?: number;
199
+ }
200
+ interface SegmentationResult {
201
+ readonly detections: readonly DetectedObject[];
202
+ readonly masks: readonly InstanceMask[];
203
+ }
204
+ /** Person-vs-background alpha (Selfie Segmenter), aligned to the ORIGINAL source frame. */
205
+ interface PersonMatte {
206
+ readonly mask: Uint8Array;
207
+ readonly width: number;
208
+ readonly height: number;
209
+ readonly coverage: number;
210
+ }
211
+ interface PoseKeypoint {
212
+ readonly x: number;
213
+ readonly y: number;
214
+ readonly visibility: number;
215
+ }
216
+ interface PosePerson {
217
+ readonly score: number;
218
+ readonly boundingBox: ObjectBoundingBox;
219
+ /** 17 keypoints in COCO-17 order (see COCO17_KEYPOINTS). */
220
+ readonly keypoints: readonly PoseKeypoint[];
221
+ readonly trackId?: number;
222
+ }
223
+ interface PoseResult {
224
+ readonly people: readonly PosePerson[];
225
+ }
226
+ /**
227
+ * Deterministic, serializable snapshot of one frame's signals.
228
+ * Built synchronously from the per-frame caches, so it is safe to call inside a frame hook.
229
+ */
230
+ interface VisionSummary {
231
+ readonly frame: number;
232
+ readonly objects: ReadonlyArray<{
233
+ readonly trackId: number;
234
+ readonly category: string;
235
+ readonly score: number;
236
+ /** Source-space pixel center, `[x, y]`. */
237
+ readonly center: readonly [number, number];
238
+ readonly speed: number;
239
+ readonly active: boolean;
240
+ }>;
241
+ readonly classes: readonly string[];
242
+ /** Per-class instance count for the frame (only non-zero classes present). */
243
+ readonly masks: Readonly<Record<string, number>>;
244
+ }
245
+ /** Per-track aggregate in an analysis report. */
246
+ interface VisionAnalysisTrackSummary {
247
+ readonly trackId: number;
248
+ readonly category: string;
249
+ /** Inclusive frame range the track was active over, `[start, end]`. */
250
+ readonly frames: readonly [number, number];
251
+ readonly speedMeanPxS: number;
252
+ /** Center path sampled every `pathStride` frames, `[x, y]` pairs. */
253
+ readonly centerPath: ReadonlyArray<readonly [number, number]>;
254
+ }
255
+ interface VisionAnalysisClassSummary {
256
+ readonly framesPresent: number;
257
+ readonly totalFrames: number;
258
+ readonly maxConfidence: number;
259
+ readonly tracks: number;
260
+ }
261
+ interface VisionAnalysisReport {
262
+ readonly source: string;
263
+ readonly totalFrames: number;
264
+ readonly fps: number;
265
+ readonly durationSec: number;
266
+ readonly tasks: readonly string[];
267
+ /** What was fetched, and how long it took — surfaced so agents can budget cold starts. */
268
+ readonly modelDownloads: Readonly<Record<string, {
269
+ readonly bytes: number;
270
+ readonly ms: number;
271
+ }>>;
272
+ readonly inferenceMs: number;
273
+ readonly trackCount: number;
274
+ readonly classes: Readonly<Record<string, VisionAnalysisClassSummary>>;
275
+ readonly tracks: readonly VisionAnalysisTrackSummary[];
276
+ readonly masks: Readonly<Record<string, {
277
+ readonly meanCoverage: number;
278
+ }>>;
279
+ }
280
+ interface PinToLandmarkOptions {
281
+ readonly offsetX?: number;
282
+ readonly offsetY?: number;
283
+ readonly offsetZ?: number;
284
+ readonly trackRotation?: boolean;
285
+ readonly scaleWithPerspective?: boolean;
286
+ }
287
+ interface LandmarkCoordinateSignals {
288
+ readonly x: _framefields_core0.ProgrammaticSignal;
289
+ readonly y: _framefields_core0.ProgrammaticSignal;
290
+ readonly z: _framefields_core0.ProgrammaticSignal;
291
+ readonly screenX: _framefields_core0.ProgrammaticSignal;
292
+ readonly screenY: _framefields_core0.ProgrammaticSignal;
293
+ }
294
+ type ObjectAnchorName = "center" | "topCenter" | "bottomCenter" | "topLeft" | "topRight" | "centerLeft" | "centerRight" | "bottomLeft" | "bottomRight";
295
+ interface PinToObjectOptions extends PinToLandmarkOptions {
296
+ readonly anchor?: ObjectAnchorName;
297
+ readonly matchWidth?: boolean;
298
+ readonly matchHeight?: boolean;
299
+ readonly smoothFrames?: number;
300
+ readonly hideWhenLost?: boolean;
301
+ }
302
+ //#endregion
303
+ //#region src/runner/vision-runner.d.ts
304
+ interface VisionRunnerOptions {
305
+ /** Model size for detect/segment/pose. Default "s". */
306
+ readonly variant?: VisionVariant;
307
+ /** Minimum detection / person score, 0..1. Default 0.3. */
308
+ readonly confidence?: number;
309
+ /** COCO class-name filter for detect/segment (all classes when omitted). */
310
+ readonly classes?: readonly string[];
311
+ readonly maskThreshold?: number;
312
+ readonly featherRadius?: number;
313
+ readonly modelsDir?: string;
314
+ /** Mirror origin for model downloads (see VisionModelStore). */
315
+ readonly baseUrl?: string;
316
+ readonly timeoutMs?: number;
317
+ /** Session backend. Default: onnxruntime-node (CPU). Browser: `createWebGPUProvider()`. */
318
+ readonly provider?: SessionProvider;
319
+ /** Inject a model store (tests, shared caches). Overrides modelsDir/baseUrl/timeoutMs. */
320
+ readonly store?: VisionModelStore;
321
+ readonly onProgress?: (key: VisionModelKey, bytesLoaded: number, totalBytes: number) => void;
322
+ }
323
+ /**
324
+ * Lazy vision runner.
325
+ *
326
+ * `create()` performs ZERO I/O: no downloads, no sessions. Each inference method downloads
327
+ * its model and warms its session on FIRST USE ONLY, then reuses them for the process
328
+ * lifetime. `detect()` and `segment()` share one RTMDet-Ins forward pass per frame.
329
+ * Call `close()` to release sessions.
330
+ */
331
+ declare class VisionRunner {
332
+ readonly variant: VisionVariant;
333
+ readonly confidence: number;
334
+ readonly classes?: readonly string[];
335
+ readonly maskThreshold: number;
336
+ readonly featherRadius?: number;
337
+ private readonly _store;
338
+ private readonly _provider;
339
+ private readonly _sessions;
340
+ private readonly _tensorScratch;
341
+ /** One RTMDet-Ins pass per frame buffer, shared by detect() and segment(). */
342
+ private _instancePass?;
343
+ private constructor();
344
+ /** Cheap: no downloads, no sessions, no side effects. */
345
+ static create(options?: VisionRunnerOptions): VisionRunner;
346
+ get modelsDir(): string;
347
+ /** Every registry model keyed to its current download state (downloaded → "ready"). */
348
+ get downloadStatus(): ReadonlyMap<VisionModelKey, ModelDownloadStatus>;
349
+ /** The model serving `task` (matte depends on frame aspect; defaults to square). */
350
+ modelKeyFor(task: VisionTask, aspect?: number): VisionModelKey;
351
+ /**
352
+ * Explicit warm-up: downloads + sessions for the given tasks, ahead of first use.
353
+ * `matte` warms both aspect variants (the frame aspect is unknown until inference).
354
+ */
355
+ preload(tasks?: readonly VisionTask[]): Promise<void>;
356
+ /** COCO-80 boxes (sorted by score, source pixels). Masks are not decoded. */
357
+ detect(image: VisionImageInput): Promise<readonly DetectedObject[]>;
358
+ /** COCO-80 boxes plus a frame-aligned soft mask per instance. */
359
+ segment(image: VisionImageInput): Promise<SegmentationResult>;
360
+ /** COCO-17 keypoints per person (source pixels), highest score first. */
361
+ pose(image: VisionImageInput): Promise<PoseResult>;
362
+ /** Person-vs-background alpha for the whole frame (Selfie Segmenter). */
363
+ matte(image: VisionImageInput): Promise<PersonMatte>;
364
+ close(): void;
365
+ private instancePass;
366
+ private instanceDecodeOptions;
367
+ private infer;
368
+ private session;
369
+ }
370
+ //#endregion
371
+ //#region src/analysis/analyze-sequence.d.ts
372
+ /**
373
+ * One-shot, ffmpeg-free sequence analysis.
374
+ *
375
+ * Pure frame-iterable analyzer: it never decodes media itself, so it stays engine-only and
376
+ * offline-testable. Callers (e.g. the framefields ingestion adapter) supply decoded frames.
377
+ * Model loading remains lazy — the runner downloads each task's model on the first frame
378
+ * that actually runs that task, and only the tasks requested here are ever touched.
379
+ */
380
+ interface AnalyzeSequenceOptions {
381
+ readonly runner: VisionRunner;
382
+ /** Label for the report ("clip.mp4" or similar). Defaults to "frames". */
383
+ readonly source?: string;
384
+ readonly fps?: number;
385
+ /** Tasks to run. Defaults to `["detect"]`. Only these models are ever downloaded. */
386
+ readonly tasks?: readonly VisionTask[];
387
+ /** Restrict class summaries to these categories (all when omitted). */
388
+ readonly categories?: readonly string[];
389
+ /** Tracker box-association IoU. Default 0.25. */
390
+ readonly iouThreshold?: number;
391
+ readonly maxMissedFrames?: number;
392
+ /** Sample the per-track center path every N frames. Default: ~50 points across the clip. */
393
+ readonly pathStride?: number;
394
+ /** Also aggregate instance-mask coverage (requires the `segment` task). Default false. */
395
+ readonly includeMasks?: boolean;
396
+ }
397
+ declare function analyzeSequence(frames: AsyncIterable<VisionImageInput> | Iterable<VisionImageInput>, options: AnalyzeSequenceOptions): Promise<VisionAnalysisReport>;
398
+ //#endregion
399
+ //#region src/decode/preprocess.d.ts
400
+ /**
401
+ * Image → NCHW float tensor preprocessing, plus the inverse mapping back to source pixels.
402
+ *
403
+ * Each model family was exported with its own preprocessing contract:
404
+ *
405
+ * | Model | Resize | Pad | Channels | Values |
406
+ * | ----------- | ---------------------------- | ---- | -------- | -------------------------- |
407
+ * | RTMDet-Ins | keep ratio, content top-left | 114 | BGR | (v − mean) / std (ImageNet) |
408
+ * | RTMO | keep ratio, content centered | 114 | BGR | raw 0..255 |
409
+ * | Selfie | stretch to input | — | RGB | v / 255 |
410
+ *
411
+ * Getting any of these wrong does not crash — it silently degrades accuracy — so the specs
412
+ * live here and every runner path goes through `imageToTensor`. Input sizes differ per model
413
+ * (RTMO-t is 416², the rest of RTMO 640²), so they come from the registry, never from here.
414
+ */
415
+ interface RgbaImage {
416
+ readonly data: Uint8ClampedArray | Uint8Array;
417
+ readonly width: number;
418
+ readonly height: number;
419
+ }
420
+ interface PreprocessSpec {
421
+ /** Model input size in pixels. */
422
+ readonly width: number;
423
+ readonly height: number;
424
+ /** `topLeft` / `center` keep aspect ratio and pad; `stretch` fills the input. */
425
+ readonly fit: "topLeft" | "center" | "stretch";
426
+ readonly channels: "rgb" | "bgr";
427
+ /** Per-channel mean, in output channel order, on the 0..255 scale. */
428
+ readonly mean: readonly [number, number, number];
429
+ /** Per-channel std, in output channel order, on the 0..255 scale. */
430
+ readonly std: readonly [number, number, number];
431
+ /** Padding value on the 0..255 scale (normalized like any other pixel). */
432
+ readonly padValue: number;
433
+ }
434
+ /** Maps model-input pixels back to source pixels: `src = (input − offset) / scale`. */
435
+ interface InputTransform {
436
+ readonly scaleX: number;
437
+ readonly scaleY: number;
438
+ readonly offsetX: number;
439
+ readonly offsetY: number;
440
+ readonly inputWidth: number;
441
+ readonly inputHeight: number;
442
+ }
443
+ type Normalization = Omit<PreprocessSpec, "width" | "height">;
444
+ /** Per-family normalization; the input size always comes from the model's registry entry. */
445
+ declare const PREPROCESS_BY_FAMILY: Readonly<Record<"rtmdet-ins" | "rtmo" | "selfie", Normalization>>;
446
+ declare function preprocessSpec(family: keyof typeof PREPROCESS_BY_FAMILY, input: readonly [number, number]): PreprocessSpec;
447
+ declare function computeInputTransform(srcW: number, srcH: number, spec: Pick<PreprocessSpec, "width" | "height" | "fit">): InputTransform;
448
+ /** Model-input point → source pixel point. */
449
+ declare function toSourcePoint(x: number, y: number, t: InputTransform): {
450
+ x: number;
451
+ y: number;
452
+ };
453
+ /**
454
+ * Resizes RGBA → NCHW float32 `[1, 3, height, width]` with bilinear sampling, padding,
455
+ * channel reordering and normalization per `spec`. Reuses `out` when large enough
456
+ * (hot-path frame loop).
457
+ */
458
+ declare function imageToTensor(image: RgbaImage, spec: PreprocessSpec, out?: Float32Array): {
459
+ tensor: Float32Array;
460
+ transform: InputTransform;
461
+ };
462
+ //#endregion
463
+ //#region src/decode/rtmdet-ins.d.ts
464
+ /**
465
+ * RTMDet-Ins decode (mmdeploy end2end export — NMS and mask assembly are baked into the graph).
466
+ *
467
+ * Outputs:
468
+ * dets [1, N, 5] x0, y0, x1, y1 (input px), score
469
+ * labels [1, N] COCO-80 class index (int64)
470
+ * masks [1, N, Hm, Wm] per-instance probabilities (already sigmoided), input-aligned
471
+ */
472
+ interface RtmdetInsOutputs {
473
+ readonly dets: ArrayLike<number>;
474
+ readonly labels: ArrayLike<number | bigint>;
475
+ /** Omit to decode boxes only (detection without masks). */
476
+ readonly masks?: ArrayLike<number>;
477
+ readonly count: number;
478
+ readonly maskWidth: number;
479
+ readonly maskHeight: number;
480
+ }
481
+ interface RtmdetInsDecodeOptions {
482
+ readonly confidence: number;
483
+ /** Class-name filter, case-insensitive (all classes when omitted). */
484
+ readonly classes?: readonly string[];
485
+ readonly classNames: readonly string[];
486
+ readonly transform: InputTransform;
487
+ readonly sourceWidth: number;
488
+ readonly sourceHeight: number;
489
+ /** Mask probability threshold. Default 0.5. */
490
+ readonly maskThreshold?: number;
491
+ /** Soft-edge band around the threshold. Default 0.05. */
492
+ readonly featherRadius?: number;
493
+ }
494
+ declare function decodeRtmdetIns(out: RtmdetInsOutputs, opts: RtmdetInsDecodeOptions): SegmentationResult;
495
+ /**
496
+ * Probability → 0..255 alpha. A hard threshold when `feather` is 0, otherwise a linear ramp
497
+ * across `[threshold − feather, threshold + feather]` for anti-aliased edges.
498
+ */
499
+ declare function probabilityToAlpha(threshold: number, feather: number): (p: number) => number;
500
+ //#endregion
501
+ //#region src/decode/rtmo.d.ts
502
+ /**
503
+ * RTMO decode (one-stage multi-person pose; NMS is baked into the export).
504
+ *
505
+ * Outputs:
506
+ * dets [1, N, 5] x0, y0, x1, y1 (input px), person score
507
+ * keypoints [1, N, 17, 3] x, y (input px), keypoint score — COCO-17 order
508
+ */
509
+ interface RtmoOutputs {
510
+ readonly dets: ArrayLike<number>;
511
+ readonly keypoints: ArrayLike<number>;
512
+ readonly count: number;
513
+ /** Values per keypoint (3 for x, y, score). */
514
+ readonly keypointStride: number;
515
+ }
516
+ interface RtmoDecodeOptions {
517
+ readonly confidence: number;
518
+ readonly transform: InputTransform;
519
+ readonly sourceWidth: number;
520
+ readonly sourceHeight: number;
521
+ /** Drop people with fewer keypoints at visibility ≥ 0.3. Default 3. */
522
+ readonly minVisibleKeypoints?: number;
523
+ /** Suppress a person whose box overlaps a higher-scored one above this IoU. Default 0.6. */
524
+ readonly iouThreshold?: number;
525
+ }
526
+ declare const COCO17_KEYPOINT_COUNT = 17;
527
+ declare function decodeRtmo(out: RtmoOutputs, opts: RtmoDecodeOptions): PoseResult;
528
+ //#endregion
529
+ //#region src/decode/selfie.d.ts
530
+ /**
531
+ * Selfie Segmenter decode: `alphas [1, 1, h, w]` person probability (stretched input) →
532
+ * frame-sized 0..255 alpha, bilinear-upsampled to the source resolution.
533
+ */
534
+ interface SelfieDecodeOptions {
535
+ readonly sourceWidth: number;
536
+ readonly sourceHeight: number;
537
+ /** Probability threshold. Default 0.5. */
538
+ readonly maskThreshold?: number;
539
+ /** Soft-edge band; the low-res matte reads best with a wide one. Default 0.2. */
540
+ readonly featherRadius?: number;
541
+ }
542
+ declare function decodeSelfie(alphas: ArrayLike<number>, matteWidth: number, matteHeight: number, opts: SelfieDecodeOptions): PersonMatte;
543
+ //#endregion
544
+ //#region src/gpu/pose-skeleton-renderer.d.ts
545
+ /**
546
+ * WebGPU skeleton renderer for pose results.
547
+ *
548
+ * Rasterizes the primary person's COCO-17 keypoints into an OpenPose-style
549
+ * conditioning texture natively in VRAM, using the COCO-17 bone set.
550
+ */
551
+ declare class PoseSkeletonRenderer {
552
+ private pipeline;
553
+ constructor(device: GPUDevice);
554
+ renderToTexture(keypoints: readonly PoseKeypoint[], options: PoseSkeletonOptions, bones?: readonly [{
555
+ readonly from: 0;
556
+ readonly to: 5;
557
+ readonly color: readonly [0, 1, 1, 1];
558
+ }, {
559
+ readonly from: 0;
560
+ readonly to: 6;
561
+ readonly color: readonly [0, 1, 1, 1];
562
+ }, {
563
+ readonly from: 5;
564
+ readonly to: 6;
565
+ readonly color: readonly [1, 0, 0, 1];
566
+ }, {
567
+ readonly from: 5;
568
+ readonly to: 7;
569
+ readonly color: readonly [1, 0.333, 0, 1];
570
+ }, {
571
+ readonly from: 7;
572
+ readonly to: 9;
573
+ readonly color: readonly [1, 0.667, 0, 1];
574
+ }, {
575
+ readonly from: 6;
576
+ readonly to: 8;
577
+ readonly color: readonly [1, 1, 0, 1];
578
+ }, {
579
+ readonly from: 8;
580
+ readonly to: 10;
581
+ readonly color: readonly [0.667, 1, 0, 1];
582
+ }, {
583
+ readonly from: 5;
584
+ readonly to: 11;
585
+ readonly color: readonly [0.333, 1, 0, 1];
586
+ }, {
587
+ readonly from: 6;
588
+ readonly to: 12;
589
+ readonly color: readonly [0, 1, 0, 1];
590
+ }, {
591
+ readonly from: 11;
592
+ readonly to: 12;
593
+ readonly color: readonly [0, 1, 0.333, 1];
594
+ }, {
595
+ readonly from: 11;
596
+ readonly to: 13;
597
+ readonly color: readonly [0, 1, 0.667, 1];
598
+ }, {
599
+ readonly from: 13;
600
+ readonly to: 15;
601
+ readonly color: readonly [0, 1, 1, 1];
602
+ }, {
603
+ readonly from: 12;
604
+ readonly to: 14;
605
+ readonly color: readonly [0, 0.667, 1, 1];
606
+ }, {
607
+ readonly from: 14;
608
+ readonly to: 16;
609
+ readonly color: readonly [0, 0.333, 1, 1];
610
+ }]): GPUTexture;
611
+ destroy(): void;
612
+ }
613
+ //#endregion
614
+ //#region src/gpu/segmentation-texture-pool.d.ts
615
+ interface SegmentationTextureOptions {
616
+ readonly width: number;
617
+ readonly height: number;
618
+ readonly format?: "r8unorm" | "rgba8unorm";
619
+ readonly label?: string;
620
+ }
621
+ declare class SegmentationTexturePool {
622
+ private device;
623
+ private pool;
624
+ constructor(device: GPUDevice);
625
+ getOrCreateTexture(key: string, options: SegmentationTextureOptions): GPUTexture;
626
+ uploadMask(key: string, maskData: Uint8Array | Float32Array, width: number, height: number): GPUTexture;
627
+ destroy(): void;
628
+ }
629
+ //#endregion
630
+ //#region src/pose/keypoints.d.ts
631
+ /** COCO-17 keypoint indices (RTMO output order). */
632
+ declare const COCO17_KEYPOINTS: {
633
+ readonly NOSE: 0;
634
+ readonly LEFT_EYE: 1;
635
+ readonly RIGHT_EYE: 2;
636
+ readonly LEFT_EAR: 3;
637
+ readonly RIGHT_EAR: 4;
638
+ readonly LEFT_SHOULDER: 5;
639
+ readonly RIGHT_SHOULDER: 6;
640
+ readonly LEFT_ELBOW: 7;
641
+ readonly RIGHT_ELBOW: 8;
642
+ readonly LEFT_WRIST: 9;
643
+ readonly RIGHT_WRIST: 10;
644
+ readonly LEFT_HIP: 11;
645
+ readonly RIGHT_HIP: 12;
646
+ readonly LEFT_KNEE: 13;
647
+ readonly RIGHT_KNEE: 14;
648
+ readonly LEFT_ANKLE: 15;
649
+ readonly RIGHT_ANKLE: 16;
650
+ };
651
+ declare const COCO17_KEYPOINT_NAMES: readonly ["nose", "leftEye", "rightEye", "leftEar", "rightEar", "leftShoulder", "rightShoulder", "leftElbow", "rightElbow", "leftWrist", "rightWrist", "leftHip", "rightHip", "leftKnee", "rightKnee", "leftAnkle", "rightAnkle"];
652
+ /** COCO-17 skeleton bones for the WebGPU skeleton renderer (indices above). */
653
+ declare const COCO17_BONES: readonly [{
654
+ readonly from: 0;
655
+ readonly to: 5;
656
+ readonly color: readonly [0, 1, 1, 1];
657
+ }, {
658
+ readonly from: 0;
659
+ readonly to: 6;
660
+ readonly color: readonly [0, 1, 1, 1];
661
+ }, {
662
+ readonly from: 5;
663
+ readonly to: 6;
664
+ readonly color: readonly [1, 0, 0, 1];
665
+ }, {
666
+ readonly from: 5;
667
+ readonly to: 7;
668
+ readonly color: readonly [1, 0.333, 0, 1];
669
+ }, {
670
+ readonly from: 7;
671
+ readonly to: 9;
672
+ readonly color: readonly [1, 0.667, 0, 1];
673
+ }, {
674
+ readonly from: 6;
675
+ readonly to: 8;
676
+ readonly color: readonly [1, 1, 0, 1];
677
+ }, {
678
+ readonly from: 8;
679
+ readonly to: 10;
680
+ readonly color: readonly [0.667, 1, 0, 1];
681
+ }, {
682
+ readonly from: 5;
683
+ readonly to: 11;
684
+ readonly color: readonly [0.333, 1, 0, 1];
685
+ }, {
686
+ readonly from: 6;
687
+ readonly to: 12;
688
+ readonly color: readonly [0, 1, 0, 1];
689
+ }, {
690
+ readonly from: 11;
691
+ readonly to: 12;
692
+ readonly color: readonly [0, 1, 0.333, 1];
693
+ }, {
694
+ readonly from: 11;
695
+ readonly to: 13;
696
+ readonly color: readonly [0, 1, 0.667, 1];
697
+ }, {
698
+ readonly from: 13;
699
+ readonly to: 15;
700
+ readonly color: readonly [0, 1, 1, 1];
701
+ }, {
702
+ readonly from: 12;
703
+ readonly to: 14;
704
+ readonly color: readonly [0, 0.667, 1, 1];
705
+ }, {
706
+ readonly from: 14;
707
+ readonly to: 16;
708
+ readonly color: readonly [0, 0.333, 1, 1];
709
+ }];
710
+ //#endregion
711
+ //#region src/runner/canonical-baselines.d.ts
712
+ /** Neutral fallbacks so signal evaluators always have a well-formed frame to read. */
713
+ declare function createNeutralObjectResult(): {
714
+ readonly objects: readonly TrackedObject[];
715
+ readonly rawDetections: readonly DetectedObject[];
716
+ };
717
+ declare function createNeutralPoseResult(): PoseResult;
718
+ /** 17 neutral COCO keypoints at frame center — used by pose signals when nobody is detected. */
719
+ declare function createNeutralLandmarks(): readonly Landmark3D[];
720
+ //#endregion
721
+ //#region src/runtime/webgpu-provider.d.ts
722
+ /** Minimal structural view of the `onnxruntime-web` module — isolates the ORT API surface. */
723
+ interface OrtWebTensor {
724
+ readonly data: Float32Array | BigInt64Array | Int32Array;
725
+ readonly dims: readonly number[];
726
+ }
727
+ interface OrtWebSession {
728
+ readonly inputNames: readonly string[];
729
+ run(feeds: Record<string, unknown>): Promise<Record<string, OrtWebTensor>>;
730
+ release(): Promise<void>;
731
+ }
732
+ interface OrtWebModule {
733
+ readonly InferenceSession: {
734
+ create(modelBytes: Uint8Array, options: {
735
+ executionProviders: string[];
736
+ graphOptimizationLevel?: "disabled" | "basic" | "extended" | "all";
737
+ /** 0 verbose … 4 fatal. */
738
+ logSeverityLevel?: 0 | 1 | 2 | 3 | 4;
739
+ }): Promise<OrtWebSession>;
740
+ };
741
+ /** ORT tensor constructor (called with `new`). */
742
+ readonly Tensor: new (type: "float32", data: Float32Array, dims: readonly number[]) => unknown;
743
+ }
744
+ interface WebGPUProviderOptions {
745
+ /** Injected ORT loader — defaults to a dynamic `import("onnxruntime-web")`. */
746
+ readonly loader?: () => Promise<OrtWebModule>;
747
+ /** Execution providers in priority order. Defaults to `["webgpu", "wasm"]`. */
748
+ readonly executionProviders?: readonly string[];
749
+ /** Fall back to WASM automatically if the WebGPU session fails to create. Default true. */
750
+ readonly wasmFallback?: boolean;
751
+ }
752
+ declare class WebGPUProvider implements SessionProvider {
753
+ readonly kind: "webgpu";
754
+ private readonly _loader;
755
+ private readonly _providers;
756
+ private readonly _wasmFallback;
757
+ private _ortPromise?;
758
+ constructor(options?: WebGPUProviderOptions);
759
+ private ort;
760
+ createSession(modelBytes: Uint8Array): Promise<VisionSession>;
761
+ }
762
+ /**
763
+ * Convenience factory: a `SessionProvider` type guard for the browser entry. Kept tiny so
764
+ * `@framefields/vision/web` stays tree-shakeable.
765
+ */
766
+ declare function createWebGPUProvider(options?: WebGPUProviderOptions): WebGPUProvider;
767
+ /** True when the environment exposes a WebGPU device (synchronous capability probe). */
768
+ declare function hasWebGPU(): boolean;
769
+ //#endregion
770
+ //#region src/runtime/node-webgpu-provider.d.ts
771
+ /**
772
+ * Ensure `globalThis.navigator.gpu` exists. In the renderer this is already the
773
+ * Dawn device; otherwise imports the `webgpu` (Dawn) package and installs its
774
+ * globals plus an adapter, which is what onnxruntime-web's WebGPU EP needs.
775
+ */
776
+ declare function ensureNodeWebGPU(): Promise<void>;
777
+ /** Options mirror the browser provider; the loader is overridden for Node. */
778
+ type NodeWebGPUProviderOptions = WebGPUProviderOptions;
779
+ declare class NodeWebGPUProvider implements SessionProvider {
780
+ readonly kind: "webgpu";
781
+ private readonly _inner;
782
+ constructor(options?: NodeWebGPUProviderOptions);
783
+ createSession(modelBytes: Uint8Array): Promise<VisionSession>;
784
+ }
785
+ declare function createNodeWebGPUProvider(options?: NodeWebGPUProviderOptions): NodeWebGPUProvider;
786
+ //#endregion
787
+ //#region src/segmentation/subject.d.ts
788
+ /**
789
+ * The frame's primary instance: the largest person when present, else the most confident
790
+ * instance (lowest `detectionIndex` — detections are score-sorted). Size alone is a poor
791
+ * signal without a person: the largest instance is usually a backdrop ("dining table").
792
+ */
793
+ declare function selectSubjectMask(masks: readonly InstanceMask[]): InstanceMask | undefined;
794
+ interface MaskBounds {
795
+ readonly x0: number;
796
+ readonly y0: number;
797
+ readonly x1: number;
798
+ readonly y1: number;
799
+ }
800
+ /** Tight pixel bounds of a mask's non-zero alpha, or null when empty. */
801
+ declare function maskBounds(mask: InstanceMask): MaskBounds | null;
802
+ /**
803
+ * Builds the full subject silhouette: the primary instance plus every comparably sized
804
+ * instance whose bounds sit mostly inside or across it (`overlap` = intersection / smaller
805
+ * box area; `maxGrowth` caps a part's box area relative to the subject's).
806
+ *
807
+ * COCO has no "clothing" class, so a flowing dress, a held guitar or a ridden bike comes back
808
+ * as its own instance (often mislabeled) — merging them keeps the whole figure in the cutout.
809
+ * The size cap keeps containers out: a small figure inside a tunnel or window detected as a
810
+ * huge "clock" must not drag the whole frame into the subject.
811
+ */
812
+ declare function mergeSubjectMask(masks: readonly InstanceMask[], overlap?: number, maxGrowth?: number): InstanceMask | undefined;
813
+ //#endregion
814
+ //#region src/signals/vision-bundle.d.ts
815
+ interface VisionBundleOptions {
816
+ readonly totalFrames?: number;
817
+ readonly fps?: number;
818
+ readonly width?: number;
819
+ readonly height?: number;
820
+ readonly cameraFov?: number;
821
+ readonly config?: types_d_exports.VisionConfig;
822
+ }
823
+ /** Named pose landmarks (COCO-17) plus numeric proxy access to any of the 17 keypoints. */
824
+ interface PoseLandmarkSignals {
825
+ readonly shoulder: LandmarkCoordinateSignals;
826
+ readonly leftShoulder: LandmarkCoordinateSignals;
827
+ readonly rightShoulder: LandmarkCoordinateSignals;
828
+ readonly leftElbow: LandmarkCoordinateSignals;
829
+ readonly rightElbow: LandmarkCoordinateSignals;
830
+ readonly leftWrist: LandmarkCoordinateSignals;
831
+ readonly rightWrist: LandmarkCoordinateSignals;
832
+ readonly leftHip: LandmarkCoordinateSignals;
833
+ readonly rightHip: LandmarkCoordinateSignals;
834
+ readonly leftKnee: LandmarkCoordinateSignals;
835
+ readonly rightKnee: LandmarkCoordinateSignals;
836
+ readonly leftAnkle: LandmarkCoordinateSignals;
837
+ readonly rightAnkle: LandmarkCoordinateSignals;
838
+ readonly nose: LandmarkCoordinateSignals;
839
+ readonly leftEye: LandmarkCoordinateSignals;
840
+ readonly rightEye: LandmarkCoordinateSignals;
841
+ readonly leftEar: LandmarkCoordinateSignals;
842
+ readonly rightEar: LandmarkCoordinateSignals;
843
+ get(keypointIndex: number): LandmarkCoordinateSignals;
844
+ }
845
+ interface TrackPoseSignals extends PoseLandmarkSignals {
846
+ readonly hasPose: ProgrammaticSignal;
847
+ readonly wristSpeed: ProgrammaticSignal;
848
+ readonly handRaised: ProgrammaticSignal;
849
+ readonly bodyTiltAngle: ProgrammaticSignal;
850
+ }
851
+ interface ObjectBoundingBoxSignals {
852
+ readonly x: ProgrammaticSignal;
853
+ readonly y: ProgrammaticSignal;
854
+ readonly width: ProgrammaticSignal;
855
+ readonly height: ProgrammaticSignal;
856
+ readonly screenX: ProgrammaticSignal;
857
+ readonly screenY: ProgrammaticSignal;
858
+ readonly screenWidth: ProgrammaticSignal;
859
+ readonly screenHeight: ProgrammaticSignal;
860
+ readonly aspectRatio: ProgrammaticSignal;
861
+ readonly area: ProgrammaticSignal;
862
+ }
863
+ interface ObjectAnchorsSignals {
864
+ readonly topLeft: LandmarkCoordinateSignals;
865
+ readonly topCenter: LandmarkCoordinateSignals;
866
+ readonly topRight: LandmarkCoordinateSignals;
867
+ readonly centerLeft: LandmarkCoordinateSignals;
868
+ readonly center: LandmarkCoordinateSignals;
869
+ readonly centerRight: LandmarkCoordinateSignals;
870
+ readonly bottomLeft: LandmarkCoordinateSignals;
871
+ readonly bottomCenter: LandmarkCoordinateSignals;
872
+ readonly bottomRight: LandmarkCoordinateSignals;
873
+ }
874
+ interface ObjectKinematicsSignals {
875
+ readonly vx: ProgrammaticSignal;
876
+ readonly vy: ProgrammaticSignal;
877
+ readonly speed: ProgrammaticSignal;
878
+ readonly acceleration: ProgrammaticSignal;
879
+ readonly headingRad: ProgrammaticSignal;
880
+ readonly headingDeg: ProgrammaticSignal;
881
+ }
882
+ interface ObjectTrackSignals {
883
+ readonly trackId: number;
884
+ readonly category: string;
885
+ readonly bounds: ObjectBoundingBoxSignals;
886
+ readonly anchors: ObjectAnchorsSignals;
887
+ readonly kinematics: ObjectKinematicsSignals;
888
+ readonly confidence: ProgrammaticSignal;
889
+ readonly active: ProgrammaticSignal;
890
+ readonly isCoasting: ProgrammaticSignal;
891
+ readonly age: ProgrammaticSignal;
892
+ readonly pose: TrackPoseSignals;
893
+ readonly center: LandmarkCoordinateSignals;
894
+ readonly topCenter: LandmarkCoordinateSignals;
895
+ readonly bottomCenter: LandmarkCoordinateSignals;
896
+ readonly topLeft: LandmarkCoordinateSignals;
897
+ readonly bottomLeft: LandmarkCoordinateSignals;
898
+ readonly [key: string]: unknown;
899
+ }
900
+ interface MaskTrackSignals {
901
+ readonly trackId: number;
902
+ readonly category: string;
903
+ readonly area: ProgrammaticSignal;
904
+ readonly coverage: ProgrammaticSignal;
905
+ readonly solidity: ProgrammaticSignal;
906
+ readonly bboxFill: ProgrammaticSignal;
907
+ /** Frame-sized {0,255} mask for the current frame (renderer-populated). */
908
+ data?: Uint8Array;
909
+ /** GPU-resident silhouette texture (renderer-populated via SegmentationTexturePool). */
910
+ texture?: GPUTexture;
911
+ }
912
+ interface MaskCollectionSignals {
913
+ get(trackId: number): MaskTrackSignals;
914
+ /** Largest instance of the current frame (person preferred when present). */
915
+ readonly subject: MaskTrackSignals;
916
+ readonly count: ProgrammaticSignal;
917
+ }
918
+ interface SegmentationSignals {
919
+ readonly humanSilhouette: MaskTrackSignals;
920
+ readonly subject: MaskTrackSignals;
921
+ readonly instanceMasks: MaskCollectionSignals;
922
+ /** Person-vs-background alpha from the Selfie Segmenter (`enableMatte`). */
923
+ readonly matte: PersonMatteSignals;
924
+ readonly stencilTexture?: GPUTexture;
925
+ }
926
+ interface PersonMatteSignals {
927
+ /** Mean person alpha over the frame, 0..1 (0 when no matte has run). */
928
+ readonly coverage: ProgrammaticSignal;
929
+ /** Frame-sized 0..255 alpha for the given frame, if a matte has run. */
930
+ at(frame: number): PersonMatte | undefined;
931
+ }
932
+ interface ClassSignals {
933
+ readonly count: ProgrammaticSignal;
934
+ readonly maxConfidence: ProgrammaticSignal;
935
+ readonly present: ProgrammaticSignal;
936
+ readonly primary: ObjectTrackSignals;
937
+ }
938
+ interface ClassCollectionSignals {
939
+ /** Cached per-class signals (stable identity). */
940
+ get(name: string): ClassSignals;
941
+ /** Sorted active category names for the current frame. */
942
+ readonly names: readonly string[];
943
+ readonly histogram: {
944
+ get(ctx?: FrameContext): TensorData;
945
+ readonly value: TensorData;
946
+ };
947
+ }
948
+ interface ObjectCollectionSignals {
949
+ get(trackId: number): ObjectTrackSignals;
950
+ byCategory(category: string, rank?: number): ObjectTrackSignals;
951
+ readonly primary: ObjectTrackSignals;
952
+ readonly count: ProgrammaticSignal;
953
+ hasCategory(category: string): ProgrammaticSignal;
954
+ getActiveTracks(frame: number): readonly TrackedObject[];
955
+ readonly detectedCategories: readonly string[];
956
+ }
957
+ declare class VisionBundle {
958
+ readonly objects: ObjectCollectionSignals;
959
+ readonly poseLandmarks: PoseLandmarkSignals;
960
+ readonly masks: MaskCollectionSignals;
961
+ readonly segmentation: SegmentationSignals;
962
+ readonly classes: ClassCollectionSignals;
963
+ readonly poseLandmarksTensor: {
964
+ get(ctx?: FrameContext): TensorData;
965
+ readonly value: TensorData;
966
+ };
967
+ readonly objectsTensor: {
968
+ get(ctx?: FrameContext): TensorData;
969
+ readonly value: TensorData;
970
+ };
971
+ readonly masksTensor: {
972
+ get(ctx?: FrameContext): TensorData;
973
+ readonly value: TensorData;
974
+ };
975
+ /** Per-class detection histogram [nc] — top-level convenience mirror of `classes.histogram`. */
976
+ readonly histogramTensor: {
977
+ get(ctx?: FrameContext): TensorData;
978
+ readonly value: TensorData;
979
+ };
980
+ private _totalFrames;
981
+ private _fps;
982
+ private _width;
983
+ private _height;
984
+ private _transformer;
985
+ private _stencilTexture?;
986
+ private _poseCache;
987
+ private _objectCache;
988
+ private _maskCache;
989
+ private _matteCache;
990
+ private _poseLmCache;
991
+ private _trackCache;
992
+ private _categoryCache;
993
+ private _classCache;
994
+ private _maskTrackCache;
995
+ private _registeredSignals;
996
+ constructor(options?: VisionBundleOptions);
997
+ setStencilTexture(texture: GPUTexture): void;
998
+ get stencilTexture(): GPUTexture | undefined;
999
+ /**
1000
+ * Reads the most recent result at or before `frame`. Vision results are written one
1001
+ * frame behind the plate (the node renderer reads back the previous frame to infer),
1002
+ * and pinned-signal layout runs before the current frame's inference, so a strict
1003
+ * per-frame lookup would always miss. Walking back a few frames keeps signals live at
1004
+ * a stable one-frame lag instead of snapping to neutral defaults.
1005
+ */
1006
+ private latestAtOrBefore;
1007
+ invalidateSignals(): void;
1008
+ setPoseResult(frame: number, result: PoseResult): void;
1009
+ getPoseResult(frame: number): PoseResult;
1010
+ setObjectResult(frame: number, result: {
1011
+ objects: readonly TrackedObject[];
1012
+ rawDetections?: readonly DetectedObject[];
1013
+ } | readonly TrackedObject[]): void;
1014
+ getObjectResult(frame: number): {
1015
+ objects: readonly TrackedObject[];
1016
+ rawDetections: readonly DetectedObject[];
1017
+ };
1018
+ setMaskResult(frame: number, masks: readonly InstanceMask[]): void;
1019
+ getMaskResult(frame: number): readonly InstanceMask[];
1020
+ setMatteResult(frame: number, matte: PersonMatte): void;
1021
+ getMatteResult(frame: number): PersonMatte | undefined;
1022
+ /**
1023
+ * Serializable snapshot of one frame's object / class / mask signals.
1024
+ * Synchronous — reads only the per-frame caches, never triggers inference.
1025
+ */
1026
+ summary(frame?: number): VisionSummary;
1027
+ /** Primary person's 17 keypoints as a [17, 3] float32 tensor. */
1028
+ getPoseLandmarksTensor(frame: number): TensorData;
1029
+ /** Up to 16 active tracks as [16, 8]: [active, categoryHash, x, y, w, h, vx, vy]. */
1030
+ getObjectsTensor(frame: number, maxObjects?: number): TensorData;
1031
+ /** Per-mask coverage + solidity as [16, 2]. */
1032
+ getMasksTensor(frame: number, maxMasks?: number): TensorData;
1033
+ /** Per-class detection histogram for the frame as [nc] float32 (COCO 80 by default). */
1034
+ getClassHistogramTensor(frame: number, nc?: number): TensorData;
1035
+ private clamp;
1036
+ }
1037
+ declare function createVisionBundle(options?: VisionBundleOptions): VisionBundle;
1038
+ //#endregion
1039
+ //#region src/spatial/camera-space-transformer.d.ts
1040
+ interface Camera3DSpec {
1041
+ readonly width: number;
1042
+ readonly height: number;
1043
+ readonly fov?: number;
1044
+ readonly cameraDistance?: number;
1045
+ }
1046
+ interface ProjectedScreenCoordinate {
1047
+ readonly x: number;
1048
+ readonly y: number;
1049
+ readonly z: number;
1050
+ readonly scale: number;
1051
+ }
1052
+ declare class SpatialLandmarkTransformer {
1053
+ private width;
1054
+ private height;
1055
+ private fovRad;
1056
+ private focalDistance;
1057
+ constructor(camera: Camera3DSpec);
1058
+ /**
1059
+ * Projects a normalized landmark [0, 1] into 2D/3D canvas coordinates.
1060
+ */
1061
+ projectNormalizedLandmark(landmark: Landmark3D, offsetZ?: number): ProjectedScreenCoordinate;
1062
+ /**
1063
+ * Projects a metric world landmark [X_w, Y_w, Z_w] in meters into canvas space.
1064
+ */
1065
+ projectWorldLandmark(worldLandmark: Landmark3D, subjectDistanceMeters?: number, scaleMetersToPixels?: number): ProjectedScreenCoordinate;
1066
+ }
1067
+ //#endregion
1068
+ //#region src/spatial/spatial-pin.d.ts
1069
+ declare function pinNodeToLandmark<T extends {
1070
+ x?: unknown;
1071
+ y?: unknown;
1072
+ z?: unknown;
1073
+ [key: string]: unknown;
1074
+ }>(node: T, target: LandmarkCoordinateSignals | {
1075
+ x: number;
1076
+ y: number;
1077
+ z?: number;
1078
+ screenX?: number;
1079
+ screenY?: number;
1080
+ }, options?: PinToLandmarkOptions): T;
1081
+ declare function pinNodeToObject<T extends {
1082
+ x?: unknown;
1083
+ y?: unknown;
1084
+ z?: unknown;
1085
+ width?: unknown;
1086
+ height?: unknown;
1087
+ opacity?: unknown;
1088
+ [key: string]: unknown;
1089
+ }>(node: T, target: ObjectTrackSignals | LandmarkCoordinateSignals | {
1090
+ x: number;
1091
+ y: number;
1092
+ z?: number;
1093
+ screenX?: number;
1094
+ screenY?: number;
1095
+ }, options?: PinToObjectOptions): T;
1096
+ //#endregion
1097
+ //#region src/tracking/temporal-object-tracker.d.ts
1098
+ interface TemporalTrackerOptions {
1099
+ /** Minimum Intersection-over-Union to associate a detection with an existing track. Default: 0.25 */
1100
+ readonly iouThreshold?: number;
1101
+ /** Number of frames a lost track is coasted via velocity extrapolation before deletion. Default: 15 */
1102
+ readonly maxMissedFrames?: number;
1103
+ /** Minimum consecutive hits before a tentative track is confirmed active. Default: 1 */
1104
+ readonly minHits?: number;
1105
+ /** Position smoothing weight [0, 1] where 1.0 is instantaneous and 0.0 is fully damped. Default: 0.75 */
1106
+ readonly positionSmoothing?: number;
1107
+ /** Whether tracks remain marked active during coasting frames. Default: true */
1108
+ readonly activeDuringCoast?: boolean;
1109
+ }
1110
+ /** Axis-aligned box in any consistent unit (pixels or normalized). */
1111
+ type PixelBox = Pick<ObjectBoundingBox, "originX" | "originY" | "width" | "height">;
1112
+ declare function computeIoU(boxA: PixelBox, boxB: PixelBox): number;
1113
+ declare class TemporalObjectTracker {
1114
+ private _nextTrackId;
1115
+ private _tracks;
1116
+ private _iouThreshold;
1117
+ private _maxMissedFrames;
1118
+ private _minHits;
1119
+ private _positionSmoothing;
1120
+ private _activeDuringCoast;
1121
+ constructor(options?: TemporalTrackerOptions);
1122
+ /**
1123
+ * Updates the multi-object tracker with new detections for the current frame.
1124
+ *
1125
+ * @param detections Raw detections found on the current frame
1126
+ * @param frame Current frame index
1127
+ * @param fps Video framerate (used for physical velocity estimation)
1128
+ * @returns Stable, temporal list of TrackedObjects
1129
+ */
1130
+ update(detections: readonly DetectedObject[], _frame: number, fps?: number): TrackedObject[];
1131
+ /**
1132
+ * Tracks objects across a full sequence of frames with gap interpolation,
1133
+ * boundary extrapolation (ensuring frame 0 through end are tracked),
1134
+ * and temporal Gaussian smoothing.
1135
+ *
1136
+ * Guarantees zero flicker, zero drift, and frame-accurate stability across the entire video.
1137
+ */
1138
+ trackSequence(perFrameDetections: readonly (readonly DetectedObject[])[], totalFrames: number, fps?: number): TrackedObject[][];
1139
+ /**
1140
+ * Resets all internal track state.
1141
+ */
1142
+ reset(): void;
1143
+ }
1144
+ //#endregion
1145
+ //#region src/tracking/pose-track-matcher.d.ts
1146
+ declare function getPersonBoundingBox(person: PosePerson): PixelBox;
1147
+ /**
1148
+ * Matches detected pose people with active person tracks using greedy bipartite IoU.
1149
+ *
1150
+ * Annotates each person with the matched trackId.
1151
+ */
1152
+ declare function matchPoseToTracks(people: readonly PosePerson[], trackedObjects: readonly TrackedObject[], minIoU?: number): PosePerson[];
1153
+ //#endregion
1154
+ //#region src/vision-node.d.ts
1155
+ /**
1156
+ * Bundle returned by `VisionNode.attach()`: the reactive signals surface plus the lazy lifecycle
1157
+ * helpers (all zero-I/O until `runner()` really needs a model).
1158
+ */
1159
+ interface VisionAttachedBundle extends VisionBundle {
1160
+ readonly node: VisionNodeSpec;
1161
+ /** Warm the enabled tasks' models/sessions ahead of first use. */
1162
+ ready(): Promise<VisionRunner>;
1163
+ /** Lazily-created runner (downloads happen on first real inference). */
1164
+ runner(): VisionRunner;
1165
+ /** Release this node's runner/sessions. */
1166
+ close(): void;
1167
+ }
1168
+ /**
1169
+ * Vision node.
1170
+ *
1171
+ * LAZY BY CONSTRUCTION: creating a VisionNode performs ZERO I/O — no model downloads, no
1172
+ * sessions, no file probes. The first inference call (or an explicit `await vision.ready()`)
1173
+ * downloads the required models.
1174
+ */
1175
+ declare class VisionNode {
1176
+ readonly id: string;
1177
+ readonly kind: "vision";
1178
+ readonly source: VisionInputSource;
1179
+ readonly config: VisionConfig;
1180
+ readonly vision: VisionBundle;
1181
+ private _runner?;
1182
+ /** Runner construction options (modelsDir, provider, store, …) — injectable for tests/agents. */
1183
+ private readonly _runnerOptions;
1184
+ constructor(source: VisionInputSource, config?: VisionConfig, bundleOptions?: VisionBundleOptions, runnerOptions?: VisionRunnerOptions);
1185
+ /** Lazily-created runner; downloads happen on its first inference. */
1186
+ runner(): VisionRunner;
1187
+ /** Explicit warm-up — downloads the models for the enabled tasks, ahead of first inference. */
1188
+ ready(): Promise<VisionRunner>;
1189
+ close(): void;
1190
+ static attach(source: VisionInputSource, config?: VisionConfig, bundleOptions?: VisionBundleOptions): VisionAttachedBundle;
1191
+ toNode(): VisionNodeSpec;
1192
+ }
1193
+ declare namespace index_d_exports {
1194
+ export { AnalyzeSequenceOptions, AutoSessionProvider, COCO17_BONES, COCO17_KEYPOINTS, COCO17_KEYPOINT_COUNT, COCO17_KEYPOINT_NAMES, COCO_CLASSES, Camera3DSpec, ClassCollectionSignals, ClassSignals, CocoClass, InputTransform, InstanceMask, LandmarkCoordinateSignals, MaskBounds, MaskCollectionSignals, MaskTrackSignals, ModelDownloadStatus, NodeSessionProvider, NodeWebGPUProvider, NodeWebGPUProviderOptions, ObjectAnchorName, ObjectAnchorsSignals, ObjectBoundingBoxSignals, ObjectCollectionSignals, ObjectKinematicsSignals, ObjectTrackSignals, PREPROCESS_BY_FAMILY, PersonMatte, PersonMatteSignals, PinToLandmarkOptions, PinToObjectOptions, PixelBox, PoseKeypoint, PoseLandmarkSignals, PosePerson, PoseResult, PoseSkeletonOptions, PoseSkeletonRenderer, PreprocessSpec, ProjectedScreenCoordinate, RgbaImage, RtmdetInsDecodeOptions, RtmdetInsOutputs, RtmoDecodeOptions, RtmoOutputs, SegmentationResult, SegmentationSignals, SegmentationTextureOptions, SegmentationTexturePool, SelfieDecodeOptions, SessionProvider, SkeletonBone, SpatialLandmarkTransformer, TemporalObjectTracker, TemporalTrackerOptions, TrackPoseSignals, VISION_MODELS, VISION_TASKS, VISION_VARIANTS, VisionAnalysisClassSummary, VisionAnalysisReport, VisionAnalysisTrackSummary, VisionAttachedBundle, VisionBundle, VisionBundleOptions, VisionImageInput, VisionInputSource, VisionModelDescriptor, VisionModelFamily, VisionModelKey, VisionModelStore, VisionModelStoreOptions, VisionNode, VisionRunner, VisionRunnerOptions, VisionSession, VisionSummary, VisionTask, VisionTensor, VisionTensorInput, VisionTensorOutputs, VisionVariant, analyzeSequence, computeInputTransform, computeIoU, createDefaultSessionProvider, createNeutralLandmarks, createNeutralObjectResult, createNeutralPoseResult, createNodeWebGPUProvider, createVisionBundle, decodeRtmdetIns, decodeRtmo, decodeSelfie, ensureNodeWebGPU, getDefaultModelsDir, getPersonBoundingBox, imageToTensor, maskBounds, matchPoseToTracks, mergeSubjectMask, modelKeyFor, pinNodeToLandmark, pinNodeToObject, preprocessSpec, probabilityToAlpha, selectSubjectMask, setDefaultSessionProvider, toSourcePoint };
1195
+ }
1196
+ //#endregion
1197
+ export { SegmentationTexturePool as $, getDefaultModelsDir as $t, createVisionBundle as A, PinToObjectOptions as At, OrtWebSession as B, VisionSummary as Bt, ObjectTrackSignals as C, VisionRunner as Ct, TrackPoseSignals as D, ObjectAnchorName as Dt, SegmentationSignals as E, LandmarkCoordinateSignals as Et, NodeWebGPUProvider as F, VisionAnalysisClassSummary as Ft, hasWebGPU as G, VisionTensor as Gt, WebGPUProvider as H, NodeSessionProvider as Ht, NodeWebGPUProviderOptions as I, VisionAnalysisReport as It, createNeutralPoseResult as J, createDefaultSessionProvider as Jt, createNeutralLandmarks as K, VisionTensorInput as Kt, createNodeWebGPUProvider as L, VisionAnalysisTrackSummary as Lt, maskBounds as M, PosePerson as Mt, mergeSubjectMask as N, PoseResult as Nt, VisionBundle as O, PersonMatte as Ot, selectSubjectMask as P, SegmentationResult as Pt, SegmentationTextureOptions as Q, VisionModelStoreOptions as Qt, ensureNodeWebGPU as R, VisionImageInput as Rt, ObjectKinematicsSignals as S, analyzeSequence as St, PoseLandmarkSignals as T, InstanceMask as Tt, WebGPUProviderOptions as U, SessionProvider as Ut, OrtWebTensor as V, AutoSessionProvider as Vt, createWebGPUProvider as W, VisionSession as Wt, COCO17_KEYPOINTS as X, ModelDownloadStatus as Xt, COCO17_BONES as Y, setDefaultSessionProvider as Yt, COCO17_KEYPOINT_NAMES as Z, VisionModelStore as Zt, MaskCollectionSignals as _, computeInputTransform as _t, matchPoseToTracks as a, VisionModelDescriptor as an, COCO17_KEYPOINT_COUNT as at, ObjectBoundingBoxSignals as b, toSourcePoint as bt, TemporalTrackerOptions as c, VisionTask as cn, decodeRtmo as ct, pinNodeToObject as d, decodeRtmdetIns as dt, COCO_CLASSES as en, PoseSkeletonOptions as et, Camera3DSpec as f, probabilityToAlpha as ft, ClassSignals as g, RgbaImage as gt, ClassCollectionSignals as h, PreprocessSpec as ht, getPersonBoundingBox as i, VISION_VARIANTS as in, decodeSelfie as it, MaskBounds as j, PoseKeypoint as jt, VisionBundleOptions as k, PinToLandmarkOptions as kt, computeIoU as l, VisionVariant as ln, RtmdetInsDecodeOptions as lt, SpatialLandmarkTransformer as m, PREPROCESS_BY_FAMILY as mt, VisionAttachedBundle as n, VISION_MODELS as nn, SkeletonBone as nt, PixelBox as o, VisionModelFamily as on, RtmoDecodeOptions as ot, ProjectedScreenCoordinate as p, InputTransform as pt, createNeutralObjectResult as q, VisionTensorOutputs as qt, VisionNode as r, VISION_TASKS as rn, SelfieDecodeOptions as rt, TemporalObjectTracker as s, VisionModelKey as sn, RtmoOutputs as st, index_d_exports as t, CocoClass as tn, PoseSkeletonRenderer as tt, pinNodeToLandmark as u, modelKeyFor as un, RtmdetInsOutputs as ut, MaskTrackSignals as v, imageToTensor as vt, PersonMatteSignals as w, VisionRunnerOptions as wt, ObjectCollectionSignals as x, AnalyzeSequenceOptions as xt, ObjectAnchorsSignals as y, preprocessSpec as yt, OrtWebModule as z, VisionInputSource as zt };
1198
+ //# sourceMappingURL=index-CoAN0xxG.d.mts.map