@framefields/vision 0.0.0-stage → 2.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1118 @@
1
+ import { n as __reExport, t as __exportAll } from "./chunk-CFhWmker.mjs";
2
+ import { CanonicalBoneDef as SkeletonBone, PoseSkeletonOptions } from "@framefields/tensor-webgpu";
3
+ import * as _framefields_core0 from "@framefields/core";
4
+ import { DetectedObject, FrameContext, Landmark3D, ObjectBoundingBox, ProgrammaticSignal, TensorData, TrackedObject, VisionConfig, VisionNodeSpec, VisionTask, VisionVariant } from "@framefields/core";
5
+
6
+ //#region src/model/registry.d.ts
7
+
8
+ declare const VISION_TASKS: readonly VisionTask[];
9
+ declare const VISION_VARIANTS: readonly VisionVariant[];
10
+ type VisionModelKey = "rtmdet-ins-t" | "rtmdet-ins-s" | "rtmdet-ins-m" | "rtmo-t" | "rtmo-s" | "rtmo-m" | "selfie-square" | "selfie-landscape";
11
+ /** Which decoder a model's outputs feed (see runner/vision-runner.ts). */
12
+ type VisionModelFamily = "rtmdet-ins" | "rtmo" | "selfie";
13
+ interface VisionModelDescriptor {
14
+ readonly key: VisionModelKey;
15
+ readonly family: VisionModelFamily;
16
+ /** Cache filename (also the path appended to a custom `baseUrl`). */
17
+ readonly filename: string;
18
+ /** Pinned download URL (immutable revision). */
19
+ readonly url: string;
20
+ /** Exact byte size of the pinned file. */
21
+ readonly bytes: number;
22
+ /** Hex SHA-256 of the pinned file. */
23
+ readonly sha256: string;
24
+ /** Model input size, `[width, height]` — must match the ONNX graph's input shape. */
25
+ readonly input: readonly [number, number];
26
+ /** Upstream project and license, surfaced in docs and errors. */
27
+ readonly source: string;
28
+ }
29
+ declare const VISION_MODELS: Readonly<Record<VisionModelKey, VisionModelDescriptor>>;
30
+ /**
31
+ * Resolves the model for a task. `detect` and `segment` share RTMDet-Ins (one forward pass
32
+ * yields both); `matte` picks the landscape Selfie Segmenter for wide frames.
33
+ */
34
+ declare function modelKeyFor(task: VisionTask, variant?: VisionVariant, aspect?: number): VisionModelKey;
35
+ /** COCO 80 class names — index order matches the RTMDet-Ins `labels` output. */
36
+ declare const COCO_CLASSES: readonly ["person", "bicycle", "car", "motorcycle", "airplane", "bus", "train", "truck", "boat", "traffic light", "fire hydrant", "stop sign", "parking meter", "bench", "bird", "cat", "dog", "horse", "sheep", "cow", "elephant", "bear", "zebra", "giraffe", "backpack", "umbrella", "handbag", "tie", "suitcase", "frisbee", "skis", "snowboard", "sports ball", "kite", "baseball bat", "baseball glove", "skateboard", "surfboard", "tennis racket", "bottle", "wine glass", "cup", "fork", "knife", "spoon", "bowl", "banana", "apple", "sandwich", "orange", "broccoli", "carrot", "hot dog", "pizza", "donut", "cake", "chair", "couch", "potted plant", "bed", "dining table", "toilet", "tv", "laptop", "mouse", "remote", "keyboard", "cell phone", "microwave", "oven", "toaster", "sink", "refrigerator", "book", "clock", "vase", "scissors", "teddy bear", "hair drier", "toothbrush"];
37
+ type CocoClass = (typeof COCO_CLASSES)[number];
38
+ //#endregion
39
+ //#region src/model/model-store.d.ts
40
+ type ModelDownloadStatus = "pending" | "downloading" | "ready" | "error";
41
+ interface VisionModelStoreOptions {
42
+ readonly modelsDir?: string;
43
+ /**
44
+ * Mirror origin: models are fetched from `<baseUrl>/<filename>` instead of their pinned
45
+ * Hugging Face URLs. Defaults to `$FRAMEFIELDS_MODELS_BASE_URL` when set.
46
+ */
47
+ readonly baseUrl?: string;
48
+ readonly timeoutMs?: number;
49
+ /** Extra attempts after a network error, HTTP 429 or 5xx. Default 2. */
50
+ readonly retries?: number;
51
+ /** Base backoff between attempts (doubles each retry). Default 1000 ms. */
52
+ readonly retryDelayMs?: number;
53
+ /**
54
+ * Verify byte size + SHA-256 against the registry (default true). Only offline tests
55
+ * that serve fake bytes should turn this off.
56
+ */
57
+ readonly verify?: boolean;
58
+ /** Injectable fetch — offline unit tests inject a mock. */
59
+ readonly fetchImpl?: typeof fetch;
60
+ readonly onProgress?: (key: VisionModelKey, bytesLoaded: number, totalBytes: number) => void;
61
+ }
62
+ /**
63
+ * Resolves the default model cache directory:
64
+ * 1. `FRAMEFIELDS_MODELS_DIR`
65
+ * 2. `~/.cache/framefields/models`
66
+ */
67
+ declare function getDefaultModelsDir(): string;
68
+ /**
69
+ * Lazy model store. The constructor performs ZERO I/O; the only entry that can hit the
70
+ * network is `ensure(key)`, called by an inference function the first time it runs.
71
+ * Concurrent callers of the same key share a single in-flight download (promise dedupe),
72
+ * downloads are written atomically (temp + rename), and verified files on disk are reused
73
+ * across processes.
74
+ */
75
+ declare class VisionModelStore {
76
+ private readonly _modelsDir;
77
+ private readonly _baseUrl?;
78
+ private readonly _timeoutMs?;
79
+ private readonly _retries;
80
+ private readonly _retryDelayMs;
81
+ private readonly _verify;
82
+ private readonly _fetch;
83
+ private readonly _onProgress?;
84
+ private _inflight;
85
+ private _buffers;
86
+ private _status;
87
+ constructor(options?: VisionModelStoreOptions);
88
+ get modelsDir(): string;
89
+ pathFor(key: VisionModelKey): string;
90
+ descriptor(key: VisionModelKey): VisionModelDescriptor;
91
+ /** The URL `ensure(key)` downloads from (mirror when `baseUrl` is set). */
92
+ urlFor(key: VisionModelKey): string;
93
+ /** True when a complete cached file exists (exact registry size when verifying). */
94
+ has(key: VisionModelKey): boolean;
95
+ /** Drops a cached model (file + memory) so the next `ensure` re-downloads it. */
96
+ evict(key: VisionModelKey): Promise<void>;
97
+ status(key: VisionModelKey): ModelDownloadStatus;
98
+ /** Snapshot of every tracked model's status (empty until anything is fetched). */
99
+ statuses(): ReadonlyMap<VisionModelKey, ModelDownloadStatus>;
100
+ /** True when the model is resident in memory (loaded or downloaded this process). */
101
+ isLoaded(key: VisionModelKey): boolean;
102
+ /**
103
+ * THE only download entry point. Returns the model bytes from memory (cached),
104
+ * loading from disk or fetching from the registry URL as needed.
105
+ */
106
+ ensure(key: VisionModelKey): Promise<Uint8Array>;
107
+ /** Explicit warm-up — fetch multiple models ahead of use (agents, renderers). */
108
+ preload(keys: readonly VisionModelKey[]): Promise<void>;
109
+ private load;
110
+ /** Fetches with retries on transient failures (network errors, HTTP 429/5xx). */
111
+ private download;
112
+ private verifyBytes;
113
+ private readBuffer;
114
+ /** Drops in-memory buffers (model stays on disk for the next run). */
115
+ clearMemoryCache(): void;
116
+ }
117
+ //#endregion
118
+ //#region src/runtime/session-provider.d.ts
119
+ /**
120
+ * Minimal ONNX Runtime surface — isolates the ORT API to exactly two files so the decode
121
+ * and signals layers never touch it. Node path uses onnxruntime-node (CPU EP); the web
122
+ * entry can swap in a WebGPU-backed provider with the same interface.
123
+ */
124
+ /** Model output tensor. Integer outputs (e.g. RTMDet-Ins `labels`) arrive as BigInt64Array. */
125
+ interface VisionTensor {
126
+ readonly data: Float32Array | BigInt64Array | Int32Array;
127
+ readonly dims: readonly number[];
128
+ }
129
+ interface VisionTensorInput {
130
+ readonly name: string;
131
+ readonly data: Float32Array;
132
+ readonly dims: readonly number[];
133
+ }
134
+ type VisionTensorOutputs = Readonly<Record<string, VisionTensor>>;
135
+ interface VisionSession {
136
+ run(input: VisionTensorInput): Promise<VisionTensorOutputs>;
137
+ release(): void;
138
+ }
139
+ interface SessionProvider {
140
+ readonly kind: "node" | "webgpu";
141
+ createSession(modelBytes: Uint8Array): Promise<VisionSession>;
142
+ }
143
+ /**
144
+ * Sets the provider runners use when none is passed, for hosts that create
145
+ * runners indirectly (e.g. a browser player rendering vision nodes). `undefined`
146
+ * restores the default, onnxruntime-node.
147
+ */
148
+ declare function setDefaultSessionProvider(factory: (() => SessionProvider) | undefined): void;
149
+ /** The provider for a runner created without one. */
150
+ declare function createDefaultSessionProvider(): SessionProvider;
151
+ /** onnxruntime-node provider. The native module is imported lazily on first session. */
152
+ declare class NodeSessionProvider implements SessionProvider {
153
+ readonly kind: "node";
154
+ private ortPromise?;
155
+ private ort;
156
+ createSession(modelBytes: Uint8Array): Promise<VisionSession>;
157
+ }
158
+ declare namespace types_d_exports {
159
+ export { InstanceMask, LandmarkCoordinateSignals, ObjectAnchorName, PersonMatte, PinToLandmarkOptions, PinToObjectOptions, PoseKeypoint, PosePerson, PoseResult, SegmentationResult, VisionAnalysisClassSummary, VisionAnalysisReport, VisionAnalysisTrackSummary, VisionImageInput, VisionInputSource, VisionSummary };
160
+ }
161
+ import * as import__framefields_core from "@framefields/core";
162
+ type VisionInputSource = string | {
163
+ inputHandleId?: string;
164
+ id?: string;
165
+ [key: string]: unknown;
166
+ };
167
+ /** RGBA pixel buffer accepted by the runner. */
168
+ interface VisionImageInput {
169
+ readonly data: Uint8ClampedArray | Uint8Array;
170
+ readonly width: number;
171
+ readonly height: number;
172
+ }
173
+ /** Instance segmentation mask aligned to the ORIGINAL source frame. */
174
+ interface InstanceMask {
175
+ readonly category: string;
176
+ readonly mask: Uint8Array;
177
+ readonly width: number;
178
+ readonly height: number;
179
+ readonly area: number;
180
+ readonly coverage: number;
181
+ /** Index of the detection this mask belongs to (see segmentation result detections). */
182
+ readonly detectionIndex: number;
183
+ /** Assigned by the caller (renderer) after temporal tracking — matches TrackedObject.trackId. */
184
+ readonly trackId?: number;
185
+ }
186
+ interface SegmentationResult {
187
+ readonly detections: readonly DetectedObject[];
188
+ readonly masks: readonly InstanceMask[];
189
+ }
190
+ /** Person-vs-background alpha (Selfie Segmenter), aligned to the ORIGINAL source frame. */
191
+ interface PersonMatte {
192
+ readonly mask: Uint8Array;
193
+ readonly width: number;
194
+ readonly height: number;
195
+ readonly coverage: number;
196
+ }
197
+ interface PoseKeypoint {
198
+ readonly x: number;
199
+ readonly y: number;
200
+ readonly visibility: number;
201
+ }
202
+ interface PosePerson {
203
+ readonly score: number;
204
+ readonly boundingBox: ObjectBoundingBox;
205
+ /** 17 keypoints in COCO-17 order (see COCO17_KEYPOINTS). */
206
+ readonly keypoints: readonly PoseKeypoint[];
207
+ readonly trackId?: number;
208
+ }
209
+ interface PoseResult {
210
+ readonly people: readonly PosePerson[];
211
+ }
212
+ /**
213
+ * Deterministic, serializable snapshot of one frame's signals.
214
+ * Built synchronously from the per-frame caches, so it is safe to call inside a frame hook.
215
+ */
216
+ interface VisionSummary {
217
+ readonly frame: number;
218
+ readonly objects: ReadonlyArray<{
219
+ readonly trackId: number;
220
+ readonly category: string;
221
+ readonly score: number;
222
+ /** Source-space pixel center, `[x, y]`. */
223
+ readonly center: readonly [number, number];
224
+ readonly speed: number;
225
+ readonly active: boolean;
226
+ }>;
227
+ readonly classes: readonly string[];
228
+ /** Per-class instance count for the frame (only non-zero classes present). */
229
+ readonly masks: Readonly<Record<string, number>>;
230
+ }
231
+ /** Per-track aggregate in an analysis report. */
232
+ interface VisionAnalysisTrackSummary {
233
+ readonly trackId: number;
234
+ readonly category: string;
235
+ /** Inclusive frame range the track was active over, `[start, end]`. */
236
+ readonly frames: readonly [number, number];
237
+ readonly speedMeanPxS: number;
238
+ /** Center path sampled every `pathStride` frames, `[x, y]` pairs. */
239
+ readonly centerPath: ReadonlyArray<readonly [number, number]>;
240
+ }
241
+ interface VisionAnalysisClassSummary {
242
+ readonly framesPresent: number;
243
+ readonly totalFrames: number;
244
+ readonly maxConfidence: number;
245
+ readonly tracks: number;
246
+ }
247
+ interface VisionAnalysisReport {
248
+ readonly source: string;
249
+ readonly totalFrames: number;
250
+ readonly fps: number;
251
+ readonly durationSec: number;
252
+ readonly tasks: readonly string[];
253
+ /** What was fetched, and how long it took — surfaced so agents can budget cold starts. */
254
+ readonly modelDownloads: Readonly<Record<string, {
255
+ readonly bytes: number;
256
+ readonly ms: number;
257
+ }>>;
258
+ readonly inferenceMs: number;
259
+ readonly trackCount: number;
260
+ readonly classes: Readonly<Record<string, VisionAnalysisClassSummary>>;
261
+ readonly tracks: readonly VisionAnalysisTrackSummary[];
262
+ readonly masks: Readonly<Record<string, {
263
+ readonly meanCoverage: number;
264
+ }>>;
265
+ }
266
+ interface PinToLandmarkOptions {
267
+ readonly offsetX?: number;
268
+ readonly offsetY?: number;
269
+ readonly offsetZ?: number;
270
+ readonly trackRotation?: boolean;
271
+ readonly scaleWithPerspective?: boolean;
272
+ }
273
+ interface LandmarkCoordinateSignals {
274
+ readonly x: _framefields_core0.ProgrammaticSignal;
275
+ readonly y: _framefields_core0.ProgrammaticSignal;
276
+ readonly z: _framefields_core0.ProgrammaticSignal;
277
+ readonly screenX: _framefields_core0.ProgrammaticSignal;
278
+ readonly screenY: _framefields_core0.ProgrammaticSignal;
279
+ }
280
+ type ObjectAnchorName = "center" | "topCenter" | "bottomCenter" | "topLeft" | "topRight" | "centerLeft" | "centerRight" | "bottomLeft" | "bottomRight";
281
+ interface PinToObjectOptions extends PinToLandmarkOptions {
282
+ readonly anchor?: ObjectAnchorName;
283
+ readonly matchWidth?: boolean;
284
+ readonly matchHeight?: boolean;
285
+ readonly smoothFrames?: number;
286
+ readonly hideWhenLost?: boolean;
287
+ }
288
+ //#endregion
289
+ //#region src/runner/vision-runner.d.ts
290
+ interface VisionRunnerOptions {
291
+ /** Model size for detect/segment/pose. Default "s". */
292
+ readonly variant?: VisionVariant;
293
+ /** Minimum detection / person score, 0..1. Default 0.3. */
294
+ readonly confidence?: number;
295
+ /** COCO class-name filter for detect/segment (all classes when omitted). */
296
+ readonly classes?: readonly string[];
297
+ readonly maskThreshold?: number;
298
+ readonly featherRadius?: number;
299
+ readonly modelsDir?: string;
300
+ /** Mirror origin for model downloads (see VisionModelStore). */
301
+ readonly baseUrl?: string;
302
+ readonly timeoutMs?: number;
303
+ /** Session backend. Default: onnxruntime-node (CPU). Browser: `createWebGPUProvider()`. */
304
+ readonly provider?: SessionProvider;
305
+ /** Inject a model store (tests, shared caches). Overrides modelsDir/baseUrl/timeoutMs. */
306
+ readonly store?: VisionModelStore;
307
+ readonly onProgress?: (key: VisionModelKey, bytesLoaded: number, totalBytes: number) => void;
308
+ }
309
+ /**
310
+ * Lazy vision runner.
311
+ *
312
+ * `create()` performs ZERO I/O: no downloads, no sessions. Each inference method downloads
313
+ * its model and warms its session on FIRST USE ONLY, then reuses them for the process
314
+ * lifetime. `detect()` and `segment()` share one RTMDet-Ins forward pass per frame.
315
+ * Call `close()` to release sessions.
316
+ */
317
+ declare class VisionRunner {
318
+ readonly variant: VisionVariant;
319
+ readonly confidence: number;
320
+ readonly classes?: readonly string[];
321
+ readonly maskThreshold: number;
322
+ readonly featherRadius?: number;
323
+ private readonly _store;
324
+ private readonly _provider;
325
+ private readonly _sessions;
326
+ private readonly _tensorScratch;
327
+ /** One RTMDet-Ins pass per frame buffer, shared by detect() and segment(). */
328
+ private _instancePass?;
329
+ private constructor();
330
+ /** Cheap: no downloads, no sessions, no side effects. */
331
+ static create(options?: VisionRunnerOptions): VisionRunner;
332
+ get modelsDir(): string;
333
+ /** Every registry model keyed to its current download state (downloaded → "ready"). */
334
+ get downloadStatus(): ReadonlyMap<VisionModelKey, ModelDownloadStatus>;
335
+ /** The model serving `task` (matte depends on frame aspect; defaults to square). */
336
+ modelKeyFor(task: VisionTask, aspect?: number): VisionModelKey;
337
+ /**
338
+ * Explicit warm-up: downloads + sessions for the given tasks, ahead of first use.
339
+ * `matte` warms both aspect variants (the frame aspect is unknown until inference).
340
+ */
341
+ preload(tasks?: readonly VisionTask[]): Promise<void>;
342
+ /** COCO-80 boxes (sorted by score, source pixels). Masks are not decoded. */
343
+ detect(image: VisionImageInput): Promise<readonly DetectedObject[]>;
344
+ /** COCO-80 boxes plus a frame-aligned soft mask per instance. */
345
+ segment(image: VisionImageInput): Promise<SegmentationResult>;
346
+ /** COCO-17 keypoints per person (source pixels), highest score first. */
347
+ pose(image: VisionImageInput): Promise<PoseResult>;
348
+ /** Person-vs-background alpha for the whole frame (Selfie Segmenter). */
349
+ matte(image: VisionImageInput): Promise<PersonMatte>;
350
+ close(): void;
351
+ private instancePass;
352
+ private instanceDecodeOptions;
353
+ private infer;
354
+ private session;
355
+ }
356
+ //#endregion
357
+ //#region src/analysis/analyze-sequence.d.ts
358
+ /**
359
+ * One-shot, ffmpeg-free sequence analysis.
360
+ *
361
+ * Pure frame-iterable analyzer: it never decodes media itself, so it stays engine-only and
362
+ * offline-testable. Callers (e.g. the framefields ingestion adapter) supply decoded frames.
363
+ * Model loading remains lazy — the runner downloads each task's model on the first frame
364
+ * that actually runs that task, and only the tasks requested here are ever touched.
365
+ */
366
+ interface AnalyzeSequenceOptions {
367
+ readonly runner: VisionRunner;
368
+ /** Label for the report ("clip.mp4" or similar). Defaults to "frames". */
369
+ readonly source?: string;
370
+ readonly fps?: number;
371
+ /** Tasks to run. Defaults to `["detect"]`. Only these models are ever downloaded. */
372
+ readonly tasks?: readonly VisionTask[];
373
+ /** Restrict class summaries to these categories (all when omitted). */
374
+ readonly categories?: readonly string[];
375
+ /** Tracker box-association IoU. Default 0.25. */
376
+ readonly iouThreshold?: number;
377
+ readonly maxMissedFrames?: number;
378
+ /** Sample the per-track center path every N frames. Default: ~50 points across the clip. */
379
+ readonly pathStride?: number;
380
+ /** Also aggregate instance-mask coverage (requires the `segment` task). Default false. */
381
+ readonly includeMasks?: boolean;
382
+ }
383
+ declare function analyzeSequence(frames: AsyncIterable<VisionImageInput> | Iterable<VisionImageInput>, options: AnalyzeSequenceOptions): Promise<VisionAnalysisReport>;
384
+ //#endregion
385
+ //#region src/decode/preprocess.d.ts
386
+ /**
387
+ * Image → NCHW float tensor preprocessing, plus the inverse mapping back to source pixels.
388
+ *
389
+ * Each model family was exported with its own preprocessing contract:
390
+ *
391
+ * | Model | Resize | Pad | Channels | Values |
392
+ * | ----------- | ---------------------------- | ---- | -------- | -------------------------- |
393
+ * | RTMDet-Ins | keep ratio, content top-left | 114 | BGR | (v − mean) / std (ImageNet) |
394
+ * | RTMO | keep ratio, content centered | 114 | BGR | raw 0..255 |
395
+ * | Selfie | stretch to input | — | RGB | v / 255 |
396
+ *
397
+ * Getting any of these wrong does not crash — it silently degrades accuracy — so the specs
398
+ * live here and every runner path goes through `imageToTensor`. Input sizes differ per model
399
+ * (RTMO-t is 416², the rest of RTMO 640²), so they come from the registry, never from here.
400
+ */
401
+ interface RgbaImage {
402
+ readonly data: Uint8ClampedArray | Uint8Array;
403
+ readonly width: number;
404
+ readonly height: number;
405
+ }
406
+ interface PreprocessSpec {
407
+ /** Model input size in pixels. */
408
+ readonly width: number;
409
+ readonly height: number;
410
+ /** `topLeft` / `center` keep aspect ratio and pad; `stretch` fills the input. */
411
+ readonly fit: "topLeft" | "center" | "stretch";
412
+ readonly channels: "rgb" | "bgr";
413
+ /** Per-channel mean, in output channel order, on the 0..255 scale. */
414
+ readonly mean: readonly [number, number, number];
415
+ /** Per-channel std, in output channel order, on the 0..255 scale. */
416
+ readonly std: readonly [number, number, number];
417
+ /** Padding value on the 0..255 scale (normalized like any other pixel). */
418
+ readonly padValue: number;
419
+ }
420
+ /** Maps model-input pixels back to source pixels: `src = (input − offset) / scale`. */
421
+ interface InputTransform {
422
+ readonly scaleX: number;
423
+ readonly scaleY: number;
424
+ readonly offsetX: number;
425
+ readonly offsetY: number;
426
+ readonly inputWidth: number;
427
+ readonly inputHeight: number;
428
+ }
429
+ type Normalization = Omit<PreprocessSpec, "width" | "height">;
430
+ /** Per-family normalization; the input size always comes from the model's registry entry. */
431
+ declare const PREPROCESS_BY_FAMILY: Readonly<Record<"rtmdet-ins" | "rtmo" | "selfie", Normalization>>;
432
+ declare function preprocessSpec(family: keyof typeof PREPROCESS_BY_FAMILY, input: readonly [number, number]): PreprocessSpec;
433
+ declare function computeInputTransform(srcW: number, srcH: number, spec: Pick<PreprocessSpec, "width" | "height" | "fit">): InputTransform;
434
+ /** Model-input point → source pixel point. */
435
+ declare function toSourcePoint(x: number, y: number, t: InputTransform): {
436
+ x: number;
437
+ y: number;
438
+ };
439
+ /**
440
+ * Resizes RGBA → NCHW float32 `[1, 3, height, width]` with bilinear sampling, padding,
441
+ * channel reordering and normalization per `spec`. Reuses `out` when large enough
442
+ * (hot-path frame loop).
443
+ */
444
+ declare function imageToTensor(image: RgbaImage, spec: PreprocessSpec, out?: Float32Array): {
445
+ tensor: Float32Array;
446
+ transform: InputTransform;
447
+ };
448
+ //#endregion
449
+ //#region src/decode/rtmdet-ins.d.ts
450
+ /**
451
+ * RTMDet-Ins decode (mmdeploy end2end export — NMS and mask assembly are baked into the graph).
452
+ *
453
+ * Outputs:
454
+ * dets [1, N, 5] x0, y0, x1, y1 (input px), score
455
+ * labels [1, N] COCO-80 class index (int64)
456
+ * masks [1, N, Hm, Wm] per-instance probabilities (already sigmoided), input-aligned
457
+ */
458
+ interface RtmdetInsOutputs {
459
+ readonly dets: ArrayLike<number>;
460
+ readonly labels: ArrayLike<number | bigint>;
461
+ /** Omit to decode boxes only (detection without masks). */
462
+ readonly masks?: ArrayLike<number>;
463
+ readonly count: number;
464
+ readonly maskWidth: number;
465
+ readonly maskHeight: number;
466
+ }
467
+ interface RtmdetInsDecodeOptions {
468
+ readonly confidence: number;
469
+ /** Class-name filter, case-insensitive (all classes when omitted). */
470
+ readonly classes?: readonly string[];
471
+ readonly classNames: readonly string[];
472
+ readonly transform: InputTransform;
473
+ readonly sourceWidth: number;
474
+ readonly sourceHeight: number;
475
+ /** Mask probability threshold. Default 0.5. */
476
+ readonly maskThreshold?: number;
477
+ /** Soft-edge band around the threshold. Default 0.05. */
478
+ readonly featherRadius?: number;
479
+ }
480
+ declare function decodeRtmdetIns(out: RtmdetInsOutputs, opts: RtmdetInsDecodeOptions): SegmentationResult;
481
+ /**
482
+ * Probability → 0..255 alpha. A hard threshold when `feather` is 0, otherwise a linear ramp
483
+ * across `[threshold − feather, threshold + feather]` for anti-aliased edges.
484
+ */
485
+ declare function probabilityToAlpha(threshold: number, feather: number): (p: number) => number;
486
+ //#endregion
487
+ //#region src/decode/rtmo.d.ts
488
+ /**
489
+ * RTMO decode (one-stage multi-person pose; NMS is baked into the export).
490
+ *
491
+ * Outputs:
492
+ * dets [1, N, 5] x0, y0, x1, y1 (input px), person score
493
+ * keypoints [1, N, 17, 3] x, y (input px), keypoint score — COCO-17 order
494
+ */
495
+ interface RtmoOutputs {
496
+ readonly dets: ArrayLike<number>;
497
+ readonly keypoints: ArrayLike<number>;
498
+ readonly count: number;
499
+ /** Values per keypoint (3 for x, y, score). */
500
+ readonly keypointStride: number;
501
+ }
502
+ interface RtmoDecodeOptions {
503
+ readonly confidence: number;
504
+ readonly transform: InputTransform;
505
+ readonly sourceWidth: number;
506
+ readonly sourceHeight: number;
507
+ /** Drop people with fewer keypoints at visibility ≥ 0.3. Default 3. */
508
+ readonly minVisibleKeypoints?: number;
509
+ /** Suppress a person whose box overlaps a higher-scored one above this IoU. Default 0.6. */
510
+ readonly iouThreshold?: number;
511
+ }
512
+ declare const COCO17_KEYPOINT_COUNT = 17;
513
+ declare function decodeRtmo(out: RtmoOutputs, opts: RtmoDecodeOptions): PoseResult;
514
+ //#endregion
515
+ //#region src/decode/selfie.d.ts
516
+ /**
517
+ * Selfie Segmenter decode: `alphas [1, 1, h, w]` person probability (stretched input) →
518
+ * frame-sized 0..255 alpha, bilinear-upsampled to the source resolution.
519
+ */
520
+ interface SelfieDecodeOptions {
521
+ readonly sourceWidth: number;
522
+ readonly sourceHeight: number;
523
+ /** Probability threshold. Default 0.5. */
524
+ readonly maskThreshold?: number;
525
+ /** Soft-edge band; the low-res matte reads best with a wide one. Default 0.2. */
526
+ readonly featherRadius?: number;
527
+ }
528
+ declare function decodeSelfie(alphas: ArrayLike<number>, matteWidth: number, matteHeight: number, opts: SelfieDecodeOptions): PersonMatte;
529
+ //#endregion
530
+ //#region src/gpu/pose-skeleton-renderer.d.ts
531
+ /**
532
+ * WebGPU skeleton renderer for pose results.
533
+ *
534
+ * Rasterizes the primary person's COCO-17 keypoints into an OpenPose-style
535
+ * conditioning texture natively in VRAM, using the COCO-17 bone set.
536
+ */
537
+ declare class PoseSkeletonRenderer {
538
+ private pipeline;
539
+ constructor(device: GPUDevice);
540
+ renderToTexture(keypoints: readonly PoseKeypoint[], options: PoseSkeletonOptions, bones?: readonly [{
541
+ readonly from: 0;
542
+ readonly to: 5;
543
+ readonly color: readonly [0, 1, 1, 1];
544
+ }, {
545
+ readonly from: 0;
546
+ readonly to: 6;
547
+ readonly color: readonly [0, 1, 1, 1];
548
+ }, {
549
+ readonly from: 5;
550
+ readonly to: 6;
551
+ readonly color: readonly [1, 0, 0, 1];
552
+ }, {
553
+ readonly from: 5;
554
+ readonly to: 7;
555
+ readonly color: readonly [1, 0.333, 0, 1];
556
+ }, {
557
+ readonly from: 7;
558
+ readonly to: 9;
559
+ readonly color: readonly [1, 0.667, 0, 1];
560
+ }, {
561
+ readonly from: 6;
562
+ readonly to: 8;
563
+ readonly color: readonly [1, 1, 0, 1];
564
+ }, {
565
+ readonly from: 8;
566
+ readonly to: 10;
567
+ readonly color: readonly [0.667, 1, 0, 1];
568
+ }, {
569
+ readonly from: 5;
570
+ readonly to: 11;
571
+ readonly color: readonly [0.333, 1, 0, 1];
572
+ }, {
573
+ readonly from: 6;
574
+ readonly to: 12;
575
+ readonly color: readonly [0, 1, 0, 1];
576
+ }, {
577
+ readonly from: 11;
578
+ readonly to: 12;
579
+ readonly color: readonly [0, 1, 0.333, 1];
580
+ }, {
581
+ readonly from: 11;
582
+ readonly to: 13;
583
+ readonly color: readonly [0, 1, 0.667, 1];
584
+ }, {
585
+ readonly from: 13;
586
+ readonly to: 15;
587
+ readonly color: readonly [0, 1, 1, 1];
588
+ }, {
589
+ readonly from: 12;
590
+ readonly to: 14;
591
+ readonly color: readonly [0, 0.667, 1, 1];
592
+ }, {
593
+ readonly from: 14;
594
+ readonly to: 16;
595
+ readonly color: readonly [0, 0.333, 1, 1];
596
+ }]): GPUTexture;
597
+ destroy(): void;
598
+ }
599
+ //#endregion
600
+ //#region src/gpu/segmentation-texture-pool.d.ts
601
+ interface SegmentationTextureOptions {
602
+ readonly width: number;
603
+ readonly height: number;
604
+ readonly format?: "r8unorm" | "rgba8unorm";
605
+ readonly label?: string;
606
+ }
607
+ declare class SegmentationTexturePool {
608
+ private device;
609
+ private pool;
610
+ constructor(device: GPUDevice);
611
+ getOrCreateTexture(key: string, options: SegmentationTextureOptions): GPUTexture;
612
+ uploadMask(key: string, maskData: Uint8Array | Float32Array, width: number, height: number): GPUTexture;
613
+ destroy(): void;
614
+ }
615
+ //#endregion
616
+ //#region src/pose/keypoints.d.ts
617
+ /** COCO-17 keypoint indices (RTMO output order). */
618
+ declare const COCO17_KEYPOINTS: {
619
+ readonly NOSE: 0;
620
+ readonly LEFT_EYE: 1;
621
+ readonly RIGHT_EYE: 2;
622
+ readonly LEFT_EAR: 3;
623
+ readonly RIGHT_EAR: 4;
624
+ readonly LEFT_SHOULDER: 5;
625
+ readonly RIGHT_SHOULDER: 6;
626
+ readonly LEFT_ELBOW: 7;
627
+ readonly RIGHT_ELBOW: 8;
628
+ readonly LEFT_WRIST: 9;
629
+ readonly RIGHT_WRIST: 10;
630
+ readonly LEFT_HIP: 11;
631
+ readonly RIGHT_HIP: 12;
632
+ readonly LEFT_KNEE: 13;
633
+ readonly RIGHT_KNEE: 14;
634
+ readonly LEFT_ANKLE: 15;
635
+ readonly RIGHT_ANKLE: 16;
636
+ };
637
+ declare const COCO17_KEYPOINT_NAMES: readonly ["nose", "leftEye", "rightEye", "leftEar", "rightEar", "leftShoulder", "rightShoulder", "leftElbow", "rightElbow", "leftWrist", "rightWrist", "leftHip", "rightHip", "leftKnee", "rightKnee", "leftAnkle", "rightAnkle"];
638
+ /** COCO-17 skeleton bones for the WebGPU skeleton renderer (indices above). */
639
+ declare const COCO17_BONES: readonly [{
640
+ readonly from: 0;
641
+ readonly to: 5;
642
+ readonly color: readonly [0, 1, 1, 1];
643
+ }, {
644
+ readonly from: 0;
645
+ readonly to: 6;
646
+ readonly color: readonly [0, 1, 1, 1];
647
+ }, {
648
+ readonly from: 5;
649
+ readonly to: 6;
650
+ readonly color: readonly [1, 0, 0, 1];
651
+ }, {
652
+ readonly from: 5;
653
+ readonly to: 7;
654
+ readonly color: readonly [1, 0.333, 0, 1];
655
+ }, {
656
+ readonly from: 7;
657
+ readonly to: 9;
658
+ readonly color: readonly [1, 0.667, 0, 1];
659
+ }, {
660
+ readonly from: 6;
661
+ readonly to: 8;
662
+ readonly color: readonly [1, 1, 0, 1];
663
+ }, {
664
+ readonly from: 8;
665
+ readonly to: 10;
666
+ readonly color: readonly [0.667, 1, 0, 1];
667
+ }, {
668
+ readonly from: 5;
669
+ readonly to: 11;
670
+ readonly color: readonly [0.333, 1, 0, 1];
671
+ }, {
672
+ readonly from: 6;
673
+ readonly to: 12;
674
+ readonly color: readonly [0, 1, 0, 1];
675
+ }, {
676
+ readonly from: 11;
677
+ readonly to: 12;
678
+ readonly color: readonly [0, 1, 0.333, 1];
679
+ }, {
680
+ readonly from: 11;
681
+ readonly to: 13;
682
+ readonly color: readonly [0, 1, 0.667, 1];
683
+ }, {
684
+ readonly from: 13;
685
+ readonly to: 15;
686
+ readonly color: readonly [0, 1, 1, 1];
687
+ }, {
688
+ readonly from: 12;
689
+ readonly to: 14;
690
+ readonly color: readonly [0, 0.667, 1, 1];
691
+ }, {
692
+ readonly from: 14;
693
+ readonly to: 16;
694
+ readonly color: readonly [0, 0.333, 1, 1];
695
+ }];
696
+ //#endregion
697
+ //#region src/runner/canonical-baselines.d.ts
698
+ /** Neutral fallbacks so signal evaluators always have a well-formed frame to read. */
699
+ declare function createNeutralObjectResult(): {
700
+ readonly objects: readonly TrackedObject[];
701
+ readonly rawDetections: readonly DetectedObject[];
702
+ };
703
+ declare function createNeutralPoseResult(): PoseResult;
704
+ /** 17 neutral COCO keypoints at frame center — used by pose signals when nobody is detected. */
705
+ declare function createNeutralLandmarks(): readonly Landmark3D[];
706
+ //#endregion
707
+ //#region src/segmentation/subject.d.ts
708
+ /**
709
+ * The frame's primary instance: the largest person when present, else the most confident
710
+ * instance (lowest `detectionIndex` — detections are score-sorted). Size alone is a poor
711
+ * signal without a person: the largest instance is usually a backdrop ("dining table").
712
+ */
713
+ declare function selectSubjectMask(masks: readonly InstanceMask[]): InstanceMask | undefined;
714
+ interface MaskBounds {
715
+ readonly x0: number;
716
+ readonly y0: number;
717
+ readonly x1: number;
718
+ readonly y1: number;
719
+ }
720
+ /** Tight pixel bounds of a mask's non-zero alpha, or null when empty. */
721
+ declare function maskBounds(mask: InstanceMask): MaskBounds | null;
722
+ /**
723
+ * Builds the full subject silhouette: the primary instance plus every comparably sized
724
+ * instance whose bounds sit mostly inside or across it (`overlap` = intersection / smaller
725
+ * box area; `maxGrowth` caps a part's box area relative to the subject's).
726
+ *
727
+ * COCO has no "clothing" class, so a flowing dress, a held guitar or a ridden bike comes back
728
+ * as its own instance (often mislabeled) — merging them keeps the whole figure in the cutout.
729
+ * The size cap keeps containers out: a small figure inside a tunnel or window detected as a
730
+ * huge "clock" must not drag the whole frame into the subject.
731
+ */
732
+ declare function mergeSubjectMask(masks: readonly InstanceMask[], overlap?: number, maxGrowth?: number): InstanceMask | undefined;
733
+ //#endregion
734
+ //#region src/signals/vision-bundle.d.ts
735
+ interface VisionBundleOptions {
736
+ readonly totalFrames?: number;
737
+ readonly fps?: number;
738
+ readonly width?: number;
739
+ readonly height?: number;
740
+ readonly cameraFov?: number;
741
+ readonly config?: types_d_exports.VisionConfig;
742
+ }
743
+ /** Named pose landmarks (COCO-17) plus numeric proxy access to any of the 17 keypoints. */
744
+ interface PoseLandmarkSignals {
745
+ readonly shoulder: LandmarkCoordinateSignals;
746
+ readonly leftShoulder: LandmarkCoordinateSignals;
747
+ readonly rightShoulder: LandmarkCoordinateSignals;
748
+ readonly leftElbow: LandmarkCoordinateSignals;
749
+ readonly rightElbow: LandmarkCoordinateSignals;
750
+ readonly leftWrist: LandmarkCoordinateSignals;
751
+ readonly rightWrist: LandmarkCoordinateSignals;
752
+ readonly leftHip: LandmarkCoordinateSignals;
753
+ readonly rightHip: LandmarkCoordinateSignals;
754
+ readonly leftKnee: LandmarkCoordinateSignals;
755
+ readonly rightKnee: LandmarkCoordinateSignals;
756
+ readonly leftAnkle: LandmarkCoordinateSignals;
757
+ readonly rightAnkle: LandmarkCoordinateSignals;
758
+ readonly nose: LandmarkCoordinateSignals;
759
+ readonly leftEye: LandmarkCoordinateSignals;
760
+ readonly rightEye: LandmarkCoordinateSignals;
761
+ readonly leftEar: LandmarkCoordinateSignals;
762
+ readonly rightEar: LandmarkCoordinateSignals;
763
+ get(keypointIndex: number): LandmarkCoordinateSignals;
764
+ }
765
+ interface TrackPoseSignals extends PoseLandmarkSignals {
766
+ readonly hasPose: ProgrammaticSignal;
767
+ readonly wristSpeed: ProgrammaticSignal;
768
+ readonly handRaised: ProgrammaticSignal;
769
+ readonly bodyTiltAngle: ProgrammaticSignal;
770
+ }
771
+ interface ObjectBoundingBoxSignals {
772
+ readonly x: ProgrammaticSignal;
773
+ readonly y: ProgrammaticSignal;
774
+ readonly width: ProgrammaticSignal;
775
+ readonly height: ProgrammaticSignal;
776
+ readonly screenX: ProgrammaticSignal;
777
+ readonly screenY: ProgrammaticSignal;
778
+ readonly screenWidth: ProgrammaticSignal;
779
+ readonly screenHeight: ProgrammaticSignal;
780
+ readonly aspectRatio: ProgrammaticSignal;
781
+ readonly area: ProgrammaticSignal;
782
+ }
783
+ interface ObjectAnchorsSignals {
784
+ readonly topLeft: LandmarkCoordinateSignals;
785
+ readonly topCenter: LandmarkCoordinateSignals;
786
+ readonly topRight: LandmarkCoordinateSignals;
787
+ readonly centerLeft: LandmarkCoordinateSignals;
788
+ readonly center: LandmarkCoordinateSignals;
789
+ readonly centerRight: LandmarkCoordinateSignals;
790
+ readonly bottomLeft: LandmarkCoordinateSignals;
791
+ readonly bottomCenter: LandmarkCoordinateSignals;
792
+ readonly bottomRight: LandmarkCoordinateSignals;
793
+ }
794
+ interface ObjectKinematicsSignals {
795
+ readonly vx: ProgrammaticSignal;
796
+ readonly vy: ProgrammaticSignal;
797
+ readonly speed: ProgrammaticSignal;
798
+ readonly acceleration: ProgrammaticSignal;
799
+ readonly headingRad: ProgrammaticSignal;
800
+ readonly headingDeg: ProgrammaticSignal;
801
+ }
802
+ interface ObjectTrackSignals {
803
+ readonly trackId: number;
804
+ readonly category: string;
805
+ readonly bounds: ObjectBoundingBoxSignals;
806
+ readonly anchors: ObjectAnchorsSignals;
807
+ readonly kinematics: ObjectKinematicsSignals;
808
+ readonly confidence: ProgrammaticSignal;
809
+ readonly active: ProgrammaticSignal;
810
+ readonly isCoasting: ProgrammaticSignal;
811
+ readonly age: ProgrammaticSignal;
812
+ readonly pose: TrackPoseSignals;
813
+ readonly center: LandmarkCoordinateSignals;
814
+ readonly topCenter: LandmarkCoordinateSignals;
815
+ readonly bottomCenter: LandmarkCoordinateSignals;
816
+ readonly topLeft: LandmarkCoordinateSignals;
817
+ readonly bottomLeft: LandmarkCoordinateSignals;
818
+ readonly [key: string]: unknown;
819
+ }
820
+ interface MaskTrackSignals {
821
+ readonly trackId: number;
822
+ readonly category: string;
823
+ readonly area: ProgrammaticSignal;
824
+ readonly coverage: ProgrammaticSignal;
825
+ readonly solidity: ProgrammaticSignal;
826
+ readonly bboxFill: ProgrammaticSignal;
827
+ /** Frame-sized {0,255} mask for the current frame (renderer-populated). */
828
+ data?: Uint8Array;
829
+ /** GPU-resident silhouette texture (renderer-populated via SegmentationTexturePool). */
830
+ texture?: GPUTexture;
831
+ }
832
+ interface MaskCollectionSignals {
833
+ get(trackId: number): MaskTrackSignals;
834
+ /** Largest instance of the current frame (person preferred when present). */
835
+ readonly subject: MaskTrackSignals;
836
+ readonly count: ProgrammaticSignal;
837
+ }
838
+ interface SegmentationSignals {
839
+ readonly humanSilhouette: MaskTrackSignals;
840
+ readonly subject: MaskTrackSignals;
841
+ readonly instanceMasks: MaskCollectionSignals;
842
+ /** Person-vs-background alpha from the Selfie Segmenter (`enableMatte`). */
843
+ readonly matte: PersonMatteSignals;
844
+ readonly stencilTexture?: GPUTexture;
845
+ }
846
+ interface PersonMatteSignals {
847
+ /** Mean person alpha over the frame, 0..1 (0 when no matte has run). */
848
+ readonly coverage: ProgrammaticSignal;
849
+ /** Frame-sized 0..255 alpha for the given frame, if a matte has run. */
850
+ at(frame: number): PersonMatte | undefined;
851
+ }
852
+ interface ClassSignals {
853
+ readonly count: ProgrammaticSignal;
854
+ readonly maxConfidence: ProgrammaticSignal;
855
+ readonly present: ProgrammaticSignal;
856
+ readonly primary: ObjectTrackSignals;
857
+ }
858
+ interface ClassCollectionSignals {
859
+ /** Cached per-class signals (stable identity). */
860
+ get(name: string): ClassSignals;
861
+ /** Sorted active category names for the current frame. */
862
+ readonly names: readonly string[];
863
+ readonly histogram: {
864
+ get(ctx?: FrameContext): TensorData;
865
+ readonly value: TensorData;
866
+ };
867
+ }
868
+ interface ObjectCollectionSignals {
869
+ get(trackId: number): ObjectTrackSignals;
870
+ byCategory(category: string, rank?: number): ObjectTrackSignals;
871
+ readonly primary: ObjectTrackSignals;
872
+ readonly count: ProgrammaticSignal;
873
+ hasCategory(category: string): ProgrammaticSignal;
874
+ getActiveTracks(frame: number): readonly TrackedObject[];
875
+ readonly detectedCategories: readonly string[];
876
+ }
877
+ declare class VisionBundle {
878
+ readonly objects: ObjectCollectionSignals;
879
+ readonly poseLandmarks: PoseLandmarkSignals;
880
+ readonly masks: MaskCollectionSignals;
881
+ readonly segmentation: SegmentationSignals;
882
+ readonly classes: ClassCollectionSignals;
883
+ readonly poseLandmarksTensor: {
884
+ get(ctx?: FrameContext): TensorData;
885
+ readonly value: TensorData;
886
+ };
887
+ readonly objectsTensor: {
888
+ get(ctx?: FrameContext): TensorData;
889
+ readonly value: TensorData;
890
+ };
891
+ readonly masksTensor: {
892
+ get(ctx?: FrameContext): TensorData;
893
+ readonly value: TensorData;
894
+ };
895
+ /** Per-class detection histogram [nc] — top-level convenience mirror of `classes.histogram`. */
896
+ readonly histogramTensor: {
897
+ get(ctx?: FrameContext): TensorData;
898
+ readonly value: TensorData;
899
+ };
900
+ private _totalFrames;
901
+ private _fps;
902
+ private _width;
903
+ private _height;
904
+ private _transformer;
905
+ private _stencilTexture?;
906
+ private _poseCache;
907
+ private _objectCache;
908
+ private _maskCache;
909
+ private _matteCache;
910
+ private _poseLmCache;
911
+ private _trackCache;
912
+ private _categoryCache;
913
+ private _classCache;
914
+ private _maskTrackCache;
915
+ private _registeredSignals;
916
+ constructor(options?: VisionBundleOptions);
917
+ setStencilTexture(texture: GPUTexture): void;
918
+ get stencilTexture(): GPUTexture | undefined;
919
+ /**
920
+ * Reads the most recent result at or before `frame`. Vision results are written one
921
+ * frame behind the plate (the node renderer reads back the previous frame to infer),
922
+ * and pinned-signal layout runs before the current frame's inference, so a strict
923
+ * per-frame lookup would always miss. Walking back a few frames keeps signals live at
924
+ * a stable one-frame lag instead of snapping to neutral defaults.
925
+ */
926
+ private latestAtOrBefore;
927
+ invalidateSignals(): void;
928
+ setPoseResult(frame: number, result: PoseResult): void;
929
+ getPoseResult(frame: number): PoseResult;
930
+ setObjectResult(frame: number, result: {
931
+ objects: readonly TrackedObject[];
932
+ rawDetections?: readonly DetectedObject[];
933
+ } | readonly TrackedObject[]): void;
934
+ getObjectResult(frame: number): {
935
+ objects: readonly TrackedObject[];
936
+ rawDetections: readonly DetectedObject[];
937
+ };
938
+ setMaskResult(frame: number, masks: readonly InstanceMask[]): void;
939
+ getMaskResult(frame: number): readonly InstanceMask[];
940
+ setMatteResult(frame: number, matte: PersonMatte): void;
941
+ getMatteResult(frame: number): PersonMatte | undefined;
942
+ /**
943
+ * Serializable snapshot of one frame's object / class / mask signals.
944
+ * Synchronous — reads only the per-frame caches, never triggers inference.
945
+ */
946
+ summary(frame?: number): VisionSummary;
947
+ /** Primary person's 17 keypoints as a [17, 3] float32 tensor. */
948
+ getPoseLandmarksTensor(frame: number): TensorData;
949
+ /** Up to 16 active tracks as [16, 8]: [active, categoryHash, x, y, w, h, vx, vy]. */
950
+ getObjectsTensor(frame: number, maxObjects?: number): TensorData;
951
+ /** Per-mask coverage + solidity as [16, 2]. */
952
+ getMasksTensor(frame: number, maxMasks?: number): TensorData;
953
+ /** Per-class detection histogram for the frame as [nc] float32 (COCO 80 by default). */
954
+ getClassHistogramTensor(frame: number, nc?: number): TensorData;
955
+ private clamp;
956
+ }
957
+ declare function createVisionBundle(options?: VisionBundleOptions): VisionBundle;
958
+ //#endregion
959
+ //#region src/spatial/camera-space-transformer.d.ts
960
+ interface Camera3DSpec {
961
+ readonly width: number;
962
+ readonly height: number;
963
+ readonly fov?: number;
964
+ readonly cameraDistance?: number;
965
+ }
966
+ interface ProjectedScreenCoordinate {
967
+ readonly x: number;
968
+ readonly y: number;
969
+ readonly z: number;
970
+ readonly scale: number;
971
+ }
972
+ declare class SpatialLandmarkTransformer {
973
+ private width;
974
+ private height;
975
+ private fovRad;
976
+ private focalDistance;
977
+ constructor(camera: Camera3DSpec);
978
+ /**
979
+ * Projects a normalized landmark [0, 1] into 2D/3D canvas coordinates.
980
+ */
981
+ projectNormalizedLandmark(landmark: Landmark3D, offsetZ?: number): ProjectedScreenCoordinate;
982
+ /**
983
+ * Projects a metric world landmark [X_w, Y_w, Z_w] in meters into canvas space.
984
+ */
985
+ projectWorldLandmark(worldLandmark: Landmark3D, subjectDistanceMeters?: number, scaleMetersToPixels?: number): ProjectedScreenCoordinate;
986
+ }
987
+ //#endregion
988
+ //#region src/spatial/spatial-pin.d.ts
989
+ declare function pinNodeToLandmark<T extends {
990
+ x?: unknown;
991
+ y?: unknown;
992
+ z?: unknown;
993
+ [key: string]: unknown;
994
+ }>(node: T, target: LandmarkCoordinateSignals | {
995
+ x: number;
996
+ y: number;
997
+ z?: number;
998
+ screenX?: number;
999
+ screenY?: number;
1000
+ }, options?: PinToLandmarkOptions): T;
1001
+ declare function pinNodeToObject<T extends {
1002
+ x?: unknown;
1003
+ y?: unknown;
1004
+ z?: unknown;
1005
+ width?: unknown;
1006
+ height?: unknown;
1007
+ opacity?: unknown;
1008
+ [key: string]: unknown;
1009
+ }>(node: T, target: ObjectTrackSignals | LandmarkCoordinateSignals | {
1010
+ x: number;
1011
+ y: number;
1012
+ z?: number;
1013
+ screenX?: number;
1014
+ screenY?: number;
1015
+ }, options?: PinToObjectOptions): T;
1016
+ //#endregion
1017
+ //#region src/tracking/temporal-object-tracker.d.ts
1018
+ interface TemporalTrackerOptions {
1019
+ /** Minimum Intersection-over-Union to associate a detection with an existing track. Default: 0.25 */
1020
+ readonly iouThreshold?: number;
1021
+ /** Number of frames a lost track is coasted via velocity extrapolation before deletion. Default: 15 */
1022
+ readonly maxMissedFrames?: number;
1023
+ /** Minimum consecutive hits before a tentative track is confirmed active. Default: 1 */
1024
+ readonly minHits?: number;
1025
+ /** Position smoothing weight [0, 1] where 1.0 is instantaneous and 0.0 is fully damped. Default: 0.75 */
1026
+ readonly positionSmoothing?: number;
1027
+ /** Whether tracks remain marked active during coasting frames. Default: true */
1028
+ readonly activeDuringCoast?: boolean;
1029
+ }
1030
+ /** Axis-aligned box in any consistent unit (pixels or normalized). */
1031
+ type PixelBox = Pick<ObjectBoundingBox, "originX" | "originY" | "width" | "height">;
1032
+ declare function computeIoU(boxA: PixelBox, boxB: PixelBox): number;
1033
+ declare class TemporalObjectTracker {
1034
+ private _nextTrackId;
1035
+ private _tracks;
1036
+ private _iouThreshold;
1037
+ private _maxMissedFrames;
1038
+ private _minHits;
1039
+ private _positionSmoothing;
1040
+ private _activeDuringCoast;
1041
+ constructor(options?: TemporalTrackerOptions);
1042
+ /**
1043
+ * Updates the multi-object tracker with new detections for the current frame.
1044
+ *
1045
+ * @param detections Raw detections found on the current frame
1046
+ * @param frame Current frame index
1047
+ * @param fps Video framerate (used for physical velocity estimation)
1048
+ * @returns Stable, temporal list of TrackedObjects
1049
+ */
1050
+ update(detections: readonly DetectedObject[], _frame: number, fps?: number): TrackedObject[];
1051
+ /**
1052
+ * Tracks objects across a full sequence of frames with gap interpolation,
1053
+ * boundary extrapolation (ensuring frame 0 through end are tracked),
1054
+ * and temporal Gaussian smoothing.
1055
+ *
1056
+ * Guarantees zero flicker, zero drift, and frame-accurate stability across the entire video.
1057
+ */
1058
+ trackSequence(perFrameDetections: readonly (readonly DetectedObject[])[], totalFrames: number, fps?: number): TrackedObject[][];
1059
+ /**
1060
+ * Resets all internal track state.
1061
+ */
1062
+ reset(): void;
1063
+ }
1064
+ //#endregion
1065
+ //#region src/tracking/pose-track-matcher.d.ts
1066
+ declare function getPersonBoundingBox(person: PosePerson): PixelBox;
1067
+ /**
1068
+ * Matches detected pose people with active person tracks using greedy bipartite IoU.
1069
+ *
1070
+ * Annotates each person with the matched trackId.
1071
+ */
1072
+ declare function matchPoseToTracks(people: readonly PosePerson[], trackedObjects: readonly TrackedObject[], minIoU?: number): PosePerson[];
1073
+ //#endregion
1074
+ //#region src/vision-node.d.ts
1075
+ /**
1076
+ * Bundle returned by `VisionNode.attach()`: the reactive signals surface plus the lazy lifecycle
1077
+ * helpers (all zero-I/O until `runner()` really needs a model).
1078
+ */
1079
+ interface VisionAttachedBundle extends VisionBundle {
1080
+ readonly node: VisionNodeSpec;
1081
+ /** Warm the enabled tasks' models/sessions ahead of first use. */
1082
+ ready(): Promise<VisionRunner>;
1083
+ /** Lazily-created runner (downloads happen on first real inference). */
1084
+ runner(): VisionRunner;
1085
+ /** Release this node's runner/sessions. */
1086
+ close(): void;
1087
+ }
1088
+ /**
1089
+ * Vision node.
1090
+ *
1091
+ * LAZY BY CONSTRUCTION: creating a VisionNode performs ZERO I/O — no model downloads, no
1092
+ * sessions, no file probes. The first inference call (or an explicit `await vision.ready()`)
1093
+ * downloads the required models.
1094
+ */
1095
+ declare class VisionNode {
1096
+ readonly id: string;
1097
+ readonly kind: "vision";
1098
+ readonly source: VisionInputSource;
1099
+ readonly config: VisionConfig;
1100
+ readonly vision: VisionBundle;
1101
+ private _runner?;
1102
+ /** Runner construction options (modelsDir, provider, store, …) — injectable for tests/agents. */
1103
+ private readonly _runnerOptions;
1104
+ constructor(source: VisionInputSource, config?: VisionConfig, bundleOptions?: VisionBundleOptions, runnerOptions?: VisionRunnerOptions);
1105
+ /** Lazily-created runner; downloads happen on its first inference. */
1106
+ runner(): VisionRunner;
1107
+ /** Explicit warm-up — downloads the models for the enabled tasks, ahead of first inference. */
1108
+ ready(): Promise<VisionRunner>;
1109
+ close(): void;
1110
+ static attach(source: VisionInputSource, config?: VisionConfig, bundleOptions?: VisionBundleOptions): VisionAttachedBundle;
1111
+ toNode(): VisionNodeSpec;
1112
+ }
1113
+ declare namespace index_d_exports {
1114
+ export { AnalyzeSequenceOptions, COCO17_BONES, COCO17_KEYPOINTS, COCO17_KEYPOINT_COUNT, COCO17_KEYPOINT_NAMES, COCO_CLASSES, Camera3DSpec, ClassCollectionSignals, ClassSignals, CocoClass, InputTransform, InstanceMask, LandmarkCoordinateSignals, MaskBounds, MaskCollectionSignals, MaskTrackSignals, ModelDownloadStatus, NodeSessionProvider, ObjectAnchorName, ObjectAnchorsSignals, ObjectBoundingBoxSignals, ObjectCollectionSignals, ObjectKinematicsSignals, ObjectTrackSignals, PREPROCESS_BY_FAMILY, PersonMatte, PersonMatteSignals, PinToLandmarkOptions, PinToObjectOptions, PixelBox, PoseKeypoint, PoseLandmarkSignals, PosePerson, PoseResult, PoseSkeletonOptions, PoseSkeletonRenderer, PreprocessSpec, ProjectedScreenCoordinate, RgbaImage, RtmdetInsDecodeOptions, RtmdetInsOutputs, RtmoDecodeOptions, RtmoOutputs, SegmentationResult, SegmentationSignals, SegmentationTextureOptions, SegmentationTexturePool, SelfieDecodeOptions, SessionProvider, SkeletonBone, SpatialLandmarkTransformer, TemporalObjectTracker, TemporalTrackerOptions, TrackPoseSignals, VISION_MODELS, VISION_TASKS, VISION_VARIANTS, VisionAnalysisClassSummary, VisionAnalysisReport, VisionAnalysisTrackSummary, VisionAttachedBundle, VisionBundle, VisionBundleOptions, VisionImageInput, VisionInputSource, VisionModelDescriptor, VisionModelFamily, VisionModelKey, VisionModelStore, VisionModelStoreOptions, VisionNode, VisionRunner, VisionRunnerOptions, VisionSession, VisionSummary, VisionTask, VisionTensor, VisionTensorInput, VisionTensorOutputs, VisionVariant, analyzeSequence, computeInputTransform, computeIoU, createDefaultSessionProvider, createNeutralLandmarks, createNeutralObjectResult, createNeutralPoseResult, createVisionBundle, decodeRtmdetIns, decodeRtmo, decodeSelfie, getDefaultModelsDir, getPersonBoundingBox, imageToTensor, maskBounds, matchPoseToTracks, mergeSubjectMask, modelKeyFor, pinNodeToLandmark, pinNodeToObject, preprocessSpec, probabilityToAlpha, selectSubjectMask, setDefaultSessionProvider, toSourcePoint };
1115
+ }
1116
+ //#endregion
1117
+ export { RtmdetInsOutputs as $, createVisionBundle as A, NodeSessionProvider as At, COCO17_KEYPOINT_NAMES as B, VisionModelStoreOptions as Bt, ObjectTrackSignals as C, SegmentationResult as Ct, TrackPoseSignals as D, VisionImageInput as Dt, SegmentationSignals as E, VisionAnalysisTrackSummary as Et, createNeutralLandmarks as F, VisionTensorOutputs as Ft, SkeletonBone as G, VISION_TASKS as Gt, SegmentationTexturePool as H, COCO_CLASSES as Ht, createNeutralObjectResult as I, createDefaultSessionProvider as It, COCO17_KEYPOINT_COUNT as J, VisionModelFamily as Jt, SelfieDecodeOptions as K, VISION_VARIANTS as Kt, createNeutralPoseResult as L, setDefaultSessionProvider as Lt, maskBounds as M, VisionSession as Mt, mergeSubjectMask as N, VisionTensor as Nt, VisionBundle as O, VisionInputSource as Ot, selectSubjectMask as P, VisionTensorInput as Pt, RtmdetInsDecodeOptions as Q, modelKeyFor as Qt, COCO17_BONES as R, ModelDownloadStatus as Rt, ObjectKinematicsSignals as S, PoseResult as St, PoseLandmarkSignals as T, VisionAnalysisReport as Tt, PoseSkeletonOptions as U, CocoClass as Ut, SegmentationTextureOptions as V, getDefaultModelsDir as Vt, PoseSkeletonRenderer as W, VISION_MODELS as Wt, RtmoOutputs as X, VisionTask as Xt, RtmoDecodeOptions as Y, VisionModelKey as Yt, decodeRtmo as Z, VisionVariant as Zt, MaskCollectionSignals as _, PersonMatte as _t, matchPoseToTracks as a, RgbaImage as at, ObjectBoundingBoxSignals as b, PoseKeypoint as bt, TemporalTrackerOptions as c, preprocessSpec as ct, pinNodeToObject as d, analyzeSequence as dt, decodeRtmdetIns as et, Camera3DSpec as f, VisionRunner as ft, ClassSignals as g, ObjectAnchorName as gt, ClassCollectionSignals as h, LandmarkCoordinateSignals as ht, getPersonBoundingBox as i, PreprocessSpec as it, MaskBounds as j, SessionProvider as jt, VisionBundleOptions as k, VisionSummary as kt, computeIoU as l, toSourcePoint as lt, SpatialLandmarkTransformer as m, InstanceMask as mt, VisionAttachedBundle as n, InputTransform as nt, PixelBox as o, computeInputTransform as ot, ProjectedScreenCoordinate as p, VisionRunnerOptions as pt, decodeSelfie as q, VisionModelDescriptor as qt, VisionNode as r, PREPROCESS_BY_FAMILY as rt, TemporalObjectTracker as s, imageToTensor as st, index_d_exports as t, probabilityToAlpha as tt, pinNodeToLandmark as u, AnalyzeSequenceOptions as ut, MaskTrackSignals as v, PinToLandmarkOptions as vt, PersonMatteSignals as w, VisionAnalysisClassSummary as wt, ObjectCollectionSignals as x, PosePerson as xt, ObjectAnchorsSignals as y, PinToObjectOptions as yt, COCO17_KEYPOINTS as z, VisionModelStore as zt };
1118
+ //# sourceMappingURL=index-DwGxgTRf.d.mts.map