gitframes 1.2.35 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,7 +4,8 @@ import { t as AudioSignal } from "./audio-CiXUMGyk.mjs";
4
4
  import { z } from "zod";
5
5
  import { existsSync, mkdirSync, readFileSync, statSync } from "node:fs";
6
6
  import { dirname, join, resolve } from "node:path";
7
- import { readFile, rename, unlink, writeFile } from "node:fs/promises";
7
+ import { rename, unlink, writeFile } from "node:fs/promises";
8
+ import { createHash } from "node:crypto";
8
9
  import { homedir } from "node:os";
9
10
 
10
11
  //#region ../tensor-webgpu/dist/index.mjs
@@ -2225,219 +2226,106 @@ var TensorPipeline = class TensorPipeline$1 {
2225
2226
  };
2226
2227
 
2227
2228
  //#endregion
2228
- //#region ../yolo/dist/src-_tHNcXK0.mjs
2229
- /**
2230
- * Default download origin. Override per-run with `baseUrl` option or the
2231
- * GITFRAMES_YOLO_BASE_URL environment variable.
2232
- */
2233
- const YOLO_ASSETS_RELEASE = "v8.3.0";
2234
- const DEFAULT_YOLO_BASE_URL = `https://github.com/ultralytics/assets/releases/download/${YOLO_ASSETS_RELEASE}/`;
2235
- const YOLO_MODELS = {
2236
- yolo11n: {
2237
- key: "yolo11n",
2238
- task: "detect",
2239
- variant: "n",
2240
- filename: "yolo11n.onnx",
2241
- bytesFp32: 10723904,
2242
- bytesFp16: 5361952,
2243
- nc: 80,
2244
- imgsz: 640
2245
- },
2246
- yolo11s: {
2247
- key: "yolo11s",
2248
- task: "detect",
2249
- variant: "s",
2250
- filename: "yolo11s.onnx",
2251
- url: "https://huggingface.co/mobilint/YOLO11s/resolve/main/yolo11s.onnx",
2252
- bytesFp32: 38502400,
2253
- bytesFp16: 19251200,
2254
- nc: 80,
2255
- imgsz: 640
2256
- },
2257
- yolo11m: {
2258
- key: "yolo11m",
2259
- task: "detect",
2260
- variant: "m",
2261
- filename: "yolo11m.onnx",
2262
- bytesFp32: 82329600,
2263
- bytesFp16: 41164800,
2264
- nc: 80,
2265
- imgsz: 640
2266
- },
2267
- yolo11l: {
2268
- key: "yolo11l",
2269
- task: "detect",
2270
- variant: "l",
2271
- filename: "yolo11l.onnx",
2272
- bytesFp32: 103628800,
2273
- bytesFp16: 51814400,
2274
- nc: 80,
2275
- imgsz: 640
2276
- },
2277
- yolo11x: {
2278
- key: "yolo11x",
2279
- task: "detect",
2280
- variant: "x",
2281
- filename: "yolo11x.onnx",
2282
- bytesFp32: 233062400,
2283
- bytesFp16: 116531200,
2284
- nc: 80,
2285
- imgsz: 640
2286
- },
2287
- "yolo11n-seg": {
2288
- key: "yolo11n-seg",
2289
- task: "segment",
2290
- variant: "n",
2291
- filename: "yolo11n-seg.onnx",
2292
- bytesFp32: 11886592,
2293
- nc: 80,
2294
- imgsz: 640
2295
- },
2296
- "yolo11s-seg": {
2297
- key: "yolo11s-seg",
2298
- task: "segment",
2299
- variant: "s",
2300
- filename: "yolo11s-seg.onnx",
2301
- url: "https://huggingface.co/mobilint/YOLO11s-seg/resolve/main/yolo11s-seg.onnx",
2302
- bytesFp32: 42631168,
2303
- nc: 80,
2304
- imgsz: 640
2229
+ //#region ../vision/dist/registry-CSFimt7C.mjs
2230
+ const hf = (repo, revision, path$1) => `https://huggingface.co/${repo}/resolve/${revision}/${path$1}`;
2231
+ const RTMDET_INS_REPO = "tori29umai/rtmdet-ins-onnx-person-masks";
2232
+ const RTMDET_INS_REV = "e5218e8d47d7c50077f27f9893a28e0ef2ad5656";
2233
+ const RTMDET_INS_SOURCE = "RTMDet-Ins (OpenMMLab mmdetection, Apache-2.0), mmdeploy ONNX export";
2234
+ const RTMO_SOURCE = "RTMO (OpenMMLab mmpose, Apache-2.0), ONNX export";
2235
+ const SELFIE_SOURCE = "MediaPipe Selfie Segmenter (Google, Apache-2.0), ONNX export";
2236
+ const VISION_MODELS = {
2237
+ "rtmdet-ins-t": {
2238
+ key: "rtmdet-ins-t",
2239
+ family: "rtmdet-ins",
2240
+ filename: "rtmdet-ins-t.onnx",
2241
+ url: hf(RTMDET_INS_REPO, RTMDET_INS_REV, "models/tiny/rtmdet-ins_tiny_640x640.onnx"),
2242
+ bytes: 24032356,
2243
+ sha256: "99ee670820c4493aa13e20578b6a8e39781c319846dc41796a77765734beac5c",
2244
+ input: [640, 640],
2245
+ source: RTMDET_INS_SOURCE
2305
2246
  },
2306
- "yolo11m-seg": {
2307
- key: "yolo11m-seg",
2308
- task: "segment",
2309
- variant: "m",
2310
- filename: "yolo11m-seg.onnx",
2311
- bytesFp32: 91128832,
2312
- nc: 80,
2313
- imgsz: 640
2247
+ "rtmdet-ins-s": {
2248
+ key: "rtmdet-ins-s",
2249
+ family: "rtmdet-ins",
2250
+ filename: "rtmdet-ins-s.onnx",
2251
+ url: hf(RTMDET_INS_REPO, RTMDET_INS_REV, "models/s/rtmdet-ins_s_640x640.onnx"),
2252
+ bytes: 43236976,
2253
+ sha256: "9cd1787fbf3eb2bd64cc2d8154f41c036f5484b85599e1cbe0227a2dc5a4a08b",
2254
+ input: [640, 640],
2255
+ source: RTMDET_INS_SOURCE
2314
2256
  },
2315
- "yolo11n-pose": {
2316
- key: "yolo11n-pose",
2317
- task: "pose",
2318
- variant: "n",
2319
- filename: "yolo11n-pose.onnx",
2320
- bytesFp32: 11450368,
2321
- nc: 1,
2322
- imgsz: 640
2257
+ "rtmdet-ins-m": {
2258
+ key: "rtmdet-ins-m",
2259
+ family: "rtmdet-ins",
2260
+ filename: "rtmdet-ins-m.onnx",
2261
+ url: hf(RTMDET_INS_REPO, RTMDET_INS_REV, "models/m/rtmdet-ins_m_640x640.onnx"),
2262
+ bytes: 115735979,
2263
+ sha256: "a003bce80316e03467c2c2b483df2d5c0398220c2357a44f9bf985929f7eb052",
2264
+ input: [640, 640],
2265
+ source: RTMDET_INS_SOURCE
2323
2266
  },
2324
- "yolo11s-pose": {
2325
- key: "yolo11s-pose",
2326
- task: "pose",
2327
- variant: "s",
2328
- filename: "yolo11s-pose.onnx",
2329
- url: "https://huggingface.co/mobilint/YOLO11s-pose/resolve/main/yolo11s-pose.onnx",
2330
- bytesFp32: 41050112,
2331
- nc: 1,
2332
- imgsz: 640
2267
+ "rtmo-t": {
2268
+ key: "rtmo-t",
2269
+ family: "rtmo",
2270
+ filename: "rtmo-t.onnx",
2271
+ url: hf("Xenova/RTMO-t", "f29d13ce5ee291fbe505e9d06a049992f05699dc", "onnx/model.onnx"),
2272
+ bytes: 27360569,
2273
+ sha256: "cb54dd042bf997df86d938721c94238940f2bff98ba8d2ccc834430de5c65c13",
2274
+ input: [416, 416],
2275
+ source: RTMO_SOURCE
2333
2276
  },
2334
- "yolo11m-pose": {
2335
- key: "yolo11m-pose",
2336
- task: "pose",
2337
- variant: "m",
2338
- filename: "yolo11m-pose.onnx",
2339
- url: "https://huggingface.co/mobilint/YOLO11m-pose/resolve/main/yolo11m-pose.onnx",
2340
- bytesFp32: 821e5,
2341
- nc: 1,
2342
- imgsz: 640
2277
+ "rtmo-s": {
2278
+ key: "rtmo-s",
2279
+ family: "rtmo",
2280
+ filename: "rtmo-s.onnx",
2281
+ url: hf("Xenova/RTMO-s", "d8c526187f341d287753831c9c8b1ecc4855bba1", "onnx/model.onnx"),
2282
+ bytes: 39636400,
2283
+ sha256: "1cd6a3517658903233e9ccb78666554862f57f0944b6886e2048093594aef4e7",
2284
+ input: [640, 640],
2285
+ source: RTMO_SOURCE
2343
2286
  },
2344
- "yolo11l-pose": {
2345
- key: "yolo11l-pose",
2346
- task: "pose",
2347
- variant: "l",
2348
- filename: "yolo11l-pose.onnx",
2349
- url: "https://huggingface.co/mobilint/YOLO11l-pose/resolve/main/yolo11l-pose.onnx",
2350
- bytesFp32: 1105e5,
2351
- nc: 1,
2352
- imgsz: 640
2287
+ "rtmo-m": {
2288
+ key: "rtmo-m",
2289
+ family: "rtmo",
2290
+ filename: "rtmo-m.onnx",
2291
+ url: hf("Xenova/RTMO-m", "3aba1280472b98ee2bf663482e27e243038d23fe", "onnx/model.onnx"),
2292
+ bytes: 89291929,
2293
+ sha256: "76d82c45e5c4810baf587ecf2638d15cbf7ed196afb2055e977382532021b59e",
2294
+ input: [640, 640],
2295
+ source: RTMO_SOURCE
2353
2296
  },
2354
- "yolo11n-obb": {
2355
- key: "yolo11n-obb",
2356
- task: "obb",
2357
- variant: "n",
2358
- filename: "yolo11n-obb.onnx",
2359
- bytesFp32: 12048384,
2360
- nc: 15,
2361
- imgsz: 1024
2297
+ "selfie-square": {
2298
+ key: "selfie-square",
2299
+ family: "selfie",
2300
+ filename: "selfie-square.onnx",
2301
+ url: hf("onnx-community/mediapipe_selfie_segmentation", "be49485c8e027524be38591817fc5cd31bd9d00e", "onnx/model.onnx"),
2302
+ bytes: 462352,
2303
+ sha256: "3241ac4ad8aa35bdaf33946776db29f7c283a413aa0b0dacb9483594b4531aad",
2304
+ input: [256, 256],
2305
+ source: SELFIE_SOURCE
2362
2306
  },
2363
- "yolo11s-obb": {
2364
- key: "yolo11s-obb",
2365
- task: "obb",
2366
- variant: "s",
2367
- filename: "yolo11s-obb.onnx",
2368
- bytesFp32: 43253760,
2369
- nc: 15,
2370
- imgsz: 1024
2371
- },
2372
- "yolo11m-obb": {
2373
- key: "yolo11m-obb",
2374
- task: "obb",
2375
- variant: "m",
2376
- filename: "yolo11m-obb.onnx",
2377
- url: "https://huggingface.co/mobilint/YOLO11m-obb/resolve/main/yolo11m-obb.onnx",
2378
- bytesFp32: 832e5,
2379
- nc: 15,
2380
- imgsz: 1024
2381
- },
2382
- "yolo11l-obb": {
2383
- key: "yolo11l-obb",
2384
- task: "obb",
2385
- variant: "l",
2386
- filename: "yolo11l-obb.onnx",
2387
- url: "https://huggingface.co/mobilint/YOLO11l-obb/resolve/main/yolo11l-obb.onnx",
2388
- bytesFp32: 111e6,
2389
- nc: 15,
2390
- imgsz: 1024
2391
- },
2392
- "yolo11n-cls": {
2393
- key: "yolo11n-cls",
2394
- task: "classify",
2395
- variant: "n",
2396
- filename: "yolo11n-cls.onnx",
2397
- bytesFp32: 11918848,
2398
- nc: 1e3,
2399
- imgsz: 224
2400
- },
2401
- "yolo11s-cls": {
2402
- key: "yolo11s-cls",
2403
- task: "classify",
2404
- variant: "s",
2405
- filename: "yolo11s-cls.onnx",
2406
- bytesFp32: 42704896,
2407
- nc: 1e3,
2408
- imgsz: 224
2409
- },
2410
- "yolov8s-worldv2": {
2411
- key: "yolov8s-worldv2",
2412
- task: "world",
2413
- variant: "s",
2414
- filename: "yolov8s-worldv2.onnx",
2415
- url: "https://huggingface.co/Instemic/yolo-world-onnx/resolve/main/yolov8s-worldv2.onnx",
2416
- bytesFp32: 52e6,
2417
- nc: 80,
2418
- imgsz: 640
2419
- },
2420
- "yolov8m-worldv2": {
2421
- key: "yolov8m-worldv2",
2422
- task: "world",
2423
- variant: "m",
2424
- filename: "yolov8m-worldv2.onnx",
2425
- url: "https://huggingface.co/ultralytics/yolov8/resolve/main/yolov8m-worldv2.onnx",
2426
- bytesFp32: 108e6,
2427
- nc: 80,
2428
- imgsz: 640
2307
+ "selfie-landscape": {
2308
+ key: "selfie-landscape",
2309
+ family: "selfie",
2310
+ filename: "selfie-landscape.onnx",
2311
+ url: hf("onnx-community/mediapipe_selfie_segmentation_landscape", "2497d5bec26c626c7b3c4edc6e1fefc21b64f6c3", "onnx/model.onnx"),
2312
+ bytes: 462338,
2313
+ sha256: "7a0adcfdb1715d3b0ff0f61486d8a181c33f4a208343a5abfccbc6540e872d24",
2314
+ input: [256, 144],
2315
+ source: SELFIE_SOURCE
2429
2316
  }
2430
2317
  };
2431
- const DEFAULT_MODEL_KEY_BY_TASK = {
2432
- detect: (v) => v === "n" ? "yolo11n" : v === "s" ? "yolo11s" : v === "m" ? "yolo11m" : v === "l" ? "yolo11l" : "yolo11x",
2433
- segment: (v) => v === "m" ? "yolo11m-seg" : v === "s" ? "yolo11s-seg" : "yolo11n-seg",
2434
- pose: (v) => v === "l" ? "yolo11l-pose" : v === "m" ? "yolo11m-pose" : v === "s" ? "yolo11s-pose" : "yolo11n-pose",
2435
- obb: (v) => v === "l" ? "yolo11l-obb" : v === "m" ? "yolo11m-obb" : v === "s" ? "yolo11s-obb" : "yolo11n-obb",
2436
- classify: (v) => v === "s" ? "yolo11s-cls" : "yolo11n-cls",
2437
- world: (v) => v === "m" || v === "l" || v === "x" ? "yolov8m-worldv2" : "yolov8s-worldv2"
2438
- };
2439
- /** COCO 80 class names — index order matches every ultralytics COCO export. */
2440
- const YOLO_COCO_CLASSES = [
2318
+ /**
2319
+ * Resolves the model for a task. `detect` and `segment` share RTMDet-Ins (one forward pass
2320
+ * yields both); `matte` picks the landscape Selfie Segmenter for wide frames.
2321
+ */
2322
+ function modelKeyFor(task, variant = "s", aspect = 1) {
2323
+ if (task === "pose") return `rtmo-${variant}`;
2324
+ if (task === "matte") return aspect > 1.3 ? "selfie-landscape" : "selfie-square";
2325
+ return `rtmdet-ins-${variant}`;
2326
+ }
2327
+ /** COCO 80 class names — index order matches the RTMDet-Ins `labels` output. */
2328
+ const COCO_CLASSES = [
2441
2329
  "person",
2442
2330
  "bicycle",
2443
2331
  "car",
@@ -2519,25 +2407,10 @@ const YOLO_COCO_CLASSES = [
2519
2407
  "hair drier",
2520
2408
  "toothbrush"
2521
2409
  ];
2522
- /** DOTA-v1.0 15 class names — index order matches ultralytics YOLO11-OBB export. */
2523
- const YOLO_DOTA_CLASSES = [
2524
- "plane",
2525
- "ship",
2526
- "storage tank",
2527
- "baseball diamond",
2528
- "tennis court",
2529
- "basketball court",
2530
- "ground track field",
2531
- "harbor",
2532
- "bridge",
2533
- "large vehicle",
2534
- "small vehicle",
2535
- "helicopter",
2536
- "roundabout",
2537
- "soccer ball field",
2538
- "swimming pool"
2539
- ];
2540
- function computeIoU$1(boxA, boxB) {
2410
+
2411
+ //#endregion
2412
+ //#region ../vision/dist/src-Dv7EKzQ2.mjs
2413
+ function computeIoU(boxA, boxB) {
2541
2414
  const xA = Math.max(boxA.originX, boxB.originX);
2542
2415
  const yA = Math.max(boxA.originY, boxB.originY);
2543
2416
  const xB = Math.min(boxA.originX + boxA.width, boxB.originX + boxB.width);
@@ -2609,7 +2482,7 @@ var TemporalObjectTracker = class {
2609
2482
  for (let d = 0; d < detections.length; d++) {
2610
2483
  const det = detections[d];
2611
2484
  if (trk.category !== det.category) continue;
2612
- const iou = computeIoU$1(trk.box, det.boundingBox);
2485
+ const iou = computeIoU(trk.box, det.boundingBox);
2613
2486
  if (iou >= this._iouThreshold) candidateMatches.push({
2614
2487
  trackIdx: t,
2615
2488
  detIdx: d,
@@ -2888,22 +2761,22 @@ async function analyzeSequence(frames, options) {
2888
2761
  const taskSet = new Set(tasks);
2889
2762
  const categoryFilter = options.categories ? new Set(options.categories.map((c) => c.toLowerCase())) : null;
2890
2763
  const includeMasks = options.includeMasks === true && taskSet.has("segment");
2891
- const detectionTask = taskSet.has("detect") ? "detect" : taskSet.has("segment") ? "segment" : taskSet.has("obb") ? "obb" : null;
2764
+ const tracksObjects = taskSet.has("detect") || taskSet.has("segment");
2892
2765
  const perFrameDetections = [];
2893
2766
  const modelDownloads = {};
2894
2767
  const maskCoverage = /* @__PURE__ */ new Map();
2895
2768
  let inferenceMs = 0;
2896
2769
  let frameIndex = 0;
2897
2770
  const accepts = (category) => !categoryFilter || categoryFilter.has(category.toLowerCase());
2898
- const runTask = async (task, fn) => {
2899
- const key = runner.modelKeyFor(task);
2771
+ const runTask = async (task, frame, fn) => {
2772
+ const key = runner.modelKeyFor(task, frame.width / frame.height);
2900
2773
  const wasReady = runner.downloadStatus.get(key) === "ready";
2901
2774
  const start = now();
2902
2775
  const result = await fn();
2903
2776
  const elapsed = now() - start;
2904
2777
  inferenceMs += elapsed;
2905
2778
  if (!wasReady && !modelDownloads[key]) modelDownloads[key] = {
2906
- bytes: YOLO_MODELS[key].bytesFp32,
2779
+ bytes: VISION_MODELS[key].bytes,
2907
2780
  ms: elapsed
2908
2781
  };
2909
2782
  return result;
@@ -2922,19 +2795,13 @@ async function analyzeSequence(frames, options) {
2922
2795
  };
2923
2796
  for await (const frame of frames) {
2924
2797
  let detections = [];
2925
- if (includeMasks || detectionTask === "segment") {
2926
- const seg = await runTask("segment", () => runner.segment(frame));
2927
- if (includeMasks) recordMasks(seg.masks);
2928
- if (detectionTask === "segment") detections = seg.detections;
2929
- }
2930
- if (detectionTask === "detect") detections = await runTask("detect", () => runner.detect(frame));
2931
- else if (detectionTask === "obb") detections = (await runTask("obb", () => runner.detectObb(frame))).detections.map((d) => ({
2932
- category: d.category,
2933
- score: d.score,
2934
- boundingBox: d.boundingBox
2935
- }));
2936
- if (taskSet.has("pose")) await runTask("pose", () => runner.pose(frame));
2937
- if (taskSet.has("classify")) await runTask("classify", () => runner.classify(frame));
2798
+ if (includeMasks) {
2799
+ const seg = await runTask("segment", frame, () => runner.segment(frame));
2800
+ recordMasks(seg.masks);
2801
+ detections = seg.detections;
2802
+ } else if (tracksObjects) detections = await runTask("detect", frame, () => runner.detect(frame));
2803
+ if (taskSet.has("pose")) await runTask("pose", frame, () => runner.pose(frame));
2804
+ if (taskSet.has("matte")) await runTask("matte", frame, () => runner.matte(frame));
2938
2805
  perFrameDetections.push(categoryFilter ? detections.filter((d) => accepts(d.category)) : detections);
2939
2806
  frameIndex++;
2940
2807
  }
@@ -3014,463 +2881,374 @@ async function analyzeSequence(frames, options) {
3014
2881
  masks
3015
2882
  };
3016
2883
  }
3017
- /** Classification head: [1, nc] logits → softmax → top-5. */
3018
- function decodeClassifyOutput(output, nc) {
3019
- const count = Math.min(nc, output.length);
3020
- let max = -Infinity;
3021
- for (let i = 0; i < count; i++) if (output[i] > max) max = output[i];
3022
- let sum = 0;
3023
- const probs = new Float32Array(count);
3024
- for (let i = 0; i < count; i++) {
3025
- const p = Math.exp(output[i] - max);
3026
- probs[i] = p;
3027
- sum += p;
3028
- }
3029
- const inv = sum > 0 ? 1 / sum : 0;
3030
- for (let i = 0; i < count; i++) probs[i] *= inv;
3031
- const top5 = [];
3032
- for (let i = 0; i < Math.min(5, probs.length); i++) {
3033
- let best = -1;
3034
- let bestScore = -1;
3035
- for (let c = 0; c < probs.length; c++) if (probs[c] > bestScore) {
3036
- bestScore = probs[c];
3037
- best = c;
3038
- }
3039
- if (best < 0) break;
3040
- top5.push({
3041
- index: best,
3042
- score: bestScore
3043
- });
3044
- probs[best] = -1;
2884
+ const IMAGENET_BGR_MEAN = [
2885
+ 103.53,
2886
+ 116.28,
2887
+ 123.675
2888
+ ];
2889
+ const IMAGENET_BGR_STD = [
2890
+ 57.375,
2891
+ 57.12,
2892
+ 58.395
2893
+ ];
2894
+ /** Per-family normalization; the input size always comes from the model's registry entry. */
2895
+ const PREPROCESS_BY_FAMILY = {
2896
+ "rtmdet-ins": {
2897
+ fit: "topLeft",
2898
+ channels: "bgr",
2899
+ mean: IMAGENET_BGR_MEAN,
2900
+ std: IMAGENET_BGR_STD,
2901
+ padValue: 114
2902
+ },
2903
+ rtmo: {
2904
+ fit: "center",
2905
+ channels: "bgr",
2906
+ mean: [
2907
+ 0,
2908
+ 0,
2909
+ 0
2910
+ ],
2911
+ std: [
2912
+ 1,
2913
+ 1,
2914
+ 1
2915
+ ],
2916
+ padValue: 114
2917
+ },
2918
+ selfie: {
2919
+ fit: "stretch",
2920
+ channels: "rgb",
2921
+ mean: [
2922
+ 0,
2923
+ 0,
2924
+ 0
2925
+ ],
2926
+ std: [
2927
+ 255,
2928
+ 255,
2929
+ 255
2930
+ ],
2931
+ padValue: 0
3045
2932
  }
2933
+ };
2934
+ function preprocessSpec(family, input) {
3046
2935
  return {
3047
- top1: top5[0]?.index ?? 0,
3048
- top1Score: top5[0]?.score ?? 0,
3049
- top5
2936
+ ...PREPROCESS_BY_FAMILY[family],
2937
+ width: input[0],
2938
+ height: input[1]
3050
2939
  };
3051
2940
  }
3052
- function computeIoU(a, b) {
3053
- const interX0 = Math.max(a.x0, b.x0);
3054
- const interY0 = Math.max(a.y0, b.y0);
3055
- const interX1 = Math.min(a.x1, b.x1);
3056
- const interY1 = Math.min(a.y1, b.y1);
3057
- const interArea = Math.max(0, interX1 - interX0) * Math.max(0, interY1 - interY0);
3058
- if (interArea <= 0) return 0;
3059
- const unionArea = (a.x1 - a.x0) * (a.y1 - a.y0) + (b.x1 - b.x0) * (b.y1 - b.y0) - interArea;
3060
- return unionArea > 0 ? interArea / unionArea : 0;
3061
- }
3062
- /**
3063
- * Greedy class-aware NMS: boxes of different classes never suppress each other
3064
- * (matches ultralytics default behavior).
3065
- */
3066
- function nms(boxes, iouThreshold = .45) {
3067
- const sorted = [...boxes].sort((a, b) => b.score - a.score);
3068
- const kept = [];
3069
- for (const candidate of sorted) {
3070
- let suppressed = false;
3071
- for (const k of kept) {
3072
- if (k.classIndex !== candidate.classIndex) continue;
3073
- if (computeIoU(k, candidate) > iouThreshold) {
3074
- suppressed = true;
3075
- break;
3076
- }
3077
- }
3078
- if (!suppressed) kept.push(candidate);
3079
- }
3080
- return kept;
3081
- }
3082
- function decodeYoloBoxes(output, anchors, nc, opts) {
3083
- const candidates = [];
3084
- const classFilter = opts.classes ? new Set(opts.classes.map((c) => c.toLowerCase())) : null;
3085
- for (let a = 0; a < anchors; a++) {
3086
- let bestScore = -1;
3087
- let bestClass = -1;
3088
- for (let c = 0; c < nc; c++) {
3089
- const score = output[(4 + c) * anchors + a];
3090
- if (score > bestScore) {
3091
- bestScore = score;
3092
- bestClass = c;
3093
- }
3094
- }
3095
- if (bestScore < opts.confidence) continue;
3096
- if (classFilter && !classFilter.has((opts.classNames[bestClass] ?? "").toLowerCase())) continue;
3097
- const cx = output[a];
3098
- const cy = output[anchors + a];
3099
- const w = output[2 * anchors + a];
3100
- const h = output[3 * anchors + a];
3101
- if (w <= 0 || h <= 0) continue;
3102
- const halfW = w / 2;
3103
- const halfH = h / 2;
3104
- candidates.push({
3105
- x0: cx - halfW,
3106
- y0: cy - halfH,
3107
- x1: cx + halfW,
3108
- y1: cy + halfH,
3109
- score: bestScore,
3110
- classIndex: bestClass,
3111
- anchorIndex: a
3112
- });
3113
- }
3114
- return nms(candidates, opts.iouThreshold);
3115
- }
3116
- /** Maps an NMS-kept box from input space to a core `DetectedObject` in source pixel space. */
3117
- function toDetectedObject(box, opts) {
3118
- const { scale, dw, dh } = opts.params;
3119
- const { sourceWidth: sw, sourceHeight: sh } = opts;
3120
- const x0 = Math.max(0, Math.min(sw, (box.x0 - dw) / scale));
3121
- const y0 = Math.max(0, Math.min(sh, (box.y0 - dh) / scale));
3122
- const x1 = Math.max(0, Math.min(sw, (box.x1 - dw) / scale));
3123
- const y1 = Math.max(0, Math.min(sh, (box.y1 - dh) / scale));
3124
- const width = Math.max(0, x1 - x0);
3125
- const height = Math.max(0, y1 - y0);
2941
+ function computeInputTransform(srcW, srcH, spec) {
2942
+ if (spec.fit === "stretch") return {
2943
+ scaleX: spec.width / srcW,
2944
+ scaleY: spec.height / srcH,
2945
+ offsetX: 0,
2946
+ offsetY: 0,
2947
+ inputWidth: spec.width,
2948
+ inputHeight: spec.height
2949
+ };
2950
+ const scale = Math.min(spec.width / srcW, spec.height / srcH);
2951
+ const contentW = Math.round(srcW * scale);
2952
+ const contentH = Math.round(srcH * scale);
2953
+ const center = spec.fit === "center";
3126
2954
  return {
3127
- category: opts.classNames[box.classIndex] ?? `class_${box.classIndex}`,
3128
- score: box.score,
3129
- boundingBox: {
3130
- originX: x0,
3131
- originY: y0,
3132
- width,
3133
- height,
3134
- normalizedX: sw > 0 ? x0 / sw : 0,
3135
- normalizedY: sh > 0 ? y0 / sh : 0,
3136
- normalizedWidth: sw > 0 ? width / sw : 0,
3137
- normalizedHeight: sh > 0 ? height / sh : 0
3138
- }
2955
+ scaleX: scale,
2956
+ scaleY: scale,
2957
+ offsetX: center ? (spec.width - contentW) / 2 : 0,
2958
+ offsetY: center ? (spec.height - contentH) / 2 : 0,
2959
+ inputWidth: spec.width,
2960
+ inputHeight: spec.height
3139
2961
  };
3140
2962
  }
3141
- /** Full detect-head decode: hidden state → DetectedObject[] (already NMS'd, source pixels). */
3142
- function decodeDetectOutput(output, anchors, nc, opts) {
3143
- return decodeYoloBoxes(output, anchors, nc, opts).map((box) => toDetectedObject(box, opts));
3144
- }
3145
- /**
3146
- * Letterbox preprocessing + inverse mapping.
3147
- *
3148
- * The YOLO ONNX export consumes a square RGB float tensor (imgsz × imgsz), values / 255,
3149
- * gray (114) padding on the short dimension — the "letterbox". Box/keypoint outputs are
3150
- * in input-pixel space; this module owns the resize and the mapping back to source pixels.
3151
- */
3152
- const YOLO_LETTERBOX_PAD_VALUE = 114;
3153
- function computeLetterbox(srcW, srcH, imgsz) {
3154
- const scale = Math.min(imgsz / srcW, imgsz / srcH);
3155
- const inputWidth = Math.round(srcW * scale);
3156
- const inputHeight = Math.round(srcH * scale);
2963
+ /** Model-input point → source pixel point. */
2964
+ function toSourcePoint(x, y, t) {
3157
2965
  return {
3158
- scale,
3159
- dw: (imgsz - inputWidth) / 2,
3160
- dh: (imgsz - inputHeight) / 2,
3161
- inputWidth,
3162
- inputHeight,
3163
- size: imgsz
2966
+ x: (x - t.offsetX) / t.scaleX,
2967
+ y: (y - t.offsetY) / t.scaleY
3164
2968
  };
3165
2969
  }
3166
2970
  /**
3167
- * Resizes RGBA -> NCHW float32 tensor with bilinear sampling + gray padding.
3168
- * Layout: [1, 3, imgsz, imgsz], channel stride = imgsz * imgsz.
3169
- * Reuses `out` / `outData` when provided (hot-path frame loop).
2971
+ * Resizes RGBA → NCHW float32 `[1, 3, height, width]` with bilinear sampling, padding,
2972
+ * channel reordering and normalization per `spec`. Reuses `out` when large enough
2973
+ * (hot-path frame loop).
3170
2974
  */
3171
- function letterboxToTensor(image, imgsz = 640, out) {
3172
- const { width: srcW, height: srcH } = image;
3173
- const params = computeLetterbox(srcW, srcH, imgsz);
3174
- const { scale, dw, dh, inputWidth, inputHeight } = params;
3175
- const channelStride = imgsz * imgsz;
3176
- if (!out || out.length < 3 * channelStride) out = new Float32Array(3 * channelStride);
3177
- else out.fill(0, 0, 3 * channelStride);
3178
- const src = image.data;
3179
- const pad = YOLO_LETTERBOX_PAD_VALUE / 255;
3180
- const x0 = Math.ceil(dw);
3181
- const y0 = Math.ceil(dh);
3182
- const x1 = Math.floor(dw + inputWidth) - 1;
3183
- const y1 = Math.floor(dh + inputHeight) - 1;
3184
- const rPlane = out.subarray(0, channelStride);
3185
- const gPlane = out.subarray(channelStride, 2 * channelStride);
3186
- const bPlane = out.subarray(2 * channelStride, 3 * channelStride);
3187
- for (let y = 0; y < imgsz; y++) {
3188
- const rowBase = y * imgsz;
2975
+ function imageToTensor(image, spec, out) {
2976
+ const { width: srcW, height: srcH, data: src } = image;
2977
+ const { width: W, height: H } = spec;
2978
+ const transform = computeInputTransform(srcW, srcH, spec);
2979
+ const { scaleX, scaleY, offsetX, offsetY } = transform;
2980
+ const plane = W * H;
2981
+ if (!out || out.length < 3 * plane) out = new Float32Array(3 * plane);
2982
+ const order = spec.channels === "bgr" ? [
2983
+ 2,
2984
+ 1,
2985
+ 0
2986
+ ] : [
2987
+ 0,
2988
+ 1,
2989
+ 2
2990
+ ];
2991
+ const [m0, m1, m2] = spec.mean;
2992
+ const inv0 = 1 / spec.std[0];
2993
+ const inv1 = 1 / spec.std[1];
2994
+ const inv2 = 1 / spec.std[2];
2995
+ const pad0 = (spec.padValue - m0) * inv0;
2996
+ const pad1 = (spec.padValue - m1) * inv1;
2997
+ const pad2 = (spec.padValue - m2) * inv2;
2998
+ const [c0, c1, c2] = order;
2999
+ const x0 = Math.ceil(offsetX);
3000
+ const y0 = Math.ceil(offsetY);
3001
+ const x1 = Math.min(W, Math.floor(offsetX + srcW * scaleX)) - 1;
3002
+ const y1 = Math.min(H, Math.floor(offsetY + srcH * scaleY)) - 1;
3003
+ for (let y = 0; y < H; y++) {
3004
+ const rowBase = y * W;
3189
3005
  const paddedRow = y < y0 || y > y1;
3190
- for (let x = 0; x < imgsz; x++) {
3191
- let r = pad;
3192
- let g = pad;
3193
- let b = pad;
3194
- if (!paddedRow && x >= x0 && x <= x1) {
3195
- const sx = (x - dw) / scale;
3196
- const sy = (y - dh) / scale;
3197
- const x0s = Math.floor(sx);
3198
- const y0s = Math.floor(sy);
3199
- const fx = sx - x0s;
3200
- const fy = sy - y0s;
3201
- const x1s = Math.min(srcW - 1, x0s + 1);
3202
- const y1s = Math.min(srcH - 1, y0s + 1);
3203
- const i00 = (y0s * srcW + x0s) * 4;
3204
- const i10 = (y0s * srcW + x1s) * 4;
3205
- const i01 = (y1s * srcW + x0s) * 4;
3206
- const i11 = (y1s * srcW + x1s) * 4;
3207
- const w00 = (1 - fx) * (1 - fy);
3208
- const w10 = fx * (1 - fy);
3209
- const w01 = (1 - fx) * fy;
3210
- const w11 = fx * fy;
3211
- r = (src[i00] * w00 + src[i10] * w10 + src[i01] * w01 + src[i11] * w11) / 255;
3212
- g = (src[i00 + 1] * w00 + src[i10 + 1] * w10 + src[i01 + 1] * w01 + src[i11 + 1] * w11) / 255;
3213
- b = (src[i00 + 2] * w00 + src[i10 + 2] * w10 + src[i01 + 2] * w01 + src[i11 + 2] * w11) / 255;
3006
+ const sy = Math.max(0, (y + .5 - offsetY) / scaleY - .5);
3007
+ const iy0 = Math.min(srcH - 1, Math.floor(sy));
3008
+ const iy1 = Math.min(srcH - 1, iy0 + 1);
3009
+ const fy = sy - iy0;
3010
+ for (let x = 0; x < W; x++) {
3011
+ const o = rowBase + x;
3012
+ if (paddedRow || x < x0 || x > x1) {
3013
+ out[o] = pad0;
3014
+ out[plane + o] = pad1;
3015
+ out[2 * plane + o] = pad2;
3016
+ continue;
3214
3017
  }
3215
- rPlane[rowBase + x] = r;
3216
- gPlane[rowBase + x] = g;
3217
- bPlane[rowBase + x] = b;
3018
+ const sx = Math.max(0, (x + .5 - offsetX) / scaleX - .5);
3019
+ const ix0 = Math.min(srcW - 1, Math.floor(sx));
3020
+ const ix1 = Math.min(srcW - 1, ix0 + 1);
3021
+ const fx = sx - ix0;
3022
+ const i00 = (iy0 * srcW + ix0) * 4;
3023
+ const i10 = (iy0 * srcW + ix1) * 4;
3024
+ const i01 = (iy1 * srcW + ix0) * 4;
3025
+ const i11 = (iy1 * srcW + ix1) * 4;
3026
+ const w00 = (1 - fx) * (1 - fy);
3027
+ const w10 = fx * (1 - fy);
3028
+ const w01 = (1 - fx) * fy;
3029
+ const w11 = fx * fy;
3030
+ out[o] = (src[i00 + c0] * w00 + src[i10 + c0] * w10 + src[i01 + c0] * w01 + src[i11 + c0] * w11 - m0) * inv0;
3031
+ out[plane + o] = (src[i00 + c1] * w00 + src[i10 + c1] * w10 + src[i01 + c1] * w01 + src[i11 + c1] * w11 - m1) * inv1;
3032
+ out[2 * plane + o] = (src[i00 + c2] * w00 + src[i10 + c2] * w10 + src[i01 + c2] * w01 + src[i11 + c2] * w11 - m2) * inv2;
3218
3033
  }
3219
3034
  }
3220
3035
  return {
3221
3036
  tensor: out,
3222
- params
3223
- };
3224
- }
3225
- /** Projects the 4 rotated corners (input px) and the axis-aligned extent back to source px. */
3226
- function projectCorners(cx, cy, w, h, angle, opts) {
3227
- const { scale, dw, dh } = opts.params;
3228
- const cos = Math.cos(angle);
3229
- const sin = Math.sin(angle);
3230
- const hw = w / 2;
3231
- const hh = h / 2;
3232
- const local = [
3233
- [-hw, -hh],
3234
- [hw, -hh],
3235
- [hw, hh],
3236
- [-hw, hh]
3237
- ];
3238
- const corners = [];
3239
- let minX = Infinity;
3240
- let minY = Infinity;
3241
- let maxX = -Infinity;
3242
- let maxY = -Infinity;
3243
- for (const [lx, ly] of local) {
3244
- const rx = cx + lx * cos - ly * sin;
3245
- const ry = cy + lx * sin + ly * cos;
3246
- const sx = (rx - dw) / scale;
3247
- const sy = (ry - dh) / scale;
3248
- corners.push([sx, sy]);
3249
- if (sx < minX) minX = sx;
3250
- if (sy < minY) minY = sy;
3251
- if (sx > maxX) maxX = sx;
3252
- if (sy > maxY) maxY = sy;
3253
- }
3254
- return {
3255
- corners,
3256
- minX,
3257
- minY,
3258
- maxX,
3259
- maxY
3037
+ transform
3260
3038
  };
3261
3039
  }
3262
- function decodeObbOutput(output, anchors, nc, opts) {
3263
- const angleRow = opts.angleRow ?? 4 + nc;
3040
+ /** Source pixels a mask may extend past its box (instance masks are not box-clipped). */
3041
+ const MASK_BOX_PAD_PX = 8;
3042
+ function decodeRtmdetIns(out, opts) {
3043
+ const { sourceWidth: sw, sourceHeight: sh, transform } = opts;
3264
3044
  const classFilter = opts.classes ? new Set(opts.classes.map((c) => c.toLowerCase())) : null;
3265
- const candidates = [];
3266
- for (let a = 0; a < anchors; a++) {
3267
- let bestScore = -1;
3268
- let bestClass = -1;
3269
- for (let c = 0; c < nc; c++) {
3270
- const score = output[(4 + c) * anchors + a];
3271
- if (score > bestScore) {
3272
- bestScore = score;
3273
- bestClass = c;
3045
+ const kept = [];
3046
+ for (let i = 0; i < out.count; i++) {
3047
+ const score = out.dets[i * 5 + 4];
3048
+ if (!(score >= opts.confidence)) continue;
3049
+ const classIndex = Number(out.labels[i]);
3050
+ const category = opts.classNames[classIndex] ?? `class_${classIndex}`;
3051
+ if (classFilter && !classFilter.has(category.toLowerCase())) continue;
3052
+ const p0 = toSourcePoint(out.dets[i * 5], out.dets[i * 5 + 1], transform);
3053
+ const p1 = toSourcePoint(out.dets[i * 5 + 2], out.dets[i * 5 + 3], transform);
3054
+ const x0 = clamp$1(p0.x, 0, sw);
3055
+ const y0 = clamp$1(p0.y, 0, sh);
3056
+ const width = clamp$1(p1.x, 0, sw) - x0;
3057
+ const height = clamp$1(p1.y, 0, sh) - y0;
3058
+ if (width <= 0 || height <= 0) continue;
3059
+ kept.push({
3060
+ index: i,
3061
+ detection: {
3062
+ category,
3063
+ score,
3064
+ boundingBox: {
3065
+ originX: x0,
3066
+ originY: y0,
3067
+ width,
3068
+ height,
3069
+ normalizedX: x0 / sw,
3070
+ normalizedY: y0 / sh,
3071
+ normalizedWidth: width / sw,
3072
+ normalizedHeight: height / sh
3073
+ }
3274
3074
  }
3275
- }
3276
- if (bestScore < opts.confidence) continue;
3277
- if (classFilter && !classFilter.has((opts.classNames[bestClass] ?? "").toLowerCase())) continue;
3278
- const cx = output[a];
3279
- const cy = output[anchors + a];
3280
- const w = output[2 * anchors + a];
3281
- const h = output[3 * anchors + a];
3282
- if (w <= 0 || h <= 0) continue;
3283
- const angle = output[angleRow * anchors + a];
3284
- if (!Number.isFinite(angle)) continue;
3285
- candidates.push({
3286
- x0: cx - w / 2,
3287
- y0: cy - h / 2,
3288
- x1: cx + w / 2,
3289
- y1: cy + h / 2,
3290
- score: bestScore,
3291
- classIndex: bestClass,
3292
- anchorIndex: a,
3293
- angle
3294
3075
  });
3295
3076
  }
3296
- const kept = nms(candidates, opts.iouThreshold);
3297
- const detections = [];
3298
- for (const box of kept) {
3299
- const cand = box;
3300
- const cx = output[cand.anchorIndex];
3301
- const cy = output[anchors + cand.anchorIndex];
3302
- const w = output[2 * anchors + cand.anchorIndex];
3303
- const h = output[3 * anchors + cand.anchorIndex];
3304
- const { corners, minX, minY, maxX, maxY } = projectCorners(cx, cy, w, h, cand.angle, opts);
3305
- const sw = opts.sourceWidth;
3306
- const sh = opts.sourceHeight;
3307
- const originX = Math.max(0, Math.min(sw, minX));
3308
- const originY = Math.max(0, Math.min(sh, minY));
3309
- const width = Math.max(0, Math.min(sw, maxX) - originX);
3310
- const height = Math.max(0, Math.min(sh, maxY) - originY);
3311
- detections.push({
3312
- category: opts.classNames[cand.classIndex] ?? `class_${cand.classIndex}`,
3313
- classIndex: cand.classIndex,
3314
- score: cand.score,
3315
- boundingBox: {
3316
- originX,
3317
- originY,
3318
- width,
3319
- height,
3320
- normalizedX: sw > 0 ? originX / sw : 0,
3321
- normalizedY: sh > 0 ? originY / sh : 0,
3322
- normalizedWidth: sw > 0 ? width / sw : 0,
3323
- normalizedHeight: sh > 0 ? height / sh : 0,
3324
- angle: cand.angle
3325
- },
3326
- corners
3077
+ kept.sort((a, b) => b.detection.score - a.detection.score);
3078
+ const detections = kept.map((k) => k.detection);
3079
+ const masks = [];
3080
+ if (!out.masks) return {
3081
+ detections,
3082
+ masks
3083
+ };
3084
+ const alphaOf = probabilityToAlpha(opts.maskThreshold ?? .5, opts.featherRadius ?? .05);
3085
+ const { maskWidth: mw, maskHeight: mh } = out;
3086
+ const plane = mw * mh;
3087
+ const gx = mw / transform.inputWidth;
3088
+ const gy = mh / transform.inputHeight;
3089
+ for (let d = 0; d < kept.length; d++) {
3090
+ const { index, detection } = kept[d];
3091
+ const box = detection.boundingBox;
3092
+ const base = index * plane;
3093
+ const bx0 = Math.max(0, Math.floor(box.originX) - MASK_BOX_PAD_PX);
3094
+ const by0 = Math.max(0, Math.floor(box.originY) - MASK_BOX_PAD_PX);
3095
+ const bx1 = Math.min(sw - 1, Math.ceil(box.originX + box.width) + MASK_BOX_PAD_PX);
3096
+ const by1 = Math.min(sh - 1, Math.ceil(box.originY + box.height) + MASK_BOX_PAD_PX);
3097
+ const mask = new Uint8Array(sw * sh);
3098
+ let area = 0;
3099
+ for (let sy = by0; sy <= by1; sy++) {
3100
+ const my = ((sy + .5) * transform.scaleY + transform.offsetY) * gy - .5;
3101
+ const my0 = clamp$1(Math.floor(my), 0, mh - 1);
3102
+ const my1 = Math.min(mh - 1, my0 + 1);
3103
+ const fy = clamp$1(my - my0, 0, 1);
3104
+ const row = sy * sw;
3105
+ for (let sx = bx0; sx <= bx1; sx++) {
3106
+ const mx = ((sx + .5) * transform.scaleX + transform.offsetX) * gx - .5;
3107
+ const mx0 = clamp$1(Math.floor(mx), 0, mw - 1);
3108
+ const mx1 = Math.min(mw - 1, mx0 + 1);
3109
+ const fx = clamp$1(mx - mx0, 0, 1);
3110
+ const alpha = alphaOf(out.masks[base + my0 * mw + mx0] * (1 - fx) * (1 - fy) + out.masks[base + my0 * mw + mx1] * fx * (1 - fy) + out.masks[base + my1 * mw + mx0] * (1 - fx) * fy + out.masks[base + my1 * mw + mx1] * fx * fy);
3111
+ if (alpha > 0) {
3112
+ mask[row + sx] = alpha;
3113
+ area += alpha / 255;
3114
+ }
3115
+ }
3116
+ }
3117
+ area = Math.round(area);
3118
+ if (area === 0) continue;
3119
+ masks.push({
3120
+ category: detection.category,
3121
+ mask,
3122
+ width: sw,
3123
+ height: sh,
3124
+ area,
3125
+ coverage: area / (sw * sh),
3126
+ detectionIndex: d
3327
3127
  });
3328
3128
  }
3329
- return { detections };
3129
+ return {
3130
+ detections,
3131
+ masks
3132
+ };
3330
3133
  }
3331
- function decodePoseOutput(output, anchors, opts) {
3332
- const { scale, dw, dh } = opts.params;
3333
- const { sourceWidth: sw, sourceHeight: sh } = opts;
3134
+ /**
3135
+ * Probability → 0..255 alpha. A hard threshold when `feather` is 0, otherwise a linear ramp
3136
+ * across `[threshold − feather, threshold + feather]` for anti-aliased edges.
3137
+ */
3138
+ function probabilityToAlpha(threshold, feather) {
3139
+ if (feather <= 0) return (p) => p > threshold ? 255 : 0;
3140
+ const lo = threshold - feather;
3141
+ const inv = 255 / (2 * feather);
3142
+ return (p) => p <= lo ? 0 : p >= threshold + feather ? 255 : (p - lo) * inv;
3143
+ }
3144
+ function clamp$1(v, lo, hi) {
3145
+ return v < lo ? lo : v > hi ? hi : v;
3146
+ }
3147
+ const COCO17_KEYPOINT_COUNT = 17;
3148
+ /** Keypoint visibility at which a joint counts as seen. */
3149
+ const VISIBLE = .3;
3150
+ function decodeRtmo(out, opts) {
3151
+ const { sourceWidth: sw, sourceHeight: sh, transform } = opts;
3152
+ const stride = out.keypointStride;
3334
3153
  const people = [];
3335
- for (let a = 0; a < anchors; a++) {
3336
- const score = output[4 * anchors + a];
3337
- if (score < opts.confidence) continue;
3338
- const cx = output[a];
3339
- const cy = output[anchors + a];
3340
- const w = output[2 * anchors + a];
3341
- const h = output[3 * anchors + a];
3342
- if (w <= 0 || h <= 0) continue;
3343
- const x0 = Math.max(0, Math.min(sw, (cx - w / 2 - dw) / scale));
3344
- const y0 = Math.max(0, Math.min(sh, (cy - h / 2 - dh) / scale));
3345
- const x1 = Math.max(0, Math.min(sw, (cx + w / 2 - dw) / scale));
3346
- const y1 = Math.max(0, Math.min(sh, (cy + h / 2 - dh) / scale));
3347
- const boxW = Math.max(0, x1 - x0);
3348
- const boxH = Math.max(0, y1 - y0);
3154
+ for (let i = 0; i < out.count; i++) {
3155
+ const score = out.dets[i * 5 + 4];
3156
+ if (!(score >= opts.confidence)) continue;
3157
+ const p0 = toSourcePoint(out.dets[i * 5], out.dets[i * 5 + 1], transform);
3158
+ const p1 = toSourcePoint(out.dets[i * 5 + 2], out.dets[i * 5 + 3], transform);
3159
+ const x0 = clamp$2(p0.x, 0, sw);
3160
+ const y0 = clamp$2(p0.y, 0, sh);
3161
+ const width = clamp$2(p1.x, 0, sw) - x0;
3162
+ const height = clamp$2(p1.y, 0, sh) - y0;
3163
+ if (width <= 0 || height <= 0) continue;
3349
3164
  const keypoints = [];
3350
- for (let k = 0; k < 17; k++) {
3351
- const kx = Math.max(0, Math.min(sw, (output[(5 + 3 * k) * anchors + a] - dw) / scale));
3352
- const ky = Math.max(0, Math.min(sh, (output[(6 + 3 * k) * anchors + a] - dh) / scale));
3353
- const vis = output[(7 + 3 * k) * anchors + a];
3165
+ const base = i * COCO17_KEYPOINT_COUNT * stride;
3166
+ for (let k = 0; k < COCO17_KEYPOINT_COUNT; k++) {
3167
+ const o = base + k * stride;
3168
+ const p = toSourcePoint(out.keypoints[o], out.keypoints[o + 1], transform);
3354
3169
  keypoints.push({
3355
- x: kx,
3356
- y: ky,
3357
- visibility: Math.max(0, Math.min(1, vis))
3170
+ x: clamp$2(p.x, 0, sw),
3171
+ y: clamp$2(p.y, 0, sh),
3172
+ visibility: clamp$2(out.keypoints[o + 2], 0, 1)
3358
3173
  });
3359
3174
  }
3175
+ let visible = 0;
3176
+ for (const k of keypoints) if (k.visibility >= VISIBLE) visible++;
3177
+ if (visible < (opts.minVisibleKeypoints ?? 3)) continue;
3360
3178
  people.push({
3361
3179
  score,
3362
3180
  boundingBox: {
3363
3181
  originX: x0,
3364
3182
  originY: y0,
3365
- width: boxW,
3366
- height: boxH,
3367
- normalizedX: sw > 0 ? x0 / sw : 0,
3368
- normalizedY: sh > 0 ? y0 / sh : 0,
3369
- normalizedWidth: sw > 0 ? boxW / sw : 0,
3370
- normalizedHeight: sh > 0 ? boxH / sh : 0
3183
+ width,
3184
+ height,
3185
+ normalizedX: x0 / sw,
3186
+ normalizedY: y0 / sh,
3187
+ normalizedWidth: width / sw,
3188
+ normalizedHeight: height / sh
3371
3189
  },
3372
3190
  keypoints
3373
3191
  });
3374
3192
  }
3375
3193
  people.sort((a, b) => b.score - a.score);
3376
- return { people };
3194
+ const iou = opts.iouThreshold ?? .6;
3195
+ const kept = [];
3196
+ for (const p of people) if (!kept.some((k) => isDuplicate(k, p, iou))) kept.push(p);
3197
+ return { people: kept };
3198
+ }
3199
+ function clamp$2(v, lo, hi) {
3200
+ return v < lo ? lo : v > hi ? hi : v;
3377
3201
  }
3378
3202
  /**
3379
- * Instance segmentation decode.
3380
- *
3381
- * Head: [1, 4 + nc + 32, 8400] — the 32 trailing rows are per-detect mask coefficients.
3382
- * Proto: [1, 32, protoDim, protoDim] (protoDim = imgsz / 4, e.g. 160 at 640).
3383
- *
3384
- * Per kept detection: mask = sigmoid(Σ coeff_c · proto_c) → threshold 0.5, then bilinear
3385
- * upscaled from the 160-grid directly into the ORIGINAL source frame (letterbox-inverse),
3386
- * restricted to the detection box so masks stay tight and cheap.
3203
+ * Same person twice: boxes overlap above `iou`, or the joints both detections can see sit
3204
+ * within 10% of the stronger detection's box size of each other (boxes may differ when one
3205
+ * of them also covers a flowing garment or a shadow).
3387
3206
  */
3388
- function decodeSegmentOutput(output, proto, anchors, nc, protoDim, opts) {
3389
- const kept = decodeYoloBoxes(output, anchors, nc, opts);
3390
- const detections = kept.map((box) => toDetectedObject(box, opts));
3391
- const masks = [];
3392
- const { scale, dw, dh, size } = opts.params;
3207
+ function isDuplicate(a, b, iou) {
3208
+ if (computeIoU(a.boundingBox, b.boundingBox) > iou) return true;
3209
+ const scale = Math.sqrt(a.boundingBox.width * a.boundingBox.height);
3210
+ let sum = 0;
3211
+ let n = 0;
3212
+ for (let k = 0; k < a.keypoints.length; k++) {
3213
+ const ka = a.keypoints[k];
3214
+ const kb = b.keypoints[k];
3215
+ if (ka.visibility < VISIBLE || kb.visibility < VISIBLE) continue;
3216
+ sum += Math.hypot(ka.x - kb.x, ka.y - kb.y);
3217
+ n++;
3218
+ }
3219
+ return n >= 3 && sum / n < .1 * scale;
3220
+ }
3221
+ function decodeSelfie(alphas, matteWidth, matteHeight, opts) {
3393
3222
  const { sourceWidth: sw, sourceHeight: sh } = opts;
3394
- const maskThreshold = opts.maskThreshold ?? .5;
3395
- const feather = opts.featherRadius !== void 0 ? opts.featherRadius : .05;
3396
- const maskLogit = Math.log(maskThreshold / (1 - maskThreshold));
3397
- const tMin = maskThreshold - feather;
3398
- const tMax = maskThreshold + feather;
3399
- const invTwoFeather = feather > 0 ? 1 / (2 * feather) : 0;
3400
- for (let i = 0; i < kept.length; i++) {
3401
- const box = kept[i];
3402
- const det = detections[i];
3403
- const coeffBase = (4 + nc) * anchors + box.anchorIndex;
3404
- const step = anchors;
3405
- const padPx = 8;
3406
- const sx0 = Math.max(0, Math.floor((box.x0 - dw) / scale) - padPx);
3407
- const sy0 = Math.max(0, Math.floor((box.y0 - dh) / scale) - padPx);
3408
- const sx1 = Math.min(sw - 1, Math.ceil((box.x1 - dw) / scale) + padPx);
3409
- const sy1 = Math.min(sh - 1, Math.ceil((box.y1 - dh) / scale) + padPx);
3410
- if (sx1 < sx0 || sy1 < sy0) continue;
3411
- const mask = new Uint8Array(sw * sh);
3412
- const inv = protoDim / size;
3413
- let area = 0;
3414
- for (let sy = sy0; sy <= sy1; sy++) {
3415
- const gyy = (dh + sy * scale) * inv;
3416
- const gy0 = Math.floor(gyy);
3417
- const fy = gyy - gy0;
3418
- const gy1 = Math.min(protoDim - 1, gy0 + 1);
3419
- const maskRow = sy * sw;
3420
- for (let sx = sx0; sx <= sx1; sx++) {
3421
- const gxx = (dw + sx * scale) * inv;
3422
- const gx0 = Math.floor(gxx);
3423
- const fx = gxx - gx0;
3424
- const gx1 = Math.min(protoDim - 1, gx0 + 1);
3425
- let acc = 0;
3426
- for (let c = 0; c < 32; c++) {
3427
- const coeff = output[coeffBase + c * step];
3428
- if (coeff === 0) continue;
3429
- const rowBase = c * protoDim * protoDim;
3430
- const v00 = proto[rowBase + gy0 * protoDim + gx0];
3431
- const v10 = proto[rowBase + gy0 * protoDim + gx1];
3432
- const v01 = proto[rowBase + gy1 * protoDim + gx0];
3433
- const v11 = proto[rowBase + gy1 * protoDim + gx1];
3434
- const interp = v00 * (1 - fx) * (1 - fy) + v10 * fx * (1 - fy) + v01 * (1 - fx) * fy + v11 * fx * fy;
3435
- acc += coeff * interp;
3436
- }
3437
- if (feather <= 0) {
3438
- if (acc > maskLogit) {
3439
- mask[maskRow + sx] = 255;
3440
- area++;
3441
- }
3442
- } else {
3443
- const alpha = acc > 16 ? 1 : acc < -16 ? 0 : 1 / (1 + Math.exp(-acc));
3444
- let alphaNorm = 0;
3445
- if (alpha >= tMax) alphaNorm = 1;
3446
- else if (alpha > tMin) alphaNorm = (alpha - tMin) * invTwoFeather;
3447
- const alphaOut = Math.round(alphaNorm * 255);
3448
- if (alphaOut > 0) {
3449
- mask[maskRow + sx] = alphaOut;
3450
- area += alphaNorm;
3451
- }
3452
- }
3453
- }
3223
+ const alphaOf = probabilityToAlpha(opts.maskThreshold ?? .5, opts.featherRadius ?? .2);
3224
+ const mask = new Uint8Array(sw * sh);
3225
+ const kx = matteWidth / sw;
3226
+ const ky = matteHeight / sh;
3227
+ let sum = 0;
3228
+ for (let y = 0; y < sh; y++) {
3229
+ const my = Math.max(0, (y + .5) * ky - .5);
3230
+ const my0 = Math.min(matteHeight - 1, Math.floor(my));
3231
+ const my1 = Math.min(matteHeight - 1, my0 + 1);
3232
+ const fy = my - my0;
3233
+ for (let x = 0; x < sw; x++) {
3234
+ const mx = Math.max(0, (x + .5) * kx - .5);
3235
+ const mx0 = Math.min(matteWidth - 1, Math.floor(mx));
3236
+ const mx1 = Math.min(matteWidth - 1, mx0 + 1);
3237
+ const fx = mx - mx0;
3238
+ const a = alphaOf(alphas[my0 * matteWidth + mx0] * (1 - fx) * (1 - fy) + alphas[my0 * matteWidth + mx1] * fx * (1 - fy) + alphas[my1 * matteWidth + mx0] * (1 - fx) * fy + alphas[my1 * matteWidth + mx1] * fx * fy);
3239
+ mask[y * sw + x] = a;
3240
+ sum += a;
3454
3241
  }
3455
- area = Math.round(area);
3456
- if (area === 0) continue;
3457
- masks.push({
3458
- category: det.category,
3459
- mask,
3460
- width: sw,
3461
- height: sh,
3462
- area,
3463
- coverage: sw * sh > 0 ? area / (sw * sh) : 0,
3464
- detectionIndex: i
3465
- });
3466
3242
  }
3467
3243
  return {
3468
- detections,
3469
- masks
3244
+ mask,
3245
+ width: sw,
3246
+ height: sh,
3247
+ coverage: sw * sh > 0 ? sum / (255 * sw * sh) : 0
3470
3248
  };
3471
3249
  }
3472
- /** COCO-17 keypoint indices for YOLO pose outputs (matches YOLO_POSE_KEYPOINTS order). */
3473
- const POSE_LANDMARKS_YOLO = {
3250
+ /** COCO-17 keypoint indices (RTMO output order). */
3251
+ const COCO17_KEYPOINTS = {
3474
3252
  NOSE: 0,
3475
3253
  LEFT_EYE: 1,
3476
3254
  RIGHT_EYE: 2,
@@ -3489,7 +3267,7 @@ const POSE_LANDMARKS_YOLO = {
3489
3267
  LEFT_ANKLE: 15,
3490
3268
  RIGHT_ANKLE: 16
3491
3269
  };
3492
- const YOLO_POSE_KEYPOINT_NAMES = [
3270
+ const COCO17_KEYPOINT_NAMES = [
3493
3271
  "nose",
3494
3272
  "leftEye",
3495
3273
  "rightEye",
@@ -3509,10 +3287,10 @@ const YOLO_POSE_KEYPOINT_NAMES = [
3509
3287
  "rightAnkle"
3510
3288
  ];
3511
3289
  /** COCO-17 skeleton bones for the WebGPU skeleton renderer (indices above). */
3512
- const YOLO_COCO17_BONES = [
3290
+ const COCO17_BONES = [
3513
3291
  {
3514
- from: POSE_LANDMARKS_YOLO.NOSE,
3515
- to: POSE_LANDMARKS_YOLO.LEFT_SHOULDER,
3292
+ from: COCO17_KEYPOINTS.NOSE,
3293
+ to: COCO17_KEYPOINTS.LEFT_SHOULDER,
3516
3294
  color: [
3517
3295
  0,
3518
3296
  1,
@@ -3521,8 +3299,8 @@ const YOLO_COCO17_BONES = [
3521
3299
  ]
3522
3300
  },
3523
3301
  {
3524
- from: POSE_LANDMARKS_YOLO.NOSE,
3525
- to: POSE_LANDMARKS_YOLO.RIGHT_SHOULDER,
3302
+ from: COCO17_KEYPOINTS.NOSE,
3303
+ to: COCO17_KEYPOINTS.RIGHT_SHOULDER,
3526
3304
  color: [
3527
3305
  0,
3528
3306
  1,
@@ -3531,8 +3309,8 @@ const YOLO_COCO17_BONES = [
3531
3309
  ]
3532
3310
  },
3533
3311
  {
3534
- from: POSE_LANDMARKS_YOLO.LEFT_SHOULDER,
3535
- to: POSE_LANDMARKS_YOLO.RIGHT_SHOULDER,
3312
+ from: COCO17_KEYPOINTS.LEFT_SHOULDER,
3313
+ to: COCO17_KEYPOINTS.RIGHT_SHOULDER,
3536
3314
  color: [
3537
3315
  1,
3538
3316
  0,
@@ -3541,8 +3319,8 @@ const YOLO_COCO17_BONES = [
3541
3319
  ]
3542
3320
  },
3543
3321
  {
3544
- from: POSE_LANDMARKS_YOLO.LEFT_SHOULDER,
3545
- to: POSE_LANDMARKS_YOLO.LEFT_ELBOW,
3322
+ from: COCO17_KEYPOINTS.LEFT_SHOULDER,
3323
+ to: COCO17_KEYPOINTS.LEFT_ELBOW,
3546
3324
  color: [
3547
3325
  1,
3548
3326
  .333,
@@ -3551,8 +3329,8 @@ const YOLO_COCO17_BONES = [
3551
3329
  ]
3552
3330
  },
3553
3331
  {
3554
- from: POSE_LANDMARKS_YOLO.LEFT_ELBOW,
3555
- to: POSE_LANDMARKS_YOLO.LEFT_WRIST,
3332
+ from: COCO17_KEYPOINTS.LEFT_ELBOW,
3333
+ to: COCO17_KEYPOINTS.LEFT_WRIST,
3556
3334
  color: [
3557
3335
  1,
3558
3336
  .667,
@@ -3561,8 +3339,8 @@ const YOLO_COCO17_BONES = [
3561
3339
  ]
3562
3340
  },
3563
3341
  {
3564
- from: POSE_LANDMARKS_YOLO.RIGHT_SHOULDER,
3565
- to: POSE_LANDMARKS_YOLO.RIGHT_ELBOW,
3342
+ from: COCO17_KEYPOINTS.RIGHT_SHOULDER,
3343
+ to: COCO17_KEYPOINTS.RIGHT_ELBOW,
3566
3344
  color: [
3567
3345
  1,
3568
3346
  1,
@@ -3571,8 +3349,8 @@ const YOLO_COCO17_BONES = [
3571
3349
  ]
3572
3350
  },
3573
3351
  {
3574
- from: POSE_LANDMARKS_YOLO.RIGHT_ELBOW,
3575
- to: POSE_LANDMARKS_YOLO.RIGHT_WRIST,
3352
+ from: COCO17_KEYPOINTS.RIGHT_ELBOW,
3353
+ to: COCO17_KEYPOINTS.RIGHT_WRIST,
3576
3354
  color: [
3577
3355
  .667,
3578
3356
  1,
@@ -3581,8 +3359,8 @@ const YOLO_COCO17_BONES = [
3581
3359
  ]
3582
3360
  },
3583
3361
  {
3584
- from: POSE_LANDMARKS_YOLO.LEFT_SHOULDER,
3585
- to: POSE_LANDMARKS_YOLO.LEFT_HIP,
3362
+ from: COCO17_KEYPOINTS.LEFT_SHOULDER,
3363
+ to: COCO17_KEYPOINTS.LEFT_HIP,
3586
3364
  color: [
3587
3365
  .333,
3588
3366
  1,
@@ -3591,8 +3369,8 @@ const YOLO_COCO17_BONES = [
3591
3369
  ]
3592
3370
  },
3593
3371
  {
3594
- from: POSE_LANDMARKS_YOLO.RIGHT_SHOULDER,
3595
- to: POSE_LANDMARKS_YOLO.RIGHT_HIP,
3372
+ from: COCO17_KEYPOINTS.RIGHT_SHOULDER,
3373
+ to: COCO17_KEYPOINTS.RIGHT_HIP,
3596
3374
  color: [
3597
3375
  0,
3598
3376
  1,
@@ -3601,8 +3379,8 @@ const YOLO_COCO17_BONES = [
3601
3379
  ]
3602
3380
  },
3603
3381
  {
3604
- from: POSE_LANDMARKS_YOLO.LEFT_HIP,
3605
- to: POSE_LANDMARKS_YOLO.RIGHT_HIP,
3382
+ from: COCO17_KEYPOINTS.LEFT_HIP,
3383
+ to: COCO17_KEYPOINTS.RIGHT_HIP,
3606
3384
  color: [
3607
3385
  0,
3608
3386
  1,
@@ -3611,8 +3389,8 @@ const YOLO_COCO17_BONES = [
3611
3389
  ]
3612
3390
  },
3613
3391
  {
3614
- from: POSE_LANDMARKS_YOLO.LEFT_HIP,
3615
- to: POSE_LANDMARKS_YOLO.LEFT_KNEE,
3392
+ from: COCO17_KEYPOINTS.LEFT_HIP,
3393
+ to: COCO17_KEYPOINTS.LEFT_KNEE,
3616
3394
  color: [
3617
3395
  0,
3618
3396
  1,
@@ -3621,8 +3399,8 @@ const YOLO_COCO17_BONES = [
3621
3399
  ]
3622
3400
  },
3623
3401
  {
3624
- from: POSE_LANDMARKS_YOLO.LEFT_KNEE,
3625
- to: POSE_LANDMARKS_YOLO.LEFT_ANKLE,
3402
+ from: COCO17_KEYPOINTS.LEFT_KNEE,
3403
+ to: COCO17_KEYPOINTS.LEFT_ANKLE,
3626
3404
  color: [
3627
3405
  0,
3628
3406
  1,
@@ -3631,8 +3409,8 @@ const YOLO_COCO17_BONES = [
3631
3409
  ]
3632
3410
  },
3633
3411
  {
3634
- from: POSE_LANDMARKS_YOLO.RIGHT_HIP,
3635
- to: POSE_LANDMARKS_YOLO.RIGHT_KNEE,
3412
+ from: COCO17_KEYPOINTS.RIGHT_HIP,
3413
+ to: COCO17_KEYPOINTS.RIGHT_KNEE,
3636
3414
  color: [
3637
3415
  0,
3638
3416
  .667,
@@ -3641,8 +3419,8 @@ const YOLO_COCO17_BONES = [
3641
3419
  ]
3642
3420
  },
3643
3421
  {
3644
- from: POSE_LANDMARKS_YOLO.RIGHT_KNEE,
3645
- to: POSE_LANDMARKS_YOLO.RIGHT_ANKLE,
3422
+ from: COCO17_KEYPOINTS.RIGHT_KNEE,
3423
+ to: COCO17_KEYPOINTS.RIGHT_ANKLE,
3646
3424
  color: [
3647
3425
  0,
3648
3426
  .333,
@@ -3651,35 +3429,18 @@ const YOLO_COCO17_BONES = [
3651
3429
  ]
3652
3430
  }
3653
3431
  ];
3654
- /** Mapping from MediaPipe 33-pose indices to YOLO COCO-17 (named props survive unchanged). */
3655
- const MEDIAPIPE_TO_YOLO_POSE_INDEX = {
3656
- 0: 0,
3657
- 11: 5,
3658
- 12: 6,
3659
- 13: 7,
3660
- 14: 8,
3661
- 15: 9,
3662
- 16: 10,
3663
- 23: 11,
3664
- 24: 12,
3665
- 25: 13,
3666
- 26: 14,
3667
- 27: 15,
3668
- 28: 16
3669
- };
3670
3432
  /**
3671
- * WebGPU skeleton renderer for YOLO pose results.
3433
+ * WebGPU skeleton renderer for pose results.
3672
3434
  *
3673
3435
  * Rasterizes the primary person's COCO-17 keypoints into an OpenPose-style
3674
- * conditioning texture natively in VRAM — same compute pipeline as the MediaPipe
3675
- * renderer, with the COCO-17 bone set instead of the 33-point OpenPose set.
3436
+ * conditioning texture natively in VRAM, using the COCO-17 bone set.
3676
3437
  */
3677
3438
  var PoseSkeletonRenderer = class {
3678
3439
  pipeline;
3679
3440
  constructor(device) {
3680
3441
  this.pipeline = new PoseSkeletonComputePipeline(device);
3681
3442
  }
3682
- renderToTexture(keypoints, options, bones = YOLO_COCO17_BONES) {
3443
+ renderToTexture(keypoints, options, bones = COCO17_BONES) {
3683
3444
  return this.pipeline.execute(keypoints, options, bones);
3684
3445
  }
3685
3446
  destroy() {
@@ -3698,7 +3459,7 @@ var SegmentationTexturePool = class {
3698
3459
  if (existing) existing.destroy();
3699
3460
  const format = options.format ?? "rgba8unorm";
3700
3461
  const texture = this.device.createTexture({
3701
- label: options.label ?? `Yolo_Segmentation_${key}`,
3462
+ label: options.label ?? `vision_segmentation_${key}`,
3702
3463
  size: [
3703
3464
  options.width,
3704
3465
  options.height,
@@ -3756,18 +3517,19 @@ function getDefaultModelsDir() {
3756
3517
  return resolve(homedir(), ".cache/gitframes/models");
3757
3518
  }
3758
3519
  /**
3759
- * Lazy YOLO model store. Constructor performs ZERO I/O; the only entry that can hit the
3520
+ * Lazy model store. The constructor performs ZERO I/O; the only entry that can hit the
3760
3521
  * network is `ensure(key)`, called by an inference function the first time it runs.
3761
3522
  * Concurrent callers of the same key share a single in-flight download (promise dedupe),
3762
- * and verified files on disk are reused across processes.
3763
- *
3764
- * Mirrors the proven mechanics of the old MediaPipe model manager (atomic temp+rename,
3765
- * size verification, `GITFRAMES_MODELS_DIR` override) minus its eager constructor downloads.
3523
+ * downloads are written atomically (temp + rename), and verified files on disk are reused
3524
+ * across processes.
3766
3525
  */
3767
- var YoloModelStore = class YoloModelStore$1 {
3526
+ var VisionModelStore = class {
3768
3527
  _modelsDir;
3769
3528
  _baseUrl;
3770
3529
  _timeoutMs;
3530
+ _retries;
3531
+ _retryDelayMs;
3532
+ _verify;
3771
3533
  _fetch;
3772
3534
  _onProgress;
3773
3535
  _inflight = /* @__PURE__ */ new Map();
@@ -3775,8 +3537,11 @@ var YoloModelStore = class YoloModelStore$1 {
3775
3537
  _status = /* @__PURE__ */ new Map();
3776
3538
  constructor(options = {}) {
3777
3539
  this._modelsDir = options.modelsDir ? resolve(options.modelsDir) : getDefaultModelsDir();
3778
- this._baseUrl = options.baseUrl ?? process.env.GITFRAMES_YOLO_BASE_URL ?? DEFAULT_YOLO_BASE_URL;
3540
+ this._baseUrl = options.baseUrl ?? process.env.GITFRAMES_MODELS_BASE_URL ?? void 0;
3779
3541
  this._timeoutMs = options.timeoutMs;
3542
+ this._retries = Math.max(0, options.retries ?? 2);
3543
+ this._retryDelayMs = options.retryDelayMs ?? 1e3;
3544
+ this._verify = options.verify !== false;
3780
3545
  this._fetch = options.fetchImpl ?? globalThis.fetch;
3781
3546
  this._onProgress = options.onProgress;
3782
3547
  }
@@ -3784,22 +3549,24 @@ var YoloModelStore = class YoloModelStore$1 {
3784
3549
  return this._modelsDir;
3785
3550
  }
3786
3551
  pathFor(key) {
3787
- return join(this._modelsDir, YOLO_MODELS[key].filename);
3552
+ return join(this._modelsDir, VISION_MODELS[key].filename);
3788
3553
  }
3789
3554
  descriptor(key) {
3790
- return YOLO_MODELS[key];
3555
+ return VISION_MODELS[key];
3791
3556
  }
3792
- /**
3793
- * Smallest file we trust as a cached model. Every YOLO11 export in the registry
3794
- * ships well above 1 MB (smallest is yolo11n-cls ≈ 1.2 MB); a smaller file is a
3795
- * partial/stub write (e.g. a 4 KiB test fixture) and must be re-downloaded.
3796
- */
3797
- static MIN_VALID_MODEL_BYTES = 512 * 1024;
3557
+ /** The URL `ensure(key)` downloads from (mirror when `baseUrl` is set). */
3558
+ urlFor(key) {
3559
+ const desc = VISION_MODELS[key];
3560
+ if (!this._baseUrl) return desc.url;
3561
+ return `${this._baseUrl.replace(/\/+$/, "")}/${desc.filename}`;
3562
+ }
3563
+ /** True when a complete cached file exists (exact registry size when verifying). */
3798
3564
  has(key) {
3799
3565
  const targetPath = this.pathFor(key);
3800
3566
  if (!existsSync(targetPath)) return false;
3801
3567
  try {
3802
- return statSync(targetPath).size >= YoloModelStore$1.MIN_VALID_MODEL_BYTES;
3568
+ const size = statSync(targetPath).size;
3569
+ return this._verify ? size === VISION_MODELS[key].bytes : size > 0;
3803
3570
  } catch {
3804
3571
  return false;
3805
3572
  }
@@ -3810,13 +3577,6 @@ var YoloModelStore = class YoloModelStore$1 {
3810
3577
  this._status.set(key, "pending");
3811
3578
  await unlink(this.pathFor(key)).catch(() => void 0);
3812
3579
  }
3813
- /** Direct download helper for arbitrary model URLs, respecting injected fetch. */
3814
- async fetchDirect(url) {
3815
- const res = await this._fetch(url);
3816
- if (!res.ok) throw new Error(`Failed to fetch model from ${url}: ${res.statusText}`);
3817
- const buf = await res.arrayBuffer();
3818
- return new Uint8Array(buf);
3819
- }
3820
3580
  status(key) {
3821
3581
  return this._status.get(key) ?? "pending";
3822
3582
  }
@@ -3843,7 +3603,7 @@ var YoloModelStore = class YoloModelStore$1 {
3843
3603
  }
3844
3604
  /** Explicit warm-up — fetch multiple models ahead of use (agents, renderers). */
3845
3605
  async preload(keys) {
3846
- await Promise.all(keys.map((key) => this.ensure(key)));
3606
+ await Promise.all([...new Set(keys)].map((key) => this.ensure(key)));
3847
3607
  }
3848
3608
  async load(key) {
3849
3609
  const targetPath = this.pathFor(key);
@@ -3851,19 +3611,15 @@ var YoloModelStore = class YoloModelStore$1 {
3851
3611
  this._status.set(key, "ready");
3852
3612
  return this.readBuffer(key, targetPath);
3853
3613
  }
3854
- const desc = YOLO_MODELS[key];
3855
- const downloadUrl = desc.url ?? `${this._baseUrl}${desc.filename}`;
3614
+ const downloadUrl = this.urlFor(key);
3856
3615
  this._status.set(key, "downloading");
3857
3616
  if (!existsSync(dirname(targetPath))) mkdirSync(dirname(targetPath), { recursive: true });
3858
3617
  const tempPath = `${targetPath}.tmp.${Date.now()}`;
3859
3618
  try {
3860
- const response = await this._fetch(downloadUrl, { signal: this._timeoutMs ? AbortSignal.timeout(this._timeoutMs) : void 0 });
3861
- if (!response.ok) throw new Error(`Failed to download YOLO model '${key}' from ${downloadUrl}: HTTP ${response.status} ${response.statusText}`);
3862
- const totalBytes = Number(response.headers.get("content-length") ?? 0);
3863
- const arrayBuffer = await response.arrayBuffer();
3864
- const buffer = new Uint8Array(arrayBuffer);
3865
- if (buffer.byteLength === 0) throw new Error(`Downloaded YOLO model '${key}' is empty (0 bytes).`);
3866
- if (this._onProgress) this._onProgress(key, buffer.byteLength, totalBytes || buffer.byteLength);
3619
+ const { buffer, totalBytes } = await this.download(key, downloadUrl);
3620
+ if (buffer.byteLength === 0) throw new Error(`Downloaded vision model '${key}' is empty (0 bytes).`);
3621
+ if (this._verify) this.verifyBytes(key, buffer, downloadUrl);
3622
+ this._onProgress?.(key, buffer.byteLength, totalBytes || buffer.byteLength);
3867
3623
  await writeFile(tempPath, buffer);
3868
3624
  await rename(tempPath, targetPath);
3869
3625
  this._buffers.set(key, buffer);
@@ -3875,6 +3631,36 @@ var YoloModelStore = class YoloModelStore$1 {
3875
3631
  throw error;
3876
3632
  }
3877
3633
  }
3634
+ /** Fetches with retries on transient failures (network errors, HTTP 429/5xx). */
3635
+ async download(key, url) {
3636
+ for (let attempt = 0;; attempt++) {
3637
+ let transient;
3638
+ try {
3639
+ const response = await this._fetch(url, { signal: this._timeoutMs ? AbortSignal.timeout(this._timeoutMs) : void 0 });
3640
+ if (response.ok) {
3641
+ const totalBytes = Number(response.headers.get("content-length") ?? 0);
3642
+ return {
3643
+ buffer: new Uint8Array(await response.arrayBuffer()),
3644
+ totalBytes
3645
+ };
3646
+ }
3647
+ const error = /* @__PURE__ */ new Error(`Failed to download vision model '${key}' from ${url}: HTTP ${response.status} ${response.statusText}`);
3648
+ if (response.status !== 429 && response.status < 500) throw error;
3649
+ transient = error;
3650
+ } catch (error) {
3651
+ if (!(error instanceof TypeError)) throw error;
3652
+ transient = error;
3653
+ }
3654
+ if (attempt >= this._retries) throw new Error(`Failed to download vision model '${key}' from ${url} after ${attempt + 1} attempts: ${String(transient)}`, { cause: transient });
3655
+ await new Promise((r) => setTimeout(r, this._retryDelayMs * 2 ** attempt));
3656
+ }
3657
+ }
3658
+ verifyBytes(key, buffer, url) {
3659
+ const desc = VISION_MODELS[key];
3660
+ if (buffer.byteLength !== desc.bytes) throw new Error(`Vision model '${key}' from ${url} is ${buffer.byteLength} bytes, expected ${desc.bytes}.`);
3661
+ const digest = createHash("sha256").update(buffer).digest("hex");
3662
+ if (digest !== desc.sha256) throw new Error(`Vision model '${key}' from ${url} failed its SHA-256 check (got ${digest}, expected ${desc.sha256}).`);
3663
+ }
3878
3664
  readBuffer(key, targetPath) {
3879
3665
  const fileBuffer = readFileSync(targetPath);
3880
3666
  const buffer = new Uint8Array(fileBuffer.buffer, fileBuffer.byteOffset, fileBuffer.byteLength);
@@ -3918,7 +3704,7 @@ var NodeSessionProvider = class {
3918
3704
  });
3919
3705
  return this.ortPromise;
3920
3706
  }
3921
- async createSession(modelBytes, _opts) {
3707
+ async createSession(modelBytes) {
3922
3708
  const ort = await this.ort();
3923
3709
  const session = await ort.InferenceSession.create(modelBytes, {
3924
3710
  executionProviders: ["cpu"],
@@ -3926,12 +3712,7 @@ var NodeSessionProvider = class {
3926
3712
  });
3927
3713
  return {
3928
3714
  run: async (input) => {
3929
- const inputs = Array.isArray(input) ? input : [input];
3930
- const feeds = {};
3931
- for (const inp of inputs) {
3932
- const inputName = session.inputNames.includes(inp.name) ? inp.name : session.inputNames[0] ?? inp.name;
3933
- feeds[inputName] = new ort.Tensor("float32", inp.data, [...inp.dims]);
3934
- }
3715
+ const feeds = { [session.inputNames.includes(input.name) ? input.name : session.inputNames[0] ?? input.name]: new ort.Tensor("float32", input.data, [...input.dims]) };
3935
3716
  const results = await session.run(feeds);
3936
3717
  const outputs = {};
3937
3718
  for (const [name, tensor] of Object.entries(results)) outputs[name] = {
@@ -3946,48 +3727,33 @@ var NodeSessionProvider = class {
3946
3727
  };
3947
3728
  }
3948
3729
  };
3949
- const YOLO_ANCHORS_FOR_SIZE = {
3950
- 320: 2100,
3951
- 640: 8400,
3952
- 1024: 21504,
3953
- 1280: 33600
3954
- };
3955
3730
  /**
3956
- * Lazy YOLO11 runner.
3731
+ * Lazy vision runner.
3957
3732
  *
3958
- * `create()` performs ZERO I/O: no model downloads, no sessions, no environment patching.
3959
- * Each inference method (`detect` / `segment` / `pose` / `classify`) downloads its model
3960
- * and warms its session on FIRST USE ONLY, then reuses them for the process lifetime.
3733
+ * `create()` performs ZERO I/O: no downloads, no sessions. Each inference method downloads
3734
+ * its model and warms its session on FIRST USE ONLY, then reuses them for the process
3735
+ * lifetime. `detect()` and `segment()` share one RTMDet-Ins forward pass per frame.
3961
3736
  * Call `close()` to release sessions.
3962
3737
  */
3963
- var YoloVisionRunner = class YoloVisionRunner$1 {
3738
+ var VisionRunner = class VisionRunner$1 {
3964
3739
  variant;
3965
- imgsz;
3966
3740
  confidence;
3967
- iouThreshold;
3968
3741
  classes;
3969
- enableWorld;
3970
- prompts;
3971
- customModel;
3972
3742
  maskThreshold;
3973
3743
  featherRadius;
3974
3744
  _store;
3975
3745
  _provider;
3976
3746
  _sessions = /* @__PURE__ */ new Map();
3977
3747
  _tensorScratch = /* @__PURE__ */ new Map();
3978
- _customSessionPromise;
3748
+ /** One RTMDet-Ins pass per frame buffer, shared by detect() and segment(). */
3749
+ _instancePass;
3979
3750
  constructor(options) {
3980
- this.variant = options.variant ?? "n";
3981
- this.imgsz = options.imgsz ?? 640;
3982
- this.confidence = options.confidence ?? .25;
3983
- this.iouThreshold = options.iouThreshold ?? .45;
3751
+ this.variant = options.variant ?? "s";
3752
+ this.confidence = options.confidence ?? .3;
3984
3753
  this.classes = options.classes;
3985
- this.enableWorld = options.enableWorld;
3986
- this.prompts = options.prompts;
3987
- this.customModel = options.customModel;
3988
3754
  this.maskThreshold = options.maskThreshold ?? .5;
3989
3755
  this.featherRadius = options.featherRadius;
3990
- this._store = new YoloModelStore({
3756
+ this._store = options.store ?? new VisionModelStore({
3991
3757
  modelsDir: options.modelsDir,
3992
3758
  baseUrl: options.baseUrl,
3993
3759
  timeoutMs: options.timeoutMs,
@@ -3997,7 +3763,7 @@ var YoloVisionRunner = class YoloVisionRunner$1 {
3997
3763
  }
3998
3764
  /** Cheap: no downloads, no sessions, no side effects. */
3999
3765
  static create(options = {}) {
4000
- return new YoloVisionRunner$1(options);
3766
+ return new VisionRunner$1(options);
4001
3767
  }
4002
3768
  get modelsDir() {
4003
3769
  return this._store.modelsDir;
@@ -4005,291 +3771,144 @@ var YoloVisionRunner = class YoloVisionRunner$1 {
4005
3771
  /** Every registry model keyed to its current download state (downloaded → "ready"). */
4006
3772
  get downloadStatus() {
4007
3773
  const statuses = /* @__PURE__ */ new Map();
4008
- for (const key of Object.keys(YOLO_MODELS)) statuses.set(key, "pending");
3774
+ for (const key of Object.keys(VISION_MODELS)) statuses.set(key, "pending");
4009
3775
  for (const [key, status] of this._store.statuses()) statuses.set(key, status);
4010
3776
  return statuses;
4011
3777
  }
4012
- modelKeyFor(task) {
4013
- return DEFAULT_MODEL_KEY_BY_TASK[task](this.variant);
4014
- }
4015
- /** Explicit warm-up: downloads + sessions for the given tasks, ahead of first use. */
4016
- async preload(tasks) {
4017
- if (this.customModel) {
4018
- await this.customSession();
4019
- return;
4020
- }
4021
- const defaultTasks = this.enableWorld || this.prompts ? [
4022
- "world",
4023
- "segment",
4024
- "pose"
4025
- ] : [
4026
- "detect",
4027
- "segment",
4028
- "pose"
4029
- ];
4030
- const keys = (tasks ?? defaultTasks).map((t) => this.modelKeyFor(t));
4031
- await this._store.preload(keys);
4032
- await Promise.all(keys.map((key) => this.session(key)));
3778
+ /** The model serving `task` (matte depends on frame aspect; defaults to square). */
3779
+ modelKeyFor(task, aspect = 1) {
3780
+ return modelKeyFor(task, this.variant, aspect);
4033
3781
  }
3782
+ /**
3783
+ * Explicit warm-up: downloads + sessions for the given tasks, ahead of first use.
3784
+ * `matte` warms both aspect variants (the frame aspect is unknown until inference).
3785
+ */
3786
+ async preload(tasks = ["detect"]) {
3787
+ const keys = /* @__PURE__ */ new Set();
3788
+ for (const task of tasks) if (task === "matte") {
3789
+ keys.add("selfie-square");
3790
+ keys.add("selfie-landscape");
3791
+ } else keys.add(this.modelKeyFor(task));
3792
+ await this._store.preload([...keys]);
3793
+ await Promise.all([...keys].map((key) => this.session(key)));
3794
+ }
3795
+ /** COCO-80 boxes (sorted by score, source pixels). Masks are not decoded. */
4034
3796
  async detect(image) {
4035
- if (this.customModel && (!this.customModel.task || this.customModel.task === "detect" || this.customModel.task === "world")) return this.detectCustom(image);
4036
- if (this.enableWorld || this.prompts) return this.detectWorld(image, this.prompts);
4037
- const key = this.modelKeyFor("detect");
4038
- const session = await this.session(key);
4039
- const { tensor, params } = this.prepareInput(image, key);
4040
- const preds = pickByDims(await session.run({
4041
- name: "images",
4042
- data: tensor,
4043
- dims: [
4044
- 1,
4045
- 3,
4046
- this.imgsz,
4047
- this.imgsz
4048
- ]
4049
- }), 3);
4050
- return decodeDetectOutput(preds.data, preds.dims[2] ?? YOLO_ANCHORS_FOR_SIZE[this.imgsz] ?? 8400, YOLO_MODELS[key].nc, this.decodeOptions(image, params));
4051
- }
4052
- async detectWorld(image, prompts) {
4053
- const activePrompts = prompts ?? this.prompts ?? YOLO_COCO_CLASSES;
4054
- const key = this.modelKeyFor("world");
4055
- const session = await this.session(key);
4056
- const { tensor, params } = this.prepareInput(image, key);
4057
- const txtFeats = new Float32Array(activePrompts.length * 512);
4058
- for (let c = 0; c < activePrompts.length; c++) txtFeats.set(encodePromptEmbedding(activePrompts[c]), c * 512);
4059
- const preds = pickByDims(await session.run([{
4060
- name: "images",
4061
- data: tensor,
4062
- dims: [
4063
- 1,
4064
- 3,
4065
- this.imgsz,
4066
- this.imgsz
4067
- ]
4068
- }, {
4069
- name: "txt_feats",
4070
- data: txtFeats,
4071
- dims: [
4072
- 1,
4073
- activePrompts.length,
4074
- 512
4075
- ]
4076
- }]), 3);
4077
- return decodeDetectOutput(preds.data, preds.dims[2] ?? YOLO_ANCHORS_FOR_SIZE[this.imgsz] ?? 8400, activePrompts.length, {
4078
- ...this.decodeOptions(image, params, "world"),
4079
- classNames: activePrompts
4080
- });
3797
+ const pass = await this.instancePass(image);
3798
+ return decodeRtmdetIns({
3799
+ ...pass.outputs,
3800
+ masks: void 0
3801
+ }, this.instanceDecodeOptions(image, pass.transform)).detections;
4081
3802
  }
3803
+ /** COCO-80 boxes plus a frame-aligned soft mask per instance. */
4082
3804
  async segment(image) {
4083
- if (this.customModel && this.customModel.task === "segment") return this.segmentCustom(image);
4084
- const key = this.modelKeyFor("segment");
4085
- const session = await this.session(key);
4086
- const { tensor, params } = this.prepareInput(image, key);
4087
- const outputs = await session.run({
4088
- name: "images",
4089
- data: tensor,
4090
- dims: [
4091
- 1,
4092
- 3,
4093
- this.imgsz,
4094
- this.imgsz
4095
- ]
4096
- });
4097
- const preds = pickByDims(outputs, 3);
4098
- const proto = pickByDims(outputs, 4);
4099
- const protoDim = proto.dims[2] ?? proto.dims[1] ?? 160;
4100
- return decodeSegmentOutput(preds.data, proto.data, preds.dims[2] ?? YOLO_ANCHORS_FOR_SIZE[this.imgsz] ?? 8400, YOLO_MODELS[key].nc, protoDim, this.decodeOptions(image, params));
3805
+ const pass = await this.instancePass(image);
3806
+ return decodeRtmdetIns(pass.outputs, this.instanceDecodeOptions(image, pass.transform));
4101
3807
  }
3808
+ /** COCO-17 keypoints per person (source pixels), highest score first. */
4102
3809
  async pose(image) {
4103
- if (this.customModel && this.customModel.task === "pose") return this.poseCustom(image);
4104
3810
  const key = this.modelKeyFor("pose");
4105
- const session = await this.session(key);
4106
- const { tensor, params } = this.prepareInput(image, key);
4107
- const preds = pickByDims(await session.run({
4108
- name: "images",
4109
- data: tensor,
4110
- dims: [
4111
- 1,
4112
- 3,
4113
- this.imgsz,
4114
- this.imgsz
4115
- ]
4116
- }), 3);
4117
- return decodePoseOutput(preds.data, preds.dims[2] ?? YOLO_ANCHORS_FOR_SIZE[this.imgsz] ?? 8400, {
3811
+ const { outputs, transform } = await this.infer(key, image);
3812
+ const dets = output(outputs, "dets", key);
3813
+ const keypoints = output(outputs, "keypoints", key);
3814
+ return decodeRtmo({
3815
+ dets: dets.data,
3816
+ keypoints: keypoints.data,
3817
+ count: dets.dims[1] ?? 0,
3818
+ keypointStride: keypoints.dims[3] ?? 3
3819
+ }, {
4118
3820
  confidence: this.confidence,
4119
- params,
3821
+ transform,
4120
3822
  sourceWidth: image.width,
4121
3823
  sourceHeight: image.height
4122
3824
  });
4123
3825
  }
4124
- async detectObb(image) {
4125
- if (this.customModel && this.customModel.task === "obb") return this.obbCustom(image);
4126
- const key = this.modelKeyFor("obb");
4127
- const session = await this.session(key);
4128
- const modelImgsz = YOLO_MODELS[key].imgsz;
4129
- const { tensor, params } = this.prepareInput(image, key);
4130
- const preds = pickByDims(await session.run({
4131
- name: "images",
4132
- data: tensor,
4133
- dims: [
4134
- 1,
4135
- 3,
4136
- modelImgsz,
4137
- modelImgsz
4138
- ]
4139
- }), 3);
4140
- return decodeObbOutput(preds.data, preds.dims[2] ?? YOLO_ANCHORS_FOR_SIZE[modelImgsz] ?? 21504, YOLO_MODELS[key].nc, this.decodeOptions(image, params, "obb"));
4141
- }
4142
- async classify(image) {
4143
- if (this.customModel && this.customModel.task === "classify") return this.classifyCustom(image);
4144
- const key = this.modelKeyFor("classify");
4145
- const session = await this.session(key);
4146
- const modelImgsz = YOLO_MODELS[key].imgsz;
4147
- const { tensor } = letterboxToTensor(image, modelImgsz);
4148
- return decodeClassifyOutput(pickByDims(await session.run({
4149
- name: "images",
4150
- data: tensor,
4151
- dims: [
4152
- 1,
4153
- 3,
4154
- modelImgsz,
4155
- modelImgsz
4156
- ]
4157
- }), 2).data, YOLO_MODELS[key].nc);
3826
+ /** Person-vs-background alpha for the whole frame (Selfie Segmenter). */
3827
+ async matte(image) {
3828
+ const key = this.modelKeyFor("matte", image.width / image.height);
3829
+ const [w, h] = VISION_MODELS[key].input;
3830
+ const { outputs } = await this.infer(key, image);
3831
+ return decodeSelfie(output(outputs, "alphas", key).data, w, h, {
3832
+ sourceWidth: image.width,
3833
+ sourceHeight: image.height,
3834
+ maskThreshold: this.maskThreshold,
3835
+ ...this.featherRadius !== void 0 ? { featherRadius: this.featherRadius } : {}
3836
+ });
4158
3837
  }
4159
3838
  close() {
4160
3839
  for (const promise of this._sessions.values()) promise.then((session) => session.release()).catch(() => void 0);
4161
3840
  this._sessions.clear();
4162
- if (this._customSessionPromise) {
4163
- this._customSessionPromise.then((session) => session.release()).catch(() => void 0);
4164
- this._customSessionPromise = void 0;
4165
- }
3841
+ this._instancePass = void 0;
4166
3842
  this._store.clearMemoryCache();
4167
3843
  }
4168
- async detectCustom(image) {
4169
- if (!this.customModel) throw new Error("No custom model configured");
4170
- const session = await this.customSession();
4171
- const imgsz = this.customModel.imgsz ?? this.imgsz;
4172
- const scratch = new Float32Array(3 * imgsz * imgsz);
4173
- const { tensor, params } = letterboxToTensor(normalizeImage(image), imgsz, scratch);
4174
- const preds = pickByDims(await session.run({
4175
- name: "images",
4176
- data: tensor,
4177
- dims: [
4178
- 1,
4179
- 3,
4180
- imgsz,
4181
- imgsz
4182
- ]
4183
- }), 3);
4184
- const anchors = preds.dims[2] ?? YOLO_ANCHORS_FOR_SIZE[imgsz] ?? 8400;
4185
- return decodeDetectOutput(preds.data, anchors, this.customModel.classes.length, {
4186
- ...this.decodeOptions(image, params),
4187
- classNames: this.customModel.classes
4188
- });
4189
- }
4190
- async segmentCustom(image) {
4191
- if (!this.customModel) throw new Error("No custom model configured");
4192
- const session = await this.customSession();
4193
- const imgsz = this.customModel.imgsz ?? this.imgsz;
4194
- const scratch = new Float32Array(3 * imgsz * imgsz);
4195
- const { tensor, params } = letterboxToTensor(normalizeImage(image), imgsz, scratch);
4196
- const outputs = await session.run({
4197
- name: "images",
4198
- data: tensor,
4199
- dims: [
4200
- 1,
4201
- 3,
4202
- imgsz,
4203
- imgsz
4204
- ]
3844
+ instancePass(image) {
3845
+ if (this._instancePass?.frame === image.data) return this._instancePass.result;
3846
+ const key = this.modelKeyFor("segment");
3847
+ const result = this.infer(key, image).then(({ outputs, transform }) => {
3848
+ const dets = output(outputs, "dets", key);
3849
+ const masks = output(outputs, "masks", key);
3850
+ return {
3851
+ transform,
3852
+ outputs: {
3853
+ dets: dets.data,
3854
+ labels: output(outputs, "labels", key).data,
3855
+ masks: masks.data,
3856
+ count: dets.dims[1] ?? 0,
3857
+ maskHeight: masks.dims[masks.dims.length - 2] ?? 0,
3858
+ maskWidth: masks.dims[masks.dims.length - 1] ?? 0
3859
+ }
3860
+ };
4205
3861
  });
4206
- const preds = pickByDims(outputs, 3);
4207
- const proto = pickByDims(outputs, 4);
4208
- const protoDim = proto.dims[2] ?? proto.dims[1] ?? 160;
4209
- return decodeSegmentOutput(preds.data, proto.data, preds.dims[2] ?? YOLO_ANCHORS_FOR_SIZE[imgsz] ?? 8400, this.customModel.classes.length, protoDim, {
4210
- ...this.decodeOptions(image, params),
4211
- classNames: this.customModel.classes
3862
+ this._instancePass = {
3863
+ frame: image.data,
3864
+ result
3865
+ };
3866
+ result.catch(() => {
3867
+ if (this._instancePass?.result === result) this._instancePass = void 0;
4212
3868
  });
3869
+ return result;
4213
3870
  }
4214
- async poseCustom(image) {
4215
- if (!this.customModel) throw new Error("No custom model configured");
4216
- const session = await this.customSession();
4217
- const imgsz = this.customModel.imgsz ?? this.imgsz;
4218
- const scratch = new Float32Array(3 * imgsz * imgsz);
4219
- const { tensor, params } = letterboxToTensor(normalizeImage(image), imgsz, scratch);
4220
- const preds = pickByDims(await session.run({
4221
- name: "images",
4222
- data: tensor,
4223
- dims: [
4224
- 1,
4225
- 3,
4226
- imgsz,
4227
- imgsz
4228
- ]
4229
- }), 3);
4230
- return decodePoseOutput(preds.data, preds.dims[2] ?? YOLO_ANCHORS_FOR_SIZE[imgsz] ?? 8400, {
3871
+ instanceDecodeOptions(image, transform) {
3872
+ return {
4231
3873
  confidence: this.confidence,
4232
- params,
3874
+ classes: this.classes,
3875
+ classNames: COCO_CLASSES,
3876
+ transform,
4233
3877
  sourceWidth: image.width,
4234
- sourceHeight: image.height
4235
- });
4236
- }
4237
- async obbCustom(image) {
4238
- if (!this.customModel) throw new Error("No custom model configured");
4239
- const session = await this.customSession();
4240
- const imgsz = this.customModel.imgsz ?? this.imgsz;
4241
- const scratch = new Float32Array(3 * imgsz * imgsz);
4242
- const { tensor, params } = letterboxToTensor(normalizeImage(image), imgsz, scratch);
4243
- const preds = pickByDims(await session.run({
4244
- name: "images",
4245
- data: tensor,
4246
- dims: [
4247
- 1,
4248
- 3,
4249
- imgsz,
4250
- imgsz
4251
- ]
4252
- }), 3);
4253
- return decodeObbOutput(preds.data, preds.dims[2] ?? YOLO_ANCHORS_FOR_SIZE[imgsz] ?? 21504, this.customModel.classes.length, {
4254
- ...this.decodeOptions(image, params, "obb"),
4255
- classNames: this.customModel.classes
4256
- });
3878
+ sourceHeight: image.height,
3879
+ maskThreshold: this.maskThreshold,
3880
+ ...this.featherRadius !== void 0 ? { featherRadius: this.featherRadius } : {}
3881
+ };
4257
3882
  }
4258
- async classifyCustom(image) {
4259
- if (!this.customModel) throw new Error("No custom model configured");
4260
- const session = await this.customSession();
4261
- const imgsz = this.customModel.imgsz ?? 224;
4262
- const { tensor } = letterboxToTensor(image, imgsz);
4263
- return decodeClassifyOutput(pickByDims(await session.run({
4264
- name: "images",
4265
- data: tensor,
4266
- dims: [
4267
- 1,
4268
- 3,
4269
- imgsz,
4270
- imgsz
4271
- ]
4272
- }), 2).data, this.customModel.classes.length);
4273
- }
4274
- customSession() {
4275
- if (!this.customModel) throw new Error("No customModel specified in YoloVisionRunner options");
4276
- if (!this._customSessionPromise) {
4277
- const cm = this.customModel;
4278
- const imgsz = cm.imgsz ?? this.imgsz;
4279
- this._customSessionPromise = this.loadCustomModelBytes(cm).then((bytes) => this._provider.createSession(bytes, { imgsz }));
3883
+ async infer(key, image) {
3884
+ const session = await this.session(key);
3885
+ const desc = VISION_MODELS[key];
3886
+ const spec = preprocessSpec(desc.family, desc.input);
3887
+ const size = 3 * spec.width * spec.height;
3888
+ let scratch = this._tensorScratch.get(key);
3889
+ if (!scratch || scratch.length < size) {
3890
+ scratch = new Float32Array(size);
3891
+ this._tensorScratch.set(key, scratch);
4280
3892
  }
4281
- return this._customSessionPromise;
4282
- }
4283
- async loadCustomModelBytes(cm) {
4284
- if (cm.path) return new Uint8Array(await readFile(cm.path));
4285
- if (cm.url) return this._store.fetchDirect(cm.url);
4286
- throw new Error("CustomModelConfig must specify either path or url");
3893
+ const { tensor, transform } = imageToTensor(image, spec, scratch);
3894
+ return {
3895
+ outputs: await session.run({
3896
+ name: "input",
3897
+ data: tensor,
3898
+ dims: [
3899
+ 1,
3900
+ 3,
3901
+ spec.height,
3902
+ spec.width
3903
+ ]
3904
+ }),
3905
+ transform
3906
+ };
4287
3907
  }
4288
3908
  session(key) {
4289
3909
  let promise = this._sessions.get(key);
4290
3910
  if (!promise) {
4291
- const modelImgsz = YOLO_MODELS[key].imgsz ?? this.imgsz;
4292
- promise = this._store.ensure(key).then((bytes) => this._provider.createSession(bytes, { imgsz: modelImgsz })).catch((err) => {
3911
+ promise = this._store.ensure(key).then((bytes) => this._provider.createSession(bytes)).catch((err) => {
4293
3912
  this._sessions.delete(key);
4294
3913
  this._store.evict(key);
4295
3914
  throw err;
@@ -4298,59 +3917,103 @@ var YoloVisionRunner = class YoloVisionRunner$1 {
4298
3917
  }
4299
3918
  return promise;
4300
3919
  }
4301
- prepareInput(image, key) {
4302
- const imgsz = YOLO_MODELS[key].imgsz ?? this.imgsz;
4303
- let scratch = this._tensorScratch.get(key);
4304
- if (!scratch || scratch.length < 3 * imgsz * imgsz) {
4305
- scratch = new Float32Array(3 * imgsz * imgsz);
4306
- this._tensorScratch.set(key, scratch);
3920
+ };
3921
+ function output(outputs, name, key) {
3922
+ const tensor = outputs[name];
3923
+ if (!tensor) throw new Error(`Vision model '${key}' returned no '${name}' output (got: ${Object.keys(outputs).join(", ") || "none"}). Expected the ${VISION_MODELS[key].source}.`);
3924
+ return tensor;
3925
+ }
3926
+ /**
3927
+ * The frame's primary instance: the largest person when present, else the most confident
3928
+ * instance (lowest `detectionIndex` — detections are score-sorted). Size alone is a poor
3929
+ * signal without a person: the largest instance is usually a backdrop ("dining table").
3930
+ */
3931
+ function selectSubjectMask(masks) {
3932
+ let best;
3933
+ for (const m of masks) {
3934
+ if (!best) {
3935
+ best = m;
3936
+ continue;
4307
3937
  }
4308
- return letterboxToTensor(normalizeImage(image), imgsz, scratch);
3938
+ const mIsPerson = isPerson(m);
3939
+ if (mIsPerson !== isPerson(best)) {
3940
+ if (mIsPerson) best = m;
3941
+ } else if (mIsPerson ? m.area > best.area : m.detectionIndex < best.detectionIndex) best = m;
4309
3942
  }
4310
- decodeOptions(image, params, task = "detect") {
4311
- return {
4312
- confidence: this.confidence,
4313
- iouThreshold: this.iouThreshold,
4314
- classes: this.classes,
4315
- classNames: task === "obb" ? YOLO_DOTA_CLASSES : YOLO_COCO_CLASSES,
4316
- params,
4317
- sourceWidth: image.width,
4318
- sourceHeight: image.height,
4319
- maskThreshold: this.maskThreshold,
4320
- featherRadius: this.featherRadius
4321
- };
3943
+ return best;
3944
+ }
3945
+ function isPerson(mask) {
3946
+ return mask.category.toLowerCase() === "person";
3947
+ }
3948
+ /** Tight pixel bounds of a mask's non-zero alpha, or null when empty. */
3949
+ function maskBounds(mask) {
3950
+ const { mask: data, width, height } = mask;
3951
+ let x0 = width;
3952
+ let y0 = height;
3953
+ let x1 = -1;
3954
+ let y1 = -1;
3955
+ for (let y = 0; y < height; y++) {
3956
+ const row = y * width;
3957
+ for (let x = 0; x < width; x++) {
3958
+ if (data[row + x] === 0) continue;
3959
+ if (x < x0) x0 = x;
3960
+ if (x > x1) x1 = x;
3961
+ if (y < y0) y0 = y;
3962
+ if (y > y1) y1 = y;
3963
+ }
4322
3964
  }
4323
- };
4324
- function normalizeImage(image) {
4325
- if (image.data instanceof Uint8ClampedArray) return image;
4326
- return image;
3965
+ return x1 < 0 ? null : {
3966
+ x0,
3967
+ y0,
3968
+ x1,
3969
+ y1
3970
+ };
4327
3971
  }
4328
- function pickByDims(outputs, rank) {
4329
- const hit = Object.values(outputs).find((t) => t.dims.length === rank);
4330
- if (!hit) throw new Error(`YOLO model returned no ${rank}D output — got [${Object.values(outputs).map((t) => t.dims.join("x")).join(", ")}]. Check the exported model task head.`);
4331
- return hit;
3972
+ /**
3973
+ * Builds the full subject silhouette: the primary instance plus every comparably sized
3974
+ * instance whose bounds sit mostly inside or across it (`overlap` = intersection / smaller
3975
+ * box area; `maxGrowth` caps a part's box area relative to the subject's).
3976
+ *
3977
+ * COCO has no "clothing" class, so a flowing dress, a held guitar or a ridden bike comes back
3978
+ * as its own instance (often mislabeled) — merging them keeps the whole figure in the cutout.
3979
+ * The size cap keeps containers out: a small figure inside a tunnel or window detected as a
3980
+ * huge "clock" must not drag the whole frame into the subject.
3981
+ */
3982
+ function mergeSubjectMask(masks, overlap = .5, maxGrowth = 2) {
3983
+ const subject = selectSubjectMask(masks);
3984
+ if (!subject) return void 0;
3985
+ const sb = maskBounds(subject);
3986
+ if (!sb) return subject;
3987
+ const subjectArea = boundsArea(sb);
3988
+ const parts = masks.filter((m) => {
3989
+ if (m === subject) return false;
3990
+ const b = maskBounds(m);
3991
+ return b !== null && boundsArea(b) <= subjectArea * maxGrowth && boundsOverlap(sb, b) >= overlap;
3992
+ });
3993
+ if (parts.length === 0) return subject;
3994
+ const merged = subject.mask.slice();
3995
+ for (const part of parts) {
3996
+ const data = part.mask;
3997
+ for (let i = 0; i < merged.length; i++) if (data[i] > merged[i]) merged[i] = data[i];
3998
+ }
3999
+ let area = 0;
4000
+ for (let i = 0; i < merged.length; i++) area += merged[i];
4001
+ area = Math.round(area / 255);
4002
+ return {
4003
+ ...subject,
4004
+ mask: merged,
4005
+ area,
4006
+ coverage: area / (subject.width * subject.height)
4007
+ };
4332
4008
  }
4333
- function hashPromptString(str) {
4334
- let h = 2166136261;
4335
- for (let i = 0; i < str.length; i++) h = Math.imul(h ^ str.charCodeAt(i), 16777619) >>> 0;
4336
- return h;
4009
+ function boundsArea(b) {
4010
+ return (b.x1 - b.x0 + 1) * (b.y1 - b.y0 + 1);
4337
4011
  }
4338
- function encodePromptEmbedding(prompt) {
4339
- let seed = hashPromptString(prompt);
4340
- const vec = new Float32Array(512);
4341
- let norm = 0;
4342
- for (let i = 0; i < 512; i++) {
4343
- seed = seed + 1831565813 >>> 0;
4344
- let t = seed;
4345
- t = Math.imul(t ^ t >>> 15, t | 1);
4346
- t ^= t + Math.imul(t ^ t >>> 7, t | 61);
4347
- const val = ((t ^ t >>> 14) >>> 0) / 4294967296 - .5;
4348
- vec[i] = val;
4349
- norm += val * val;
4350
- }
4351
- const invNorm = 1 / (Math.sqrt(norm) || 1);
4352
- for (let i = 0; i < 512; i++) vec[i] *= invNorm;
4353
- return vec;
4012
+ function boundsOverlap(a, b) {
4013
+ const iw = Math.min(a.x1, b.x1) - Math.max(a.x0, b.x0) + 1;
4014
+ const ih = Math.min(a.y1, b.y1) - Math.max(a.y0, b.y0) + 1;
4015
+ if (iw <= 0 || ih <= 0) return 0;
4016
+ return iw * ih / Math.min(boundsArea(a), boundsArea(b));
4354
4017
  }
4355
4018
  var SpatialLandmarkTransformer = class {
4356
4019
  width;
@@ -4446,7 +4109,7 @@ function matchPoseToTracks(people, trackedObjects, minIoU = .15) {
4446
4109
  const pBox = getPersonBoundingBox(person);
4447
4110
  const isNormalized = pBox.width <= 1.05 && pBox.height <= 1.05;
4448
4111
  for (const trk of personTracks) {
4449
- const iou = computeIoU$1(pBox, isNormalized && trk.boundingBox.normalizedWidth > 0 ? {
4112
+ const iou = computeIoU(pBox, isNormalized && trk.boundingBox.normalizedWidth > 0 ? {
4450
4113
  originX: trk.boundingBox.normalizedX,
4451
4114
  originY: trk.boundingBox.normalizedY,
4452
4115
  width: trk.boundingBox.normalizedWidth,
@@ -4479,13 +4142,12 @@ function matchPoseToTracks(people, trackedObjects, minIoU = .15) {
4479
4142
  } : person;
4480
4143
  });
4481
4144
  }
4482
- var YoloVisionBundle = class {
4145
+ var VisionBundle = class {
4483
4146
  objects;
4484
4147
  poseLandmarks;
4485
4148
  masks;
4486
4149
  segmentation;
4487
4150
  classes;
4488
- classification;
4489
4151
  poseLandmarksTensor;
4490
4152
  objectsTensor;
4491
4153
  masksTensor;
@@ -4500,8 +4162,7 @@ var YoloVisionBundle = class {
4500
4162
  _poseCache = /* @__PURE__ */ new Map();
4501
4163
  _objectCache = /* @__PURE__ */ new Map();
4502
4164
  _maskCache = /* @__PURE__ */ new Map();
4503
- _classifyCache = /* @__PURE__ */ new Map();
4504
- _obbCache = /* @__PURE__ */ new Map();
4165
+ _matteCache = /* @__PURE__ */ new Map();
4505
4166
  _poseLmCache = /* @__PURE__ */ new Map();
4506
4167
  _trackCache = /* @__PURE__ */ new Map();
4507
4168
  _categoryCache = /* @__PURE__ */ new Map();
@@ -4559,30 +4220,30 @@ var YoloVisionBundle = class {
4559
4220
  const getOrCreatePoseLm = (index) => {
4560
4221
  let sig = this._poseLmCache.get(index);
4561
4222
  if (!sig) {
4562
- sig = createPoseCoord(index, POSE_KEYPOINT_NAMES[index] ?? `kpt_${index}`);
4223
+ sig = createPoseCoord(index, COCO17_KEYPOINT_NAMES[index] ?? `kpt_${index}`);
4563
4224
  this._poseLmCache.set(index, sig);
4564
4225
  }
4565
4226
  return sig;
4566
4227
  };
4567
4228
  const poseNamed = {
4568
- nose: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.NOSE),
4569
- leftEye: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.LEFT_EYE),
4570
- rightEye: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.RIGHT_EYE),
4571
- leftEar: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.LEFT_EAR),
4572
- rightEar: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.RIGHT_EAR),
4573
- shoulder: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.LEFT_SHOULDER),
4574
- leftShoulder: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.LEFT_SHOULDER),
4575
- rightShoulder: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.RIGHT_SHOULDER),
4576
- leftElbow: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.LEFT_ELBOW),
4577
- rightElbow: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.RIGHT_ELBOW),
4578
- leftWrist: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.LEFT_WRIST),
4579
- rightWrist: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.RIGHT_WRIST),
4580
- leftHip: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.LEFT_HIP),
4581
- rightHip: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.RIGHT_HIP),
4582
- leftKnee: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.LEFT_KNEE),
4583
- rightKnee: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.RIGHT_KNEE),
4584
- leftAnkle: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.LEFT_ANKLE),
4585
- rightAnkle: getOrCreatePoseLm(POSE_LANDMARKS_YOLO.RIGHT_ANKLE),
4229
+ nose: getOrCreatePoseLm(COCO17_KEYPOINTS.NOSE),
4230
+ leftEye: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_EYE),
4231
+ rightEye: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_EYE),
4232
+ leftEar: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_EAR),
4233
+ rightEar: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_EAR),
4234
+ shoulder: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_SHOULDER),
4235
+ leftShoulder: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_SHOULDER),
4236
+ rightShoulder: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_SHOULDER),
4237
+ leftElbow: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_ELBOW),
4238
+ rightElbow: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_ELBOW),
4239
+ leftWrist: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_WRIST),
4240
+ rightWrist: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_WRIST),
4241
+ leftHip: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_HIP),
4242
+ rightHip: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_HIP),
4243
+ leftKnee: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_KNEE),
4244
+ rightKnee: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_KNEE),
4245
+ leftAnkle: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_ANKLE),
4246
+ rightAnkle: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_ANKLE),
4586
4247
  get: getOrCreatePoseLm
4587
4248
  };
4588
4249
  this.poseLandmarks = new Proxy(poseNamed, { get(target, prop, receiver) {
@@ -4658,7 +4319,7 @@ var YoloVisionBundle = class {
4658
4319
  let bestIoU = .1;
4659
4320
  for (const p of poseRes.people) {
4660
4321
  const pBox = getPersonBoundingBox(p);
4661
- const iou = computeIoU$1(pBox, pBox.width <= 1.05 && pBox.height <= 1.05 && trk.boundingBox.normalizedWidth > 0 ? {
4322
+ const iou = computeIoU(pBox, pBox.width <= 1.05 && pBox.height <= 1.05 && trk.boundingBox.normalizedWidth > 0 ? {
4662
4323
  originX: trk.boundingBox.normalizedX,
4663
4324
  originY: trk.boundingBox.normalizedY,
4664
4325
  width: trk.boundingBox.normalizedWidth,
@@ -4717,7 +4378,7 @@ var YoloVisionBundle = class {
4717
4378
  const getOrCreateCoord = (idx) => {
4718
4379
  let c = coordsCache.get(idx);
4719
4380
  if (!c) {
4720
- c = createCoord(idx, POSE_KEYPOINT_NAMES[idx] ?? `kpt_${idx}`);
4381
+ c = createCoord(idx, COCO17_KEYPOINT_NAMES[idx] ?? `kpt_${idx}`);
4721
4382
  coordsCache.set(idx, c);
4722
4383
  }
4723
4384
  return c;
@@ -4731,16 +4392,16 @@ var YoloVisionBundle = class {
4731
4392
  const prev = resolvePerson(Math.max(0, f - 1));
4732
4393
  const currKp = curr.keypoints;
4733
4394
  const prevKp = prev?.keypoints;
4734
- const rwCurr = currKp[POSE_LANDMARKS_YOLO.RIGHT_WRIST];
4735
- const rwPrev = prevKp?.[POSE_LANDMARKS_YOLO.RIGHT_WRIST];
4395
+ const rwCurr = currKp[COCO17_KEYPOINTS.RIGHT_WRIST];
4396
+ const rwPrev = prevKp?.[COCO17_KEYPOINTS.RIGHT_WRIST];
4736
4397
  let rwSpeed = 0;
4737
4398
  if (rwCurr && rwPrev && rwCurr.visibility > .1 && rwPrev.visibility > .1) {
4738
4399
  const dx = (rwCurr.x - rwPrev.x) * this._width;
4739
4400
  const dy = (rwCurr.y - rwPrev.y) * this._height;
4740
4401
  rwSpeed = Math.sqrt(dx * dx + dy * dy) * this._fps;
4741
4402
  }
4742
- const lwCurr = currKp[POSE_LANDMARKS_YOLO.LEFT_WRIST];
4743
- const lwPrev = prevKp?.[POSE_LANDMARKS_YOLO.LEFT_WRIST];
4403
+ const lwCurr = currKp[COCO17_KEYPOINTS.LEFT_WRIST];
4404
+ const lwPrev = prevKp?.[COCO17_KEYPOINTS.LEFT_WRIST];
4744
4405
  let lwSpeed = 0;
4745
4406
  if (lwCurr && lwPrev && lwCurr.visibility > .1 && lwPrev.visibility > .1) {
4746
4407
  const dx = (lwCurr.x - lwPrev.x) * this._width;
@@ -4756,10 +4417,10 @@ var YoloVisionBundle = class {
4756
4417
  const p = resolvePerson(ctx.frame);
4757
4418
  if (!p) return 0;
4758
4419
  const k = p.keypoints;
4759
- const lw = k[POSE_LANDMARKS_YOLO.LEFT_WRIST];
4760
- const ls = k[POSE_LANDMARKS_YOLO.LEFT_SHOULDER];
4761
- const rw = k[POSE_LANDMARKS_YOLO.RIGHT_WRIST];
4762
- const rs = k[POSE_LANDMARKS_YOLO.RIGHT_SHOULDER];
4420
+ const lw = k[COCO17_KEYPOINTS.LEFT_WRIST];
4421
+ const ls = k[COCO17_KEYPOINTS.LEFT_SHOULDER];
4422
+ const rw = k[COCO17_KEYPOINTS.RIGHT_WRIST];
4423
+ const rs = k[COCO17_KEYPOINTS.RIGHT_SHOULDER];
4763
4424
  const lRaised = lw && ls && lw.visibility > .1 && ls.visibility > .1 && lw.y < ls.y;
4764
4425
  const rRaised = rw && rs && rw.visibility > .1 && rs.visibility > .1 && rw.y < rs.y;
4765
4426
  return lRaised || rRaised ? 1 : 0;
@@ -4771,10 +4432,10 @@ var YoloVisionBundle = class {
4771
4432
  const p = resolvePerson(ctx.frame);
4772
4433
  if (!p) return 0;
4773
4434
  const k = p.keypoints;
4774
- const ls = k[POSE_LANDMARKS_YOLO.LEFT_SHOULDER];
4775
- const rs = k[POSE_LANDMARKS_YOLO.RIGHT_SHOULDER];
4776
- const lh = k[POSE_LANDMARKS_YOLO.LEFT_HIP];
4777
- const rh = k[POSE_LANDMARKS_YOLO.RIGHT_HIP];
4435
+ const ls = k[COCO17_KEYPOINTS.LEFT_SHOULDER];
4436
+ const rs = k[COCO17_KEYPOINTS.RIGHT_SHOULDER];
4437
+ const lh = k[COCO17_KEYPOINTS.LEFT_HIP];
4438
+ const rh = k[COCO17_KEYPOINTS.RIGHT_HIP];
4778
4439
  if (!ls || !rs || !lh || !rh) return 0;
4779
4440
  const sx = (ls.x + rs.x) / 2 * this._width;
4780
4441
  const sy = (ls.y + rs.y) / 2 * this._height;
@@ -4788,24 +4449,24 @@ var YoloVisionBundle = class {
4788
4449
  fps: this._fps,
4789
4450
  label: `${label}_bodyTiltAngle`
4790
4451
  }),
4791
- nose: getOrCreateCoord(POSE_LANDMARKS_YOLO.NOSE),
4792
- leftEye: getOrCreateCoord(POSE_LANDMARKS_YOLO.LEFT_EYE),
4793
- rightEye: getOrCreateCoord(POSE_LANDMARKS_YOLO.RIGHT_EYE),
4794
- leftEar: getOrCreateCoord(POSE_LANDMARKS_YOLO.LEFT_EAR),
4795
- rightEar: getOrCreateCoord(POSE_LANDMARKS_YOLO.RIGHT_EAR),
4796
- shoulder: getOrCreateCoord(POSE_LANDMARKS_YOLO.LEFT_SHOULDER),
4797
- leftShoulder: getOrCreateCoord(POSE_LANDMARKS_YOLO.LEFT_SHOULDER),
4798
- rightShoulder: getOrCreateCoord(POSE_LANDMARKS_YOLO.RIGHT_SHOULDER),
4799
- leftElbow: getOrCreateCoord(POSE_LANDMARKS_YOLO.LEFT_ELBOW),
4800
- rightElbow: getOrCreateCoord(POSE_LANDMARKS_YOLO.RIGHT_ELBOW),
4801
- leftWrist: getOrCreateCoord(POSE_LANDMARKS_YOLO.LEFT_WRIST),
4802
- rightWrist: getOrCreateCoord(POSE_LANDMARKS_YOLO.RIGHT_WRIST),
4803
- leftHip: getOrCreateCoord(POSE_LANDMARKS_YOLO.LEFT_HIP),
4804
- rightHip: getOrCreateCoord(POSE_LANDMARKS_YOLO.RIGHT_HIP),
4805
- leftKnee: getOrCreateCoord(POSE_LANDMARKS_YOLO.LEFT_KNEE),
4806
- rightKnee: getOrCreateCoord(POSE_LANDMARKS_YOLO.RIGHT_KNEE),
4807
- leftAnkle: getOrCreateCoord(POSE_LANDMARKS_YOLO.LEFT_ANKLE),
4808
- rightAnkle: getOrCreateCoord(POSE_LANDMARKS_YOLO.RIGHT_ANKLE),
4452
+ nose: getOrCreateCoord(COCO17_KEYPOINTS.NOSE),
4453
+ leftEye: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_EYE),
4454
+ rightEye: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_EYE),
4455
+ leftEar: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_EAR),
4456
+ rightEar: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_EAR),
4457
+ shoulder: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_SHOULDER),
4458
+ leftShoulder: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_SHOULDER),
4459
+ rightShoulder: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_SHOULDER),
4460
+ leftElbow: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_ELBOW),
4461
+ rightElbow: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_ELBOW),
4462
+ leftWrist: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_WRIST),
4463
+ rightWrist: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_WRIST),
4464
+ leftHip: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_HIP),
4465
+ rightHip: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_HIP),
4466
+ leftKnee: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_KNEE),
4467
+ rightKnee: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_KNEE),
4468
+ leftAnkle: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_ANKLE),
4469
+ rightAnkle: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_ANKLE),
4809
4470
  get: getOrCreateCoord
4810
4471
  };
4811
4472
  return new Proxy(namedPose, { get(target, prop, receiver) {
@@ -4879,26 +4540,6 @@ var YoloVisionBundle = class {
4879
4540
  }, {
4880
4541
  fps: this._fps,
4881
4542
  label: `${label}_area`
4882
- }),
4883
- angle: programmaticSignal$1((ctx) => {
4884
- const b = getBox(ctx.frame);
4885
- const tcx = b.originX + b.width / 2;
4886
- const tcy = b.originY + b.height / 2;
4887
- let bestAngle = 0;
4888
- let bestDist = Infinity;
4889
- for (const d of self.getObbResult(ctx.frame).detections) {
4890
- const cx = d.boundingBox.originX + d.boundingBox.width / 2;
4891
- const cy = d.boundingBox.originY + d.boundingBox.height / 2;
4892
- const dist = (cx - tcx) * (cx - tcx) + (cy - tcy) * (cy - tcy);
4893
- if (dist < bestDist) {
4894
- bestDist = dist;
4895
- bestAngle = d.boundingBox.angle;
4896
- }
4897
- }
4898
- return bestAngle;
4899
- }, {
4900
- fps: this._fps,
4901
- label: `${label}_angle`
4902
4543
  })
4903
4544
  },
4904
4545
  anchors: {
@@ -5057,21 +4698,7 @@ var YoloVisionBundle = class {
5057
4698
  }
5058
4699
  };
5059
4700
  const maskForTrack = (frame, trackId) => this.getMaskResult(frame).find((m) => m.trackId === trackId);
5060
- const largestMask = (frame) => {
5061
- const masks = this.getMaskResult(frame);
5062
- if (masks.length === 0) return void 0;
5063
- let best = masks[0];
5064
- for (const m of masks) {
5065
- const bestIsPerson = best.category.toLowerCase() === "person";
5066
- const mIsPerson = m.category.toLowerCase() === "person";
5067
- if (mIsPerson && !bestIsPerson) {
5068
- best = m;
5069
- continue;
5070
- }
5071
- if (mIsPerson === bestIsPerson && m.area > best.area) best = m;
5072
- }
5073
- return best;
5074
- };
4701
+ const subjectMask = (frame) => selectSubjectMask(this.getMaskResult(frame));
5075
4702
  const buildMaskTrackSignals = (resolve$1, label) => {
5076
4703
  const boxFor = (frame) => {
5077
4704
  const mask = resolve$1(frame);
@@ -5124,7 +4751,7 @@ var YoloVisionBundle = class {
5124
4751
  }
5125
4752
  return sig;
5126
4753
  },
5127
- subject: buildMaskTrackSignals(largestMask, "subject"),
4754
+ subject: buildMaskTrackSignals(subjectMask, "subject"),
5128
4755
  count: programmaticSignal$1((ctx) => new Set(this.getMaskResult(ctx.frame).map((m) => m.trackId ?? -1)).size, {
5129
4756
  fps: this._fps,
5130
4757
  label: "mask_count"
@@ -5134,24 +4761,17 @@ var YoloVisionBundle = class {
5134
4761
  humanSilhouette: this.masks.subject,
5135
4762
  subject: this.masks.subject,
5136
4763
  instanceMasks: this.masks,
4764
+ matte: {
4765
+ coverage: programmaticSignal$1((ctx) => self.getMatteResult(ctx.frame)?.coverage ?? 0, {
4766
+ fps: this._fps,
4767
+ label: "matte_coverage"
4768
+ }),
4769
+ at: (frame) => self.getMatteResult(frame)
4770
+ },
5137
4771
  get stencilTexture() {
5138
4772
  return self._stencilTexture;
5139
4773
  }
5140
4774
  };
5141
- this.classification = {
5142
- top1: programmaticSignal$1((ctx) => self.getClassifyResult(ctx.frame).top1, {
5143
- fps: this._fps,
5144
- label: "classify_top1"
5145
- }),
5146
- top1Confidence: programmaticSignal$1((ctx) => self.getClassifyResult(ctx.frame).top1Score, {
5147
- fps: this._fps,
5148
- label: "classify_top1_conf"
5149
- }),
5150
- get top5() {
5151
- return self.getClassifyResult(frameSignal.value).top5;
5152
- },
5153
- top5At: (frame) => self.getClassifyResult(frame).top5
5154
- };
5155
4775
  this.poseLandmarksTensor = {
5156
4776
  get: (ctx) => self.getPoseLandmarksTensor(ctx?.frame ?? frameSignal.value),
5157
4777
  get value() {
@@ -5229,27 +4849,16 @@ var YoloVisionBundle = class {
5229
4849
  getMaskResult(frame) {
5230
4850
  return this.latestAtOrBefore(this._maskCache, frame) ?? [];
5231
4851
  }
5232
- setClassifyResult(frame, result) {
5233
- this._classifyCache.set(this.clamp(frame), result);
4852
+ setMatteResult(frame, matte) {
4853
+ this._matteCache.set(this.clamp(frame), matte);
5234
4854
  this.invalidateSignals();
5235
4855
  }
5236
- getClassifyResult(frame) {
5237
- return this.latestAtOrBefore(this._classifyCache, frame) ?? {
5238
- top1: 0,
5239
- top1Score: 0,
5240
- top5: []
5241
- };
5242
- }
5243
- setObbResult(frame, result) {
5244
- this._obbCache.set(this.clamp(frame), result);
5245
- this.invalidateSignals();
5246
- }
5247
- getObbResult(frame) {
5248
- return this.latestAtOrBefore(this._obbCache, frame) ?? { detections: [] };
4856
+ getMatteResult(frame) {
4857
+ return this.latestAtOrBefore(this._matteCache, frame);
5249
4858
  }
5250
4859
  /**
5251
- * Serializable snapshot of one frame's object / class / mask signals (specs/yolov4plan.ts
5252
- * §Phase A). Synchronous — reads only the per-frame caches, never triggers inference.
4860
+ * Serializable snapshot of one frame's object / class / mask signals.
4861
+ * Synchronous — reads only the per-frame caches, never triggers inference.
5253
4862
  */
5254
4863
  summary(frame = frameSignal.value) {
5255
4864
  const f = this.clamp(frame);
@@ -5335,7 +4944,7 @@ var YoloVisionBundle = class {
5335
4944
  getClassHistogramTensor(frame, nc = 80) {
5336
4945
  const data = new Float32Array(nc);
5337
4946
  for (const o of this.getObjectResult(frame).objects) if (o.active) {
5338
- const idx = YOLO_CLASS_INDEX_BY_NAME.get(o.category.toLowerCase());
4947
+ const idx = COCO_CLASS_INDEX_BY_NAME.get(o.category.toLowerCase());
5339
4948
  if (idx !== void 0) data[idx] += 1;
5340
4949
  }
5341
4950
  return {
@@ -5348,107 +4957,7 @@ var YoloVisionBundle = class {
5348
4957
  return Math.max(0, Math.min(this._totalFrames - 1, Math.round(frame)));
5349
4958
  }
5350
4959
  };
5351
- const POSE_KEYPOINT_NAMES = [
5352
- "nose",
5353
- "leftEye",
5354
- "rightEye",
5355
- "leftEar",
5356
- "rightEar",
5357
- "leftShoulder",
5358
- "rightShoulder",
5359
- "leftElbow",
5360
- "rightElbow",
5361
- "leftWrist",
5362
- "rightWrist",
5363
- "leftHip",
5364
- "rightHip",
5365
- "leftKnee",
5366
- "rightKnee",
5367
- "leftAnkle",
5368
- "rightAnkle"
5369
- ];
5370
- const YOLO_CLASS_INDEX_BY_NAME = new Map([
5371
- "person",
5372
- "bicycle",
5373
- "car",
5374
- "motorcycle",
5375
- "airplane",
5376
- "bus",
5377
- "train",
5378
- "truck",
5379
- "boat",
5380
- "traffic light",
5381
- "fire hydrant",
5382
- "stop sign",
5383
- "parking meter",
5384
- "bench",
5385
- "bird",
5386
- "cat",
5387
- "dog",
5388
- "horse",
5389
- "sheep",
5390
- "cow",
5391
- "elephant",
5392
- "bear",
5393
- "zebra",
5394
- "giraffe",
5395
- "backpack",
5396
- "umbrella",
5397
- "handbag",
5398
- "tie",
5399
- "suitcase",
5400
- "frisbee",
5401
- "skis",
5402
- "snowboard",
5403
- "sports ball",
5404
- "kite",
5405
- "baseball bat",
5406
- "baseball glove",
5407
- "skateboard",
5408
- "surfboard",
5409
- "tennis racket",
5410
- "bottle",
5411
- "wine glass",
5412
- "cup",
5413
- "fork",
5414
- "knife",
5415
- "spoon",
5416
- "bowl",
5417
- "banana",
5418
- "apple",
5419
- "sandwich",
5420
- "orange",
5421
- "broccoli",
5422
- "carrot",
5423
- "hot dog",
5424
- "pizza",
5425
- "donut",
5426
- "cake",
5427
- "chair",
5428
- "couch",
5429
- "potted plant",
5430
- "bed",
5431
- "dining table",
5432
- "toilet",
5433
- "tv",
5434
- "laptop",
5435
- "mouse",
5436
- "remote",
5437
- "keyboard",
5438
- "cell phone",
5439
- "microwave",
5440
- "oven",
5441
- "toaster",
5442
- "sink",
5443
- "refrigerator",
5444
- "book",
5445
- "clock",
5446
- "vase",
5447
- "scissors",
5448
- "teddy bear",
5449
- "hair drier",
5450
- "toothbrush"
5451
- ].map((name, index) => [name, index]));
4960
+ const COCO_CLASS_INDEX_BY_NAME = new Map(COCO_CLASSES.map((name, index) => [name, index]));
5452
4961
  function pinNodeToLandmark(node, target, options = {}) {
5453
4962
  const offX = options.offsetX ?? 0;
5454
4963
  const offY = options.offsetY ?? 0;
@@ -5493,46 +5002,50 @@ function pinNodeToObject(node, target, options = {}) {
5493
5002
  }
5494
5003
  return pinNodeToLandmark(node, target, options);
5495
5004
  }
5005
+ /** Tasks a config turns on; detection is the default when nothing is enabled explicitly. */
5006
+ function enabledTasks(config) {
5007
+ const tasks = [];
5008
+ if (config.enableDetection === true) tasks.push("detect");
5009
+ if (config.enableSegmentation === true) tasks.push("segment");
5010
+ if (config.enablePose === true) tasks.push("pose");
5011
+ if (config.enableMatte === true) tasks.push("matte");
5012
+ return tasks.length > 0 ? tasks : ["detect"];
5013
+ }
5496
5014
  /**
5497
- * YOLO vision node.
5015
+ * Vision node.
5498
5016
  *
5499
- * LAZY BY CONSTRUCTION: creating a YoloNode performs ZERO I/O — no model
5500
- * downloads, no sessions, no file probes. The first inference call (or an
5501
- * explicit `await vision.ready()`) downloads the required models.
5017
+ * LAZY BY CONSTRUCTION: creating a VisionNode performs ZERO I/O — no model downloads, no
5018
+ * sessions, no file probes. The first inference call (or an explicit `await vision.ready()`)
5019
+ * downloads the required models.
5502
5020
  */
5503
- var YoloNode$1 = class YoloNode$1$1 {
5021
+ var VisionNode = class VisionNode$1 {
5504
5022
  id;
5505
- kind = "yolo";
5023
+ kind = "vision";
5506
5024
  source;
5507
5025
  config;
5508
5026
  vision;
5509
5027
  _runner;
5510
- /** Runner construction options (modelsDir, provider, …) — injectable for tests/agents. */
5028
+ /** Runner construction options (modelsDir, provider, store, …) — injectable for tests/agents. */
5511
5029
  _runnerOptions;
5512
5030
  constructor(source, config = {}, bundleOptions = {}, runnerOptions = {}) {
5513
- this.id = "yolo-" + Math.random().toString(36).slice(2, 9);
5031
+ this.id = `vision-${Math.random().toString(36).slice(2, 9)}`;
5514
5032
  this.source = source;
5515
5033
  this.config = config;
5516
5034
  this._runnerOptions = runnerOptions;
5517
- this.vision = new YoloVisionBundle({
5035
+ this.vision = new VisionBundle({
5518
5036
  ...bundleOptions,
5519
5037
  config
5520
5038
  });
5521
5039
  }
5522
- /** Lazily-created runner; downloads happen here on first access. */
5040
+ /** Lazily-created runner; downloads happen on its first inference. */
5523
5041
  runner() {
5524
5042
  if (!this._runner) {
5525
- const { variant, imgsz, confidence, iouThreshold, classes, enableWorld, prompts, customModel, modelsDir, baseUrl, maskThreshold, featherRadius } = this.config;
5526
- this._runner = YoloVisionRunner.create({
5043
+ const { variant, confidence, classes, modelsDir, baseUrl, maskThreshold, featherRadius } = this.config;
5044
+ this._runner = VisionRunner.create({
5527
5045
  ...this._runnerOptions,
5528
5046
  ...variant !== void 0 ? { variant } : {},
5529
- ...imgsz !== void 0 ? { imgsz } : {},
5530
5047
  ...confidence !== void 0 ? { confidence } : {},
5531
- ...iouThreshold !== void 0 ? { iouThreshold } : {},
5532
5048
  ...classes !== void 0 ? { classes } : {},
5533
- ...enableWorld !== void 0 ? { enableWorld } : {},
5534
- ...prompts !== void 0 ? { prompts } : {},
5535
- ...customModel !== void 0 ? { customModel } : {},
5536
5049
  ...modelsDir !== void 0 ? { modelsDir } : {},
5537
5050
  ...baseUrl !== void 0 ? { baseUrl } : {},
5538
5051
  ...maskThreshold !== void 0 ? { maskThreshold } : {},
@@ -5544,15 +5057,7 @@ var YoloNode$1 = class YoloNode$1$1 {
5544
5057
  /** Explicit warm-up — downloads the models for the enabled tasks, ahead of first inference. */
5545
5058
  async ready() {
5546
5059
  const runner = this.runner();
5547
- const tasks = [];
5548
- if (this.config.enableDetection === true) tasks.push("detect");
5549
- if (this.config.enableSegmentation === true) tasks.push("segment");
5550
- if (this.config.enablePose === true) tasks.push("pose");
5551
- if (this.config.enableObb === true) tasks.push("obb");
5552
- if (this.config.enableClassification === true) tasks.push("classify");
5553
- if (this.config.enableWorld === true) tasks.push("world");
5554
- const defaultTask = this.config.enableWorld === true ? "world" : "detect";
5555
- await runner.preload(tasks.length > 0 ? tasks : [defaultTask]);
5060
+ await runner.preload(enabledTasks(this.config));
5556
5061
  return runner;
5557
5062
  }
5558
5063
  close() {
@@ -5560,36 +5065,23 @@ var YoloNode$1 = class YoloNode$1$1 {
5560
5065
  this._runner = void 0;
5561
5066
  }
5562
5067
  static attach(source, config = {}, bundleOptions = {}) {
5563
- const nodeInstance = new YoloNode$1$1(source, config, bundleOptions);
5564
- const vision = nodeInstance.vision;
5565
- Object.defineProperty(vision, "node", {
5566
- value: {
5567
- id: nodeInstance.id,
5568
- kind: "yolo",
5569
- source: nodeInstance.source,
5570
- config: nodeInstance.config
5068
+ const node = new VisionNode$1(source, config, bundleOptions);
5069
+ const vision = node.vision;
5070
+ Object.defineProperties(vision, {
5071
+ node: {
5072
+ value: node.toNode(),
5073
+ enumerable: true
5571
5074
  },
5572
- enumerable: true,
5573
- writable: false
5574
- });
5575
- Object.defineProperty(vision, "ready", {
5576
- value: () => nodeInstance.ready(),
5577
- enumerable: false
5578
- });
5579
- Object.defineProperty(vision, "runner", {
5580
- value: () => nodeInstance.runner(),
5581
- enumerable: false
5582
- });
5583
- Object.defineProperty(vision, "close", {
5584
- value: () => nodeInstance.close(),
5585
- enumerable: false
5075
+ ready: { value: () => node.ready() },
5076
+ runner: { value: () => node.runner() },
5077
+ close: { value: () => node.close() }
5586
5078
  });
5587
5079
  return vision;
5588
5080
  }
5589
5081
  toNode() {
5590
5082
  return {
5591
5083
  id: this.id,
5592
- kind: "yolo",
5084
+ kind: "vision",
5593
5085
  source: this.source,
5594
5086
  config: this.config
5595
5087
  };
@@ -19639,11 +19131,12 @@ const VignetteWebGPURenderer = async (args) => {
19639
19131
  var renderers_default$1 = defineRenderer({ WebGPURenderer: VignetteWebGPURenderer });
19640
19132
 
19641
19133
  //#endregion
19642
- //#region ../../nodes/node-yolo/dist/renderers-mupYaxMa.mjs
19134
+ //#region ../../nodes/node-vision/dist/renderers-CpOzS0Wl.mjs
19643
19135
  let sharedRunner = null;
19644
19136
  let sharedSkeletonRenderer = null;
19645
19137
  let sharedTexturePool = null;
19646
- let sharedObjectTracker = null;
19138
+ /** One tracker per vision node — track ids must not mix across nodes/sources. */
19139
+ const objectTrackers = /* @__PURE__ */ new Map();
19647
19140
  /**
19648
19141
  * Persistent per-node child textures. The shared frame encoder is submitted at
19649
19142
  * the END of the frame, so a node renderer cannot read back the child it just
@@ -19651,46 +19144,50 @@ let sharedObjectTracker = null;
19651
19144
  * the start of the next call — a stable one-frame delay instead of the
19652
19145
  * unpredictable multi-frame lag you get from pooled textures.
19653
19146
  *
19654
- * Keyed on the stable Yolo `Effect` instance (render ids change every frame),
19147
+ * Keyed on the stable Vision `Effect` instance (render ids change every frame),
19655
19148
  * with a size-keyed fallback for raw operation nodes.
19656
19149
  */
19657
- const yoloChildTexturesByEffect = /* @__PURE__ */ new WeakMap();
19658
- const yoloChildTexturesByKey = /* @__PURE__ */ new Map();
19150
+ const childTexturesByEffect = /* @__PURE__ */ new WeakMap();
19151
+ const childTexturesByKey = /* @__PURE__ */ new Map();
19659
19152
  /**
19660
19153
  * Last successful subject mask, reused for a few frames when the model misses —
19661
19154
  * a transient miss then holds the silhouette instead of flashing the raw plate.
19662
- * Keyed per node/effect so multiple YOLO nodes do not collide.
19155
+ * Keyed per node/effect so multiple vision nodes do not collide.
19663
19156
  */
19664
19157
  const lastSubjectByNode = /* @__PURE__ */ new Map();
19665
19158
  const SUBJECT_HOLD_FRAMES = 3;
19666
19159
  /**
19667
19160
  * Lazy shared runner — `create()` is a pure constructor (zero I/O); the first frame that
19668
19161
  * requests a task triggers that task's model download at inference time (never at init).
19162
+ * Recreated only when an option that changes inference changes.
19669
19163
  */
19670
19164
  function getSharedRunner(op) {
19671
- if (!sharedRunner || sharedRunner.variant !== op.variant || sharedRunner.enableWorld !== op.enableWorld || sharedRunner.prompts !== op.prompts || sharedRunner.customModel !== op.customModel || sharedRunner.featherRadius !== op.featherRadius || sharedRunner.maskThreshold !== op.maskThreshold) sharedRunner = YoloVisionRunner.create({
19165
+ const options = {
19672
19166
  variant: op.variant,
19673
- imgsz: op.imgsz,
19674
19167
  confidence: op.confidence,
19675
- iouThreshold: op.iouThreshold,
19676
19168
  classes: op.classes,
19677
- enableWorld: op.enableWorld,
19678
- prompts: op.prompts,
19679
- customModel: op.customModel,
19680
19169
  maskThreshold: op.maskThreshold,
19681
19170
  featherRadius: op.featherRadius,
19682
19171
  modelsDir: op.modelsDir,
19683
19172
  baseUrl: op.baseUrl
19684
- });
19685
- return sharedRunner;
19173
+ };
19174
+ const key = JSON.stringify(options);
19175
+ if (sharedRunner?.key !== key) {
19176
+ sharedRunner?.runner.close();
19177
+ sharedRunner = {
19178
+ key,
19179
+ runner: VisionRunner.create(options)
19180
+ };
19181
+ }
19182
+ return sharedRunner.runner;
19686
19183
  }
19687
- /** Lazy shared tracker — detections (incl. OBB) are tracked under one stable identity space. */
19688
- function getSharedObjectTracker(op) {
19689
- if (!sharedObjectTracker) sharedObjectTracker = new TemporalObjectTracker({
19690
- iouThreshold: op.iouThreshold ?? .25,
19691
- maxMissedFrames: op.maxMissedFrames ?? 15
19692
- });
19693
- return sharedObjectTracker;
19184
+ function getObjectTracker(nodeKey, op) {
19185
+ let tracker = objectTrackers.get(nodeKey);
19186
+ if (!tracker) {
19187
+ tracker = new TemporalObjectTracker({ maxMissedFrames: op.maxMissedFrames ?? 15 });
19188
+ objectTrackers.set(nodeKey, tracker);
19189
+ }
19190
+ return tracker;
19694
19191
  }
19695
19192
  /**
19696
19193
  * Normalizes the runner's pose keypoints (plate pixel space) to [0, 1] landmarks, which is
@@ -19727,26 +19224,11 @@ function assignMasksToTracks(masks, tracked) {
19727
19224
  } : mask;
19728
19225
  });
19729
19226
  }
19730
- /** Picks the subject mask: the largest instance, preferring a person when present. */
19731
- function selectSubjectMask(masks) {
19732
- if (masks.length === 0) return void 0;
19733
- let best = masks[0];
19734
- for (const m of masks) {
19735
- const mIsPerson = m.category.toLowerCase() === "person";
19736
- const bestIsPerson = best.category.toLowerCase() === "person";
19737
- if (mIsPerson && !bestIsPerson) {
19738
- best = m;
19739
- continue;
19740
- }
19741
- if (mIsPerson === bestIsPerson && m.area > best.area) best = m;
19742
- }
19743
- return best;
19744
- }
19745
19227
  /**
19746
- * Grows a person mask into connected foreground pixels — the COCO person head
19747
- * drops a flowing dress (treats it as non-person), so seeding a flood fill from
19748
- * the body through pixels that differ from the sampled backdrop recovers the
19749
- * fabric without adding the near-uniform studio wall or its soft grey shadow.
19228
+ * Grows a subject mask into connected foreground pixels — seeding a flood fill
19229
+ * from the subject through pixels that differ from the sampled backdrop recovers
19230
+ * thin or fast-moving edges (hair, fabric) without adding a near-uniform studio
19231
+ * wall or its soft grey shadow.
19750
19232
  */
19751
19233
  function fillInternalHoles(mask, width, height) {
19752
19234
  const exterior = new Uint8Array(width * height);
@@ -19889,26 +19371,25 @@ function maskCentroid(mask) {
19889
19371
  cy: sumY / count
19890
19372
  };
19891
19373
  }
19892
- const YoloWebGPURenderer = async (args) => {
19374
+ const VisionWebGPURenderer = async (args) => {
19893
19375
  const { ctx, encoder, pass, targetView, targetWidth, targetHeight, props, drawChild } = args;
19894
19376
  const { virtualMedia } = props;
19895
19377
  const rawOp = virtualMedia?.operation;
19896
- if (!rawOp) return;
19897
- const op = normalizeOperation(rawOp);
19898
- if (!op) return;
19378
+ if (rawOp?.op !== "Vision") return;
19379
+ const op = rawOp;
19899
19380
  pass.end();
19900
19381
  const childMedia = virtualMedia?.children?.[0];
19901
19382
  if (!childMedia) return;
19902
19383
  const effectKey = op.effect;
19903
19384
  const hasEffectKey = effectKey !== void 0 && effectKey !== null;
19904
19385
  const nodeKeyStr = virtualMedia?.id ?? rawOp?.id ?? `${targetWidth}x${targetHeight}_${op.mode}_${op.variant}_${op.keyBackground}_${op.backgroundKeyThreshold}`;
19905
- const fallbackKey = `yolo-${nodeKeyStr}`;
19906
- let childTex = hasEffectKey ? yoloChildTexturesByEffect.get(effectKey) : yoloChildTexturesByKey.get(fallbackKey);
19386
+ const fallbackKey = `vision-${nodeKeyStr}`;
19387
+ let childTex = hasEffectKey ? childTexturesByEffect.get(effectKey) : childTexturesByKey.get(fallbackKey);
19907
19388
  if (childTex && (childTex.width !== targetWidth || childTex.height !== targetHeight)) {
19908
19389
  childTex.destroy();
19909
19390
  childTex = void 0;
19910
- if (hasEffectKey) yoloChildTexturesByEffect.delete(effectKey);
19911
- else yoloChildTexturesByKey.delete(fallbackKey);
19391
+ if (hasEffectKey) childTexturesByEffect.delete(effectKey);
19392
+ else childTexturesByKey.delete(fallbackKey);
19912
19393
  }
19913
19394
  const hasPreviousFrame = childTex !== void 0;
19914
19395
  if (!childTex) {
@@ -19916,15 +19397,15 @@ const YoloWebGPURenderer = async (args) => {
19916
19397
  size: [targetWidth, targetHeight],
19917
19398
  format: ctx.renderer.format,
19918
19399
  usage: GPUTextureUsage.RENDER_ATTACHMENT | GPUTextureUsage.TEXTURE_BINDING | GPUTextureUsage.COPY_SRC,
19919
- label: "yolo_child_persistent"
19400
+ label: "vision_child_persistent"
19920
19401
  });
19921
- if (hasEffectKey) yoloChildTexturesByEffect.set(effectKey, childTex);
19922
- else yoloChildTexturesByKey.set(fallbackKey, childTex);
19923
- if (yoloChildTexturesByKey.size > 4) {
19924
- const oldest = yoloChildTexturesByKey.keys().next().value;
19402
+ if (hasEffectKey) childTexturesByEffect.set(effectKey, childTex);
19403
+ else childTexturesByKey.set(fallbackKey, childTex);
19404
+ if (childTexturesByKey.size > 4) {
19405
+ const oldest = childTexturesByKey.keys().next().value;
19925
19406
  if (oldest && oldest !== fallbackKey) {
19926
- yoloChildTexturesByKey.get(oldest)?.destroy();
19927
- yoloChildTexturesByKey.delete(oldest);
19407
+ childTexturesByKey.get(oldest)?.destroy();
19408
+ childTexturesByKey.delete(oldest);
19928
19409
  }
19929
19410
  }
19930
19411
  }
@@ -19936,9 +19417,9 @@ const YoloWebGPURenderer = async (args) => {
19936
19417
  const stagingBuffer = ctx.device.createBuffer({
19937
19418
  size: bufferSize,
19938
19419
  usage: GPUBufferUsage.COPY_DST | GPUBufferUsage.MAP_READ,
19939
- label: "yolo_frame_staging"
19420
+ label: "vision_frame_staging"
19940
19421
  });
19941
- const readbackEncoder = ctx.device.createCommandEncoder({ label: "yolo_readback_encoder" });
19422
+ const readbackEncoder = ctx.device.createCommandEncoder({ label: "vision_readback_encoder" });
19942
19423
  readbackEncoder.copyTextureToBuffer({ texture: childTex }, {
19943
19424
  buffer: stagingBuffer,
19944
19425
  bytesPerRow,
@@ -19966,62 +19447,45 @@ const YoloWebGPURenderer = async (args) => {
19966
19447
  }, targetWidth, targetHeight, "clear").end();
19967
19448
  await drawChild(childMedia, { ...props }, childView, childTex, targetWidth, targetHeight);
19968
19449
  const frameIdx = props.frame ?? 0;
19450
+ const fps = props.fps ?? 24;
19451
+ const mode = op.mode ?? "passthrough";
19452
+ const isMatteMode = mode === "mask" || mode === "matte" || mode === "crop";
19969
19453
  const visionBundle = op.visionBundle ?? virtualMedia.visionBundle;
19970
- const runner = await getSharedRunner(op);
19454
+ const runner = getSharedRunner(op);
19971
19455
  const image = framePixels ? {
19972
19456
  data: framePixels,
19973
19457
  width: targetWidth,
19974
19458
  height: targetHeight
19975
19459
  } : null;
19976
- if (image && op.mode !== "obb" && (op.enableDetection !== false || op.mode === "boxes" || op.mode === "tracking")) {
19460
+ let frameObjects = [];
19461
+ if (image && (op.enableDetection !== false || mode === "boxes" || mode === "tracking")) {
19977
19462
  const detections = await runner.detect(image);
19978
- const tracker = getSharedObjectTracker(op);
19463
+ const tracker = getObjectTracker(nodeKeyStr, op);
19979
19464
  if (frameIdx === 0) tracker.reset();
19980
- const tracked = tracker.update(detections, frameIdx, props.fps ?? 24);
19465
+ frameObjects = tracker.update(detections, frameIdx, fps);
19981
19466
  visionBundle?.setObjectResult(frameIdx, {
19982
- objects: tracked,
19467
+ objects: frameObjects,
19983
19468
  rawDetections: detections
19984
19469
  });
19985
19470
  }
19986
- const needsSegmentation = op.enableSegmentation === true || op.mode === "mask" || op.mode === "matte" || op.mode === "crop";
19987
19471
  let frameMasks = [];
19988
- if (image && needsSegmentation) {
19989
- const segRes = await runner.segment(image);
19990
- const tracked = visionBundle?.getObjectResult(frameIdx).objects ?? [];
19991
- frameMasks = assignMasksToTracks(segRes.masks, tracked);
19472
+ if (image && (op.enableSegmentation === true || isMatteMode && op.matteSource !== "selfie")) {
19473
+ frameMasks = assignMasksToTracks((await runner.segment(image)).masks, frameObjects);
19992
19474
  visionBundle?.setMaskResult(frameIdx, frameMasks);
19993
19475
  }
19476
+ let personMatte;
19477
+ if (image && (op.enableMatte === true || isMatteMode && op.matteSource === "selfie")) {
19478
+ personMatte = await runner.matte(image);
19479
+ visionBundle?.setMatteResult(frameIdx, personMatte);
19480
+ }
19994
19481
  let currentPoseRes;
19995
- if (image && (op.enablePose === true || op.mode === "skeleton")) {
19996
- const normalized = normalizePoseKeypoints(await runner.pose(image), image.width, image.height);
19997
- const tracked = visionBundle?.getObjectResult(frameIdx).objects ?? [];
19998
- currentPoseRes = { people: matchPoseToTracks(normalized.people, tracked) };
19482
+ if (image && (op.enablePose === true || mode === "skeleton")) {
19483
+ currentPoseRes = { people: matchPoseToTracks(normalizePoseKeypoints(await runner.pose(image), image.width, image.height).people, frameObjects) };
19999
19484
  visionBundle?.setPoseResult(frameIdx, currentPoseRes);
20000
19485
  }
20001
- if (image && op.enableClassification === true) {
20002
- const classifyRes = await runner.classify(image);
20003
- visionBundle?.setClassifyResult(frameIdx, classifyRes);
20004
- }
20005
- if (image && (op.enableObb === true || op.mode === "obb")) {
20006
- const obbRes = await runner.detectObb(image);
20007
- visionBundle?.setObbResult(frameIdx, obbRes);
20008
- const detections = obbRes.detections.map((d) => ({
20009
- category: d.category,
20010
- score: d.score,
20011
- boundingBox: d.boundingBox
20012
- }));
20013
- const tracker = getSharedObjectTracker(op);
20014
- if (frameIdx === 0) tracker.reset();
20015
- const tracked = tracker.update(detections, frameIdx, props.fps ?? 24);
20016
- visionBundle?.setObjectResult(frameIdx, {
20017
- objects: tracked,
20018
- rawDetections: detections
20019
- });
20020
- }
20021
- const mode = op.mode ?? "passthrough";
20022
- if (mode === "mask" || mode === "matte" || mode === "crop") {
19486
+ if (isMatteMode) {
20023
19487
  if (!sharedTexturePool) sharedTexturePool = new SegmentationTexturePool(ctx.device);
20024
- const selected = selectSubjectMask(frameMasks.length > 0 ? frameMasks : visionBundle?.getMaskResult(frameIdx) ?? []);
19488
+ const selected = op.matteSource === "selfie" ? personMatteAsMask(personMatte) : mergeSubjectMask(frameMasks.length > 0 ? frameMasks : visionBundle?.getMaskResult(frameIdx) ?? []);
20025
19489
  let subject;
20026
19490
  const lastSubjectEntry = lastSubjectByNode.get(nodeKeyStr);
20027
19491
  const lastSubjectMask = lastSubjectEntry?.mask ?? null;
@@ -20095,39 +19559,6 @@ const YoloWebGPURenderer = async (args) => {
20095
19559
  return;
20096
19560
  }
20097
19561
  }
20098
- if (mode === "obb") {
20099
- const outPass$1 = ctx.renderer.beginFrame(encoder, targetView, {
20100
- r: 0,
20101
- g: 0,
20102
- b: 0,
20103
- a: 0
20104
- }, targetWidth, targetHeight, "clear");
20105
- ctx.renderer.drawTexture(outPass$1, childTex, {
20106
- x: 0,
20107
- y: 0,
20108
- width: targetWidth,
20109
- height: targetHeight
20110
- });
20111
- const detections = visionBundle?.getObbResult(frameIdx).detections ?? [];
20112
- const obbColor = "#f59e0b";
20113
- for (const det of detections) {
20114
- if (det.corners.length < 4) continue;
20115
- const path$1 = det.corners.map((corner, i) => `${i === 0 ? "M" : "L"} ${corner[0]} ${corner[1]}`).join(" ") + " Z";
20116
- try {
20117
- ctx.renderer.drawPath(outPass$1, path$1, obbColor, 2);
20118
- } catch {
20119
- const b = det.boundingBox;
20120
- ctx.renderer.drawRect(outPass$1, {
20121
- x: b.originX,
20122
- y: b.originY,
20123
- width: b.width,
20124
- height: b.height
20125
- }, obbColor, 2);
20126
- }
20127
- }
20128
- outPass$1.end();
20129
- return;
20130
- }
20131
19562
  if (mode === "boxes" || mode === "tracking") {
20132
19563
  const outPass$1 = ctx.renderer.beginFrame(encoder, targetView, {
20133
19564
  r: 0,
@@ -20141,11 +19572,11 @@ const YoloWebGPURenderer = async (args) => {
20141
19572
  width: targetWidth,
20142
19573
  height: targetHeight
20143
19574
  });
20144
- const objRes = visionBundle?.getObjectResult(frameIdx);
20145
- if (objRes && objRes.objects.length > 0) {
19575
+ const objects = frameObjects.length > 0 ? frameObjects : visionBundle?.getObjectResult(frameIdx).objects ?? [];
19576
+ if (objects.length > 0) {
20146
19577
  const boxColor = "#38bdf8";
20147
19578
  const stroke = 2;
20148
- for (const obj of objRes.objects) {
19579
+ for (const obj of objects) {
20149
19580
  if (!obj.active) continue;
20150
19581
  const b = obj.boundingBox;
20151
19582
  ctx.renderer.drawRect(outPass$1, {
@@ -20235,73 +19666,48 @@ function uploadComposite(pool, mode, subject, framePixels, width, height, frameI
20235
19666
  out[px + 3] = alpha;
20236
19667
  }
20237
19668
  }
20238
- const key = `yolo_composite_${nodeKeyStr}_${mode}_${frameIdx}`;
19669
+ const key = `vision_composite_${nodeKeyStr}_${mode}_${frameIdx}`;
20239
19670
  const tex = pool.uploadMask(key, out, width, height);
20240
19671
  if (mode === "crop") {
20241
- const box = subjectBoundingBox(subject);
20242
- if (box) {
20243
- const cropW = Math.max(1, Math.round(box.width));
20244
- const cropH = Math.max(1, Math.round(box.height));
20245
- const crop = new Uint8Array(width * height * 4);
20246
- for (let y = 0; y < height; y++) {
20247
- const srcY = Math.min(cropH - 1, Math.max(0, Math.round(y / height * cropH)));
20248
- for (let x = 0; x < width; x++) {
20249
- const srcX = Math.min(cropW - 1, Math.max(Math.round(x / width * cropW)));
20250
- const srcIdx = (srcY * width + srcX) * 4;
20251
- const dstIdx = (y * width + x) * 4;
20252
- crop[dstIdx] = out[srcIdx];
20253
- crop[dstIdx + 1] = out[srcIdx + 1];
20254
- crop[dstIdx + 2] = out[srcIdx + 2];
20255
- crop[dstIdx + 3] = out[srcIdx + 3];
20256
- }
20257
- }
20258
- return pool.uploadMask(`yolo_crop_${frameIdx}`, crop, width, height);
20259
- }
19672
+ const box = maskBounds(subject);
19673
+ if (box) return pool.uploadMask(`vision_crop_${nodeKeyStr}_${frameIdx}`, cropToBounds(out, width, height, box), width, height);
20260
19674
  }
20261
19675
  return tex;
20262
19676
  }
20263
- function subjectBoundingBox(subject) {
20264
- const { mask, width } = subject;
20265
- let minX = width;
20266
- let minY = Infinity;
20267
- let maxX = -1;
20268
- let maxY = -1;
20269
- for (let y = 0; y < subject.height; y++) for (let x = 0; x < width; x++) if (mask[y * width + x] > 0) {
20270
- if (x < minX) minX = x;
20271
- if (x > maxX) maxX = x;
20272
- if (y < minY) minY = y;
20273
- if (y > maxY) maxY = y;
20274
- }
20275
- if (maxX < minX || maxY < minY) return null;
19677
+ /** Nearest-neighbour resample of `bounds` inside an RGBA frame up to the full frame. */
19678
+ function cropToBounds(rgba, width, height, bounds) {
19679
+ const cropW = bounds.x1 - bounds.x0 + 1;
19680
+ const cropH = bounds.y1 - bounds.y0 + 1;
19681
+ const crop = new Uint8Array(width * height * 4);
19682
+ for (let y = 0; y < height; y++) {
19683
+ const srcY = bounds.y0 + Math.min(cropH - 1, Math.floor(y / height * cropH));
19684
+ for (let x = 0; x < width; x++) {
19685
+ const srcX = bounds.x0 + Math.min(cropW - 1, Math.floor(x / width * cropW));
19686
+ const src = (srcY * width + srcX) * 4;
19687
+ const dst = (y * width + x) * 4;
19688
+ crop[dst] = rgba[src];
19689
+ crop[dst + 1] = rgba[src + 1];
19690
+ crop[dst + 2] = rgba[src + 2];
19691
+ crop[dst + 3] = rgba[src + 3];
19692
+ }
19693
+ }
19694
+ return crop;
19695
+ }
19696
+ /** Adapts a Selfie Segmenter matte to the instance-mask shape the composite path takes. */
19697
+ function personMatteAsMask(matte) {
19698
+ if (!matte || matte.coverage <= 0) return void 0;
19699
+ const area = Math.round(matte.coverage * matte.width * matte.height);
20276
19700
  return {
20277
- width: maxX - minX + 1,
20278
- height: maxY - minY + 1
19701
+ category: "person",
19702
+ mask: matte.mask,
19703
+ width: matte.width,
19704
+ height: matte.height,
19705
+ area,
19706
+ coverage: matte.coverage,
19707
+ detectionIndex: -1
20279
19708
  };
20280
19709
  }
20281
- /**
20282
- * Legacy-compat: accepts a `MediaPipe` operation (from compositions targeting the old
20283
- * node) and maps its flags onto the YOLO config — pose, object tracking, segmentation,
20284
- * and passthrough/mask/skeleton/boxes/tracking modes all carry over.
20285
- */
20286
- function normalizeOperation(raw) {
20287
- if (raw.op === "Yolo") return raw;
20288
- if (raw.op === "MediaPipe") {
20289
- const legacy = raw;
20290
- return {
20291
- ...raw,
20292
- op: "Yolo",
20293
- enableDetection: legacy.enableObjectTracking === true || legacy.mode === "boxes" || legacy.mode === "tracking",
20294
- enablePose: legacy.enablePoseLandmarks === true || legacy.mode === "skeleton",
20295
- enableSegmentation: legacy.enableSegmentation === true || legacy.mode === "mask" || legacy.mode === "matte",
20296
- confidence: legacy.objectScoreThreshold ?? .25,
20297
- iouThreshold: legacy.iouThreshold ?? .45,
20298
- classes: legacy.objectCategories ?? void 0,
20299
- modelsDir: legacy.modelsDir ?? void 0
20300
- };
20301
- }
20302
- return null;
20303
- }
20304
- var renderers_default = defineRenderer({ WebGPURenderer: YoloWebGPURenderer });
19710
+ var renderers_default = defineRenderer({ WebGPURenderer: VisionWebGPURenderer });
20305
19711
 
20306
19712
  //#endregion
20307
19713
  //#region src/effects/with-defaults.ts
@@ -21036,42 +20442,26 @@ var Crop = class extends Effect {
21036
20442
  });
21037
20443
  }
21038
20444
  };
21039
- var Yolo = class extends Effect {
21040
- op = "Yolo";
20445
+ /**
20446
+ * On-device vision: tracks objects, segments instances, estimates pose and mattes people on
20447
+ * the rendered frame, feeding reactive signals. Models download lazily on first use.
20448
+ */
20449
+ var Vision = class extends Effect {
20450
+ op = "Vision";
21041
20451
  constructor(config = {}) {
21042
20452
  super({
21043
20453
  enableDetection: config.enableDetection !== false,
21044
20454
  enableSegmentation: config.enableSegmentation === true,
21045
20455
  enablePose: config.enablePose === true,
21046
- enableClassification: config.enableClassification === true,
21047
- enableObb: config.enableObb === true,
21048
- confidence: config.confidence ?? .25,
21049
- iouThreshold: config.iouThreshold ?? .45,
21050
- variant: config.variant ?? "n",
21051
- imgsz: config.imgsz ?? 640,
20456
+ enableMatte: config.enableMatte === true,
20457
+ confidence: config.confidence ?? .3,
20458
+ variant: config.variant ?? "s",
21052
20459
  mode: config.mode ?? "passthrough",
21053
- delegate: config.delegate ?? "CPU",
20460
+ matteSource: config.matteSource ?? "instance",
21054
20461
  ...config
21055
20462
  });
21056
20463
  }
21057
20464
  };
21058
- /**
21059
- * @deprecated Use `Yolo`. Kept as a source-compatible alias that maps the legacy MediaPipe
21060
- * option shape onto the YOLO engine (`op` is `"Yolo"`).
21061
- */
21062
- var MediaPipe = class extends Yolo {
21063
- constructor(config = {}) {
21064
- super({
21065
- enableDetection: true,
21066
- enablePose: config.enablePoseLandmarks !== false,
21067
- enableSegmentation: config.enableSegmentation === true,
21068
- mode: config.mode === "skeleton" ? "skeleton" : config.mode ?? "passthrough",
21069
- ...config.delegate !== void 0 ? { delegate: config.delegate } : {},
21070
- ...config.modelsDir !== void 0 ? { modelsDir: config.modelsDir } : {},
21071
- ...config.visionBundle !== void 0 ? { visionBundle: config.visionBundle } : {}
21072
- });
21073
- }
21074
- };
21075
20465
  var Relight3D = class extends Effect {
21076
20466
  op = "Relight3D";
21077
20467
  constructor(config = {}) {
@@ -21126,5 +20516,5 @@ var Modulate = class extends Effect {
21126
20516
  };
21127
20517
 
21128
20518
  //#endregion
21129
- export { renderers_default$23 as $, renderers_default as A, renderers_default$10 as B, Yolo as C, DepthNormalsComputePipeline as Ct, ColorKey as D, ColorBalance as E, TensorPipeline as Et, renderers_default$5 as F, renderers_default$15 as G, renderers_default$12 as H, renderers_default$6 as I, renderers_default$18 as J, renderers_default$16 as K, renderers_default$7 as L, renderers_default$2 as M, renderers_default$3 as N, Vignette as O, renderers_default$4 as P, renderers_default$22 as Q, renderers_default$8 as R, UnsharpMask as S, ControlNetMultiplexer as St, Blur as T, TemporalDeflickerPipeline as Tt, renderers_default$13 as U, renderers_default$11 as V, renderers_default$14 as W, renderers_default$20 as X, renderers_default$19 as Y, renderers_default$21 as Z, SSAO as _, YoloVisionRunner as _t, CustomEffect as a, renderers_default$29 as at, TemporalDeflicker as b, pinNodeToObject as bt, GradientMap as c, POSE_LANDMARKS_YOLO as ct, Levels as d, SpatialLandmarkTransformer as dt, renderers_default$24 as et, MediaPipe as f, TemporalObjectTracker as ft, Relight3D as g, YoloVisionBundle as gt, PBRGlass as h, YoloNode$1 as ht, Curves as i, renderers_default$28 as it, renderers_default$1 as j, effectMeta as k, HalftoneScreen as l, PoseSkeletonRenderer as lt, MotionBlur as m, YOLO_POSE_KEYPOINT_NAMES as mt, AudioSignalExtractor as n, renderers_default$26 as nt, DepthOfField as o, renderers_default$30 as ot, Modulate as p, YOLO_COCO17_BONES as pt, renderers_default$17 as q, Crop as r, renderers_default$27 as rt, FilmGrain as s, MEDIAPIPE_TO_YOLO_POSE_INDEX as st, ApplyLUT as t, renderers_default$25 as tt, HighPass as u, SegmentationTexturePool as ut, SelectiveColor as v, analyzeSequence as vt, effectFactories as w, OpticalFlowComputePipeline as wt, TileOffset as x, CannyComputePipeline as xt, ShadowsHighlights as y, pinNodeToLandmark as yt, renderers_default$9 as z };
21130
- //# sourceMappingURL=effects-4t4qk-BR.mjs.map
20519
+ export { renderers_default$24 as $, renderers_default$1 as A, renderers_default$11 as B, effectFactories as C, OpticalFlowComputePipeline as Ct, Vignette as D, ColorKey as E, renderers_default$6 as F, renderers_default$16 as G, renderers_default$13 as H, renderers_default$7 as I, renderers_default$19 as J, renderers_default$17 as K, renderers_default$8 as L, renderers_default$3 as M, renderers_default$4 as N, effectMeta as O, renderers_default$5 as P, renderers_default$23 as Q, renderers_default$9 as R, Vision as S, DepthNormalsComputePipeline as St, ColorBalance as T, TensorPipeline as Tt, renderers_default$14 as U, renderers_default$12 as V, renderers_default$15 as W, renderers_default$21 as X, renderers_default$20 as Y, renderers_default$22 as Z, SelectiveColor as _, pinNodeToLandmark as _t, CustomEffect as a, renderers_default$30 as at, TileOffset as b, CannyComputePipeline as bt, GradientMap as c, COCO17_KEYPOINT_NAMES as ct, Levels as d, SpatialLandmarkTransformer as dt, renderers_default$25 as et, Modulate as f, TemporalObjectTracker as ft, SSAO as g, analyzeSequence as gt, Relight3D as h, VisionRunner as ht, Curves as i, renderers_default$29 as it, renderers_default$2 as j, renderers_default as k, HalftoneScreen as l, PoseSkeletonRenderer as lt, PBRGlass as m, VisionNode as mt, AudioSignalExtractor as n, renderers_default$27 as nt, DepthOfField as o, COCO17_BONES as ot, MotionBlur as p, VisionBundle as pt, renderers_default$18 as q, Crop as r, renderers_default$28 as rt, FilmGrain as s, COCO17_KEYPOINTS as st, ApplyLUT as t, renderers_default$26 as tt, HighPass as u, SegmentationTexturePool as ut, ShadowsHighlights as v, pinNodeToObject as vt, Blur as w, TemporalDeflickerPipeline as wt, UnsharpMask as x, ControlNetMultiplexer as xt, TemporalDeflicker as y, COCO_CLASSES as yt, renderers_default$10 as z };
20520
+ //# sourceMappingURL=effects-C9connag.mjs.map