@framefields/node-vision 2.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,953 @@
1
+ import { defineRenderer } from "@framefields/node-sdk/renderer";
2
+ import { PoseSkeletonRenderer, TemporalObjectTracker, VisionRunner, maskBounds, matchPoseToTracks, mergeSubjectMask } from "@framefields/vision";
3
+
4
+ //#region src/renderers/frame-cache.ts
5
+ let scratchValues = new Uint8Array(0);
6
+ let scratchLengths = new Uint32Array(0);
7
+ function encodeMask(mask, width, height) {
8
+ if (scratchValues.length < mask.length) {
9
+ scratchValues = new Uint8Array(mask.length);
10
+ scratchLengths = new Uint32Array(mask.length);
11
+ }
12
+ const values = scratchValues;
13
+ const lengths = scratchLengths;
14
+ let runs = 0;
15
+ let start = 0;
16
+ for (let i = 1; i <= mask.length; i++) if (i === mask.length || mask[i] !== mask[start]) {
17
+ values[runs] = mask[start];
18
+ lengths[runs] = i - start;
19
+ runs++;
20
+ start = i;
21
+ }
22
+ return {
23
+ width,
24
+ height,
25
+ values: values.slice(0, runs),
26
+ lengths: lengths.slice(0, runs)
27
+ };
28
+ }
29
+ function decodeMask(rle) {
30
+ const out = new Uint8Array(rle.width * rle.height);
31
+ let at = 0;
32
+ for (let r = 0; r < rle.values.length; r++) {
33
+ const end = at + rle.lengths[r];
34
+ if (rle.values[r] !== 0) out.fill(rle.values[r], at, end);
35
+ at = end;
36
+ }
37
+ return out;
38
+ }
39
+ function rleBytes(rle) {
40
+ return rle.values.byteLength + rle.lengths.byteLength;
41
+ }
42
+ function toCachedMask(mask) {
43
+ const { mask: data, ...rest } = mask;
44
+ return {
45
+ ...rest,
46
+ rle: encodeMask(data, mask.width, mask.height)
47
+ };
48
+ }
49
+ function fromCachedMask(mask) {
50
+ const { rle, ...rest } = mask;
51
+ return {
52
+ ...rest,
53
+ mask: decodeMask(rle)
54
+ };
55
+ }
56
+ function resultBytes(result) {
57
+ let bytes = 512 + result.detections.length * 128 + result.objects.length * 256;
58
+ for (const m of result.masks ?? []) bytes += rleBytes(m.rle);
59
+ if (result.subject) bytes += rleBytes(result.subject.mask.rle);
60
+ return bytes;
61
+ }
62
+ /** Results by source (node + inference settings) and frame, least recently used out first. */
63
+ var VisionFrameCache = class {
64
+ entries = /* @__PURE__ */ new Map();
65
+ bytes = 0;
66
+ constructor(maxBytes = 256 * 1024 * 1024) {
67
+ this.maxBytes = maxBytes;
68
+ }
69
+ get(source, frame) {
70
+ const key = `${source}#${frame}`;
71
+ const hit = this.entries.get(key);
72
+ if (hit) {
73
+ this.entries.delete(key);
74
+ this.entries.set(key, hit);
75
+ }
76
+ return hit;
77
+ }
78
+ set(source, frame, result) {
79
+ const key = `${source}#${frame}`;
80
+ const old = this.entries.get(key);
81
+ if (old) {
82
+ this.bytes -= resultBytes(old);
83
+ this.entries.delete(key);
84
+ }
85
+ this.entries.set(key, result);
86
+ this.bytes += resultBytes(result);
87
+ for (const [k, v] of this.entries) {
88
+ if (this.bytes <= this.maxBytes) break;
89
+ this.entries.delete(k);
90
+ this.bytes -= resultBytes(v);
91
+ }
92
+ }
93
+ clear() {
94
+ this.entries.clear();
95
+ this.bytes = 0;
96
+ }
97
+ get size() {
98
+ return this.entries.size;
99
+ }
100
+ };
101
+
102
+ //#endregion
103
+ //#region src/renderers/matte-compositor.ts
104
+ /**
105
+ * GPU composite for the mask-family modes: the picture times the subject mask,
106
+ * in one full-screen pass. Replaces building the RGBA result on the CPU and
107
+ * uploading it; only the 8-bit mask goes to the GPU.
108
+ *
109
+ * - matte: picture × mask (premultiplied, so the background is transparent)
110
+ * - mask: white silhouette (rgb = mask, a = 1)
111
+ * - crop: matte, with the subject's bounding box stretched to the frame
112
+ */
113
+ const WGSL = `
114
+ struct Params {
115
+ mode: u32, // 0 matte, 1 mask, 2 crop
116
+ _pad: u32,
117
+ cropMin: vec2<f32>, // uv of the subject bounds (crop)
118
+ cropMax: vec2<f32>,
119
+ _pad2: vec2<f32>,
120
+ };
121
+
122
+ @group(0) @binding(0) var<uniform> params: Params;
123
+ @group(0) @binding(1) var picture: texture_2d<f32>;
124
+ @group(0) @binding(2) var mask: texture_2d<f32>;
125
+ @group(0) @binding(3) var samp: sampler;
126
+
127
+ struct VSOut {
128
+ @builtin(position) pos: vec4<f32>,
129
+ @location(0) uv: vec2<f32>,
130
+ };
131
+
132
+ @vertex fn vs(@builtin(vertex_index) i: u32) -> VSOut {
133
+ var p = array<vec2<f32>, 3>(vec2(-1.0, -1.0), vec2(3.0, -1.0), vec2(-1.0, 3.0));
134
+ var out: VSOut;
135
+ out.pos = vec4(p[i], 0.0, 1.0);
136
+ out.uv = vec2(p[i].x * 0.5 + 0.5, 0.5 - p[i].y * 0.5);
137
+ return out;
138
+ }
139
+
140
+ @fragment fn fs(in: VSOut) -> @location(0) vec4<f32> {
141
+ var uv = in.uv;
142
+ if (params.mode == 2u) {
143
+ uv = mix(params.cropMin, params.cropMax, uv);
144
+ }
145
+ let m = textureSampleLevel(mask, samp, uv, 0.0).r;
146
+ if (params.mode == 1u) {
147
+ return vec4(m, m, m, 1.0);
148
+ }
149
+ return textureSampleLevel(picture, samp, uv, 0.0) * m;
150
+ }
151
+ `;
152
+ var MatteCompositor = class {
153
+ pipeline;
154
+ sampler;
155
+ uniforms = [];
156
+ uniformIndex = 0;
157
+ /** One 8-bit mask texture per node, rewritten each frame. */
158
+ masks = /* @__PURE__ */ new Map();
159
+ constructor(device, format) {
160
+ this.device = device;
161
+ const module = device.createShaderModule({
162
+ label: "vision_matte_composite",
163
+ code: WGSL
164
+ });
165
+ this.pipeline = device.createRenderPipeline({
166
+ label: "vision_matte_composite",
167
+ layout: "auto",
168
+ vertex: {
169
+ module,
170
+ entryPoint: "vs"
171
+ },
172
+ fragment: {
173
+ module,
174
+ entryPoint: "fs",
175
+ targets: [{ format }]
176
+ },
177
+ primitive: { topology: "triangle-list" }
178
+ });
179
+ this.sampler = device.createSampler({
180
+ magFilter: "linear",
181
+ minFilter: "linear"
182
+ });
183
+ }
184
+ /** Uploads a frame-sized 0..255 mask for `key` and returns its texture. */
185
+ uploadMask(key, mask, width, height) {
186
+ let tex = this.masks.get(key);
187
+ if (!tex || tex.width !== width || tex.height !== height) {
188
+ tex?.destroy();
189
+ tex = this.device.createTexture({
190
+ label: `vision_mask_${key}`,
191
+ size: [width, height],
192
+ format: "r8unorm",
193
+ usage: GPUTextureUsage.TEXTURE_BINDING | GPUTextureUsage.COPY_DST
194
+ });
195
+ this.masks.set(key, tex);
196
+ }
197
+ const aligned = Math.ceil(width / 256) * 256;
198
+ let bytes = mask;
199
+ if (aligned !== width) {
200
+ bytes = new Uint8Array(aligned * height);
201
+ for (let y = 0; y < height; y++) bytes.set(mask.subarray(y * width, (y + 1) * width), y * aligned);
202
+ }
203
+ this.device.queue.writeTexture({ texture: tex }, bytes.buffer, {
204
+ offset: bytes.byteOffset,
205
+ bytesPerRow: aligned,
206
+ rowsPerImage: height
207
+ }, {
208
+ width,
209
+ height
210
+ });
211
+ return tex;
212
+ }
213
+ /** Clears `target` and draws `picture` through `mask` into it. */
214
+ draw(encoder, target, picture, mask, mode, bounds) {
215
+ const w = mask.width;
216
+ const h = mask.height;
217
+ const data = new Float32Array(8);
218
+ const view = new Uint32Array(data.buffer);
219
+ view[0] = mode === "matte" ? 0 : mode === "mask" ? 1 : 2;
220
+ if (mode === "crop" && bounds) {
221
+ data[2] = bounds.x0 / w;
222
+ data[3] = bounds.y0 / h;
223
+ data[4] = (bounds.x1 + 1) / w;
224
+ data[5] = (bounds.y1 + 1) / h;
225
+ } else {
226
+ data[2] = 0;
227
+ data[3] = 0;
228
+ data[4] = 1;
229
+ data[5] = 1;
230
+ if (mode === "crop") view[0] = 0;
231
+ }
232
+ if (this.uniformIndex >= this.uniforms.length) this.uniforms.push(this.device.createBuffer({
233
+ size: data.byteLength,
234
+ usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST
235
+ }));
236
+ const uniform = this.uniforms[this.uniformIndex];
237
+ this.uniformIndex = (this.uniformIndex + 1) % 16;
238
+ this.device.queue.writeBuffer(uniform, 0, data);
239
+ const bindGroup = this.device.createBindGroup({
240
+ layout: this.pipeline.getBindGroupLayout(0),
241
+ entries: [
242
+ {
243
+ binding: 0,
244
+ resource: { buffer: uniform }
245
+ },
246
+ {
247
+ binding: 1,
248
+ resource: picture.createView()
249
+ },
250
+ {
251
+ binding: 2,
252
+ resource: mask.createView()
253
+ },
254
+ {
255
+ binding: 3,
256
+ resource: this.sampler
257
+ }
258
+ ]
259
+ });
260
+ const pass = encoder.beginRenderPass({
261
+ label: "vision_matte_composite",
262
+ colorAttachments: [{
263
+ view: target,
264
+ loadOp: "clear",
265
+ storeOp: "store",
266
+ clearValue: {
267
+ r: 0,
268
+ g: 0,
269
+ b: 0,
270
+ a: 0
271
+ }
272
+ }]
273
+ });
274
+ pass.setPipeline(this.pipeline);
275
+ pass.setBindGroup(0, bindGroup);
276
+ pass.draw(3);
277
+ pass.end();
278
+ }
279
+ };
280
+
281
+ //#endregion
282
+ //#region src/renderers/webgpu-renderer.ts
283
+ let sharedRunner = null;
284
+ let sharedSkeletonRenderer = null;
285
+ let sharedCompositor = null;
286
+ let sharedCompositorFormat = null;
287
+ /** Analysed frames, shared by every vision node in the process. */
288
+ const frameCache = new VisionFrameCache();
289
+ /** The frame each node's child texture holds (drawn on its last call). */
290
+ const heldFrameByNode = /* @__PURE__ */ new Map();
291
+ /** Per node: a copy of the held frame, and the composited matte. */
292
+ const heldTextures = /* @__PURE__ */ new Map();
293
+ const matteTextures = /* @__PURE__ */ new Map();
294
+ /** One tracker per vision node — track ids must not mix across nodes/sources. */
295
+ const objectTrackers = /* @__PURE__ */ new Map();
296
+ /**
297
+ * Persistent per-node child textures. The shared frame encoder is submitted at
298
+ * the END of the frame, so a node renderer cannot read back the child it just
299
+ * drew. We keep the texture around and read back the PREVIOUS frame's pixels at
300
+ * the start of the next call — a stable one-frame delay instead of the
301
+ * unpredictable multi-frame lag you get from pooled textures.
302
+ *
303
+ * Keyed on the stable Vision `Effect` instance (render ids change every frame),
304
+ * with a size-keyed fallback for raw operation nodes.
305
+ */
306
+ const childTexturesByEffect = /* @__PURE__ */ new WeakMap();
307
+ const childTexturesByKey = /* @__PURE__ */ new Map();
308
+ /**
309
+ * Last successful subject mask, reused for a few frames when the model misses —
310
+ * a transient miss then holds the silhouette instead of flashing the raw plate.
311
+ * Keyed per node/effect so multiple vision nodes do not collide.
312
+ */
313
+ const lastSubjectByNode = /* @__PURE__ */ new Map();
314
+ const SUBJECT_HOLD_FRAMES = 3;
315
+ /** A per-node texture shaped like `like`, recreated when the size changes. */
316
+ function nodeTexture(textures, device, key, like, usage, label) {
317
+ let tex = textures.get(key);
318
+ if (!tex || tex.width !== like.width || tex.height !== like.height || tex.format !== like.format) {
319
+ tex?.destroy();
320
+ tex = device.createTexture({
321
+ size: [like.width, like.height],
322
+ format: like.format,
323
+ usage,
324
+ label
325
+ });
326
+ textures.set(key, tex);
327
+ }
328
+ return tex;
329
+ }
330
+ const heldTexture = (device, key, like) => nodeTexture(heldTextures, device, key, like, GPUTextureUsage.COPY_DST | GPUTextureUsage.TEXTURE_BINDING, "vision_held_frame");
331
+ const matteTexture = (device, key, like) => nodeTexture(matteTextures, device, key, like, GPUTextureUsage.RENDER_ATTACHMENT | GPUTextureUsage.TEXTURE_BINDING | GPUTextureUsage.COPY_SRC, "vision_matte");
332
+ /** Everything that decides a frame's results besides the frame itself. */
333
+ function cacheSource(nodeKey, child, op, width, height) {
334
+ const c = child;
335
+ return JSON.stringify([
336
+ nodeKey,
337
+ c?.operation?.id ?? c?.id ?? null,
338
+ width,
339
+ height,
340
+ op.mode ?? "passthrough",
341
+ op.variant,
342
+ op.confidence,
343
+ op.classes,
344
+ op.maskThreshold,
345
+ op.featherRadius,
346
+ op.matteSource,
347
+ op.keyBackground === true,
348
+ op.maxMissedFrames
349
+ ]);
350
+ }
351
+ /** The background-key threshold at `frame` (it may be animated). */
352
+ function backgroundKeyThreshold(op, frame, fps) {
353
+ const raw = op.backgroundKeyThreshold;
354
+ if (typeof raw === "number") return raw;
355
+ if (typeof raw?.get === "function") return Number(raw.get({
356
+ frame,
357
+ fps
358
+ }));
359
+ if (typeof raw?._value === "number") return Number(raw._value);
360
+ return 70;
361
+ }
362
+ /**
363
+ * The model's subject, or the last one through brief misses: a transient miss
364
+ * then holds the silhouette instead of flashing the raw plate.
365
+ */
366
+ function selectSubject(selected, nodeKey, frame) {
367
+ if (selected) {
368
+ lastSubjectByNode.set(nodeKey, {
369
+ mask: selected,
370
+ frame
371
+ });
372
+ return selected;
373
+ }
374
+ const last = lastSubjectByNode.get(nodeKey);
375
+ if (last && frame - last.frame <= SUBJECT_HOLD_FRAMES) return last.mask;
376
+ }
377
+ /**
378
+ * Lazy shared runner — `create()` is a pure constructor (zero I/O); the first frame that
379
+ * requests a task triggers that task's model download at inference time (never at init).
380
+ * Recreated only when an option that changes inference changes.
381
+ */
382
+ function getSharedRunner(op) {
383
+ const options = {
384
+ variant: op.variant,
385
+ confidence: op.confidence,
386
+ classes: op.classes,
387
+ maskThreshold: op.maskThreshold,
388
+ featherRadius: op.featherRadius,
389
+ modelsDir: op.modelsDir,
390
+ baseUrl: op.baseUrl
391
+ };
392
+ const key = JSON.stringify(options);
393
+ if (sharedRunner?.key !== key) {
394
+ sharedRunner?.runner.close();
395
+ sharedRunner = {
396
+ key,
397
+ runner: VisionRunner.create(options)
398
+ };
399
+ }
400
+ return sharedRunner.runner;
401
+ }
402
+ function getObjectTracker(nodeKey, op) {
403
+ let tracker = objectTrackers.get(nodeKey);
404
+ if (!tracker) {
405
+ tracker = new TemporalObjectTracker({ maxMissedFrames: op.maxMissedFrames ?? 15 });
406
+ objectTrackers.set(nodeKey, tracker);
407
+ }
408
+ return tracker;
409
+ }
410
+ /**
411
+ * Normalizes the runner's pose keypoints (plate pixel space) to [0, 1] landmarks, which is
412
+ * the contract the signal bundle and the skeleton pipeline expect.
413
+ */
414
+ function normalizePoseKeypoints(result, width, height) {
415
+ if (width <= 0 || height <= 0) return result;
416
+ return { people: result.people.map((person) => ({
417
+ ...person,
418
+ keypoints: person.keypoints.map((k) => ({
419
+ x: k.x / width,
420
+ y: k.y / height,
421
+ visibility: k.visibility
422
+ }))
423
+ })) };
424
+ }
425
+ /** Assigns each decoded instance mask to the temporal track whose box center is nearest the mask's centroid. */
426
+ function assignMasksToTracks(masks, tracked) {
427
+ return masks.map((mask) => {
428
+ const { cx, cy } = maskCentroid(mask);
429
+ let best = null;
430
+ for (const obj of tracked) {
431
+ const dx = obj.centerX - cx;
432
+ const dy = obj.centerY - cy;
433
+ const distance = dx * dx + dy * dy;
434
+ if (!best || distance < best.distance) best = {
435
+ trackId: obj.trackId,
436
+ distance
437
+ };
438
+ }
439
+ return best ? {
440
+ ...mask,
441
+ trackId: best.trackId
442
+ } : mask;
443
+ });
444
+ }
445
+ /**
446
+ * Grows a subject mask into connected foreground pixels — seeding a flood fill
447
+ * from the subject through pixels that differ from the sampled backdrop recovers
448
+ * thin or fast-moving edges (hair, fabric) without adding a near-uniform studio
449
+ * wall or its soft grey shadow.
450
+ */
451
+ function fillInternalHoles(mask, width, height) {
452
+ const n = width * height;
453
+ const exterior = new Uint8Array(n);
454
+ const queue = new Int32Array(n);
455
+ let tail = 0;
456
+ const seed = (i) => {
457
+ if (exterior[i] || mask[i] > 0) return;
458
+ exterior[i] = 1;
459
+ queue[tail++] = i;
460
+ };
461
+ for (let x = 0; x < width; x++) {
462
+ seed(x);
463
+ seed((height - 1) * width + x);
464
+ }
465
+ for (let y = 0; y < height; y++) {
466
+ seed(y * width);
467
+ seed(y * width + width - 1);
468
+ }
469
+ let head = 0;
470
+ while (head < tail) {
471
+ const i = queue[head++];
472
+ const x = i % width;
473
+ let j = i + 1;
474
+ if (x + 1 < width && !exterior[j] && mask[j] === 0) {
475
+ exterior[j] = 1;
476
+ queue[tail++] = j;
477
+ }
478
+ j = i - 1;
479
+ if (x > 0 && !exterior[j] && mask[j] === 0) {
480
+ exterior[j] = 1;
481
+ queue[tail++] = j;
482
+ }
483
+ j = i + width;
484
+ if (j < n && !exterior[j] && mask[j] === 0) {
485
+ exterior[j] = 1;
486
+ queue[tail++] = j;
487
+ }
488
+ j = i - width;
489
+ if (j >= 0 && !exterior[j] && mask[j] === 0) {
490
+ exterior[j] = 1;
491
+ queue[tail++] = j;
492
+ }
493
+ }
494
+ for (let i = 0; i < n; i++) if (mask[i] === 0 && exterior[i] === 0) mask[i] = 255;
495
+ }
496
+ function smoothMaskBoundary(mask, width, height, radius = 2) {
497
+ if (radius <= 0) return mask;
498
+ const temp = new Uint8Array(width * height);
499
+ const out = new Uint8Array(width * height);
500
+ const div = 2 * radius + 1;
501
+ for (let y = 0; y < height; y++) {
502
+ const row = y * width;
503
+ let sum = 0;
504
+ for (let k = -radius; k <= radius; k++) {
505
+ const x = Math.max(0, Math.min(width - 1, k));
506
+ sum += mask[row + x];
507
+ }
508
+ for (let x = 0; x < width; x++) {
509
+ temp[row + x] = Math.round(sum / div);
510
+ const xOut = Math.max(0, x - radius);
511
+ const xIn = Math.min(width - 1, x + radius + 1);
512
+ sum += mask[row + xIn] - mask[row + xOut];
513
+ }
514
+ }
515
+ const sums = new Float64Array(width);
516
+ for (let k = -radius; k <= radius; k++) {
517
+ const row = Math.max(0, Math.min(height - 1, k)) * width;
518
+ for (let x = 0; x < width; x++) sums[x] += temp[row + x];
519
+ }
520
+ for (let y = 0; y < height; y++) {
521
+ const row = y * width;
522
+ const rowOut = Math.max(0, y - radius) * width;
523
+ const rowIn = Math.min(height - 1, y + radius + 1) * width;
524
+ for (let x = 0; x < width; x++) {
525
+ out[row + x] = Math.round(sums[x] / div);
526
+ sums[x] += temp[rowIn + x] - temp[rowOut + x];
527
+ }
528
+ }
529
+ return out;
530
+ }
531
+ function growMaskIntoForeground(mask, width, height, pixels, threshold, featherRadius) {
532
+ let br = 0;
533
+ let bg = 0;
534
+ let bb = 0;
535
+ let count = 0;
536
+ const sample = (x, y) => {
537
+ const i = y * width + x;
538
+ if (mask[i]) return;
539
+ const p = i * 4;
540
+ br += pixels[p];
541
+ bg += pixels[p + 1];
542
+ bb += pixels[p + 2];
543
+ count++;
544
+ };
545
+ for (let x = 0; x < width; x += 4) {
546
+ sample(x, 0);
547
+ sample(x, height - 1);
548
+ }
549
+ for (let y = 0; y < height; y += 4) {
550
+ sample(0, y);
551
+ sample(width - 1, y);
552
+ }
553
+ if (count === 0) return mask;
554
+ br /= count;
555
+ bg /= count;
556
+ bb /= count;
557
+ const n = width * height;
558
+ const fg = new Uint8Array(n);
559
+ for (let i = 0, p = 0; i < n; i++, p += 4) fg[i] = Math.max(Math.abs(pixels[p] - br), Math.abs(pixels[p + 1] - bg), Math.abs(pixels[p + 2] - bb)) > threshold ? 1 : 0;
560
+ const queue = new Int32Array(n);
561
+ let tail = 0;
562
+ for (let i = 0; i < n; i++) if (mask[i] > 0) {
563
+ if (fg[i]) mask[i] = 255;
564
+ queue[tail++] = i;
565
+ }
566
+ let head = 0;
567
+ while (head < tail) {
568
+ const i = queue[head++];
569
+ const x = i % width;
570
+ let j = i + 1;
571
+ if (x + 1 < width && mask[j] !== 255 && fg[j]) {
572
+ mask[j] = 255;
573
+ queue[tail++] = j;
574
+ }
575
+ j = i - 1;
576
+ if (x > 0 && mask[j] !== 255 && fg[j]) {
577
+ mask[j] = 255;
578
+ queue[tail++] = j;
579
+ }
580
+ j = i + width;
581
+ if (j < n && mask[j] !== 255 && fg[j]) {
582
+ mask[j] = 255;
583
+ queue[tail++] = j;
584
+ }
585
+ j = i - width;
586
+ if (j >= 0 && mask[j] !== 255 && fg[j]) {
587
+ mask[j] = 255;
588
+ queue[tail++] = j;
589
+ }
590
+ }
591
+ fillInternalHoles(mask, width, height);
592
+ return smoothMaskBoundary(mask, width, height, featherRadius !== void 0 && featherRadius > 0 ? Math.max(1, Math.min(8, Math.round(featherRadius * 50))) : 2);
593
+ }
594
+ function maskCentroid(mask) {
595
+ const { mask: data, width, height } = mask;
596
+ let sumX = 0;
597
+ let sumY = 0;
598
+ let count = 0;
599
+ for (let y = 0; y < height; y++) {
600
+ const row = y * width;
601
+ for (let x = 0; x < width; x++) if (data[row + x] > 0) {
602
+ sumX += x;
603
+ sumY += y;
604
+ count++;
605
+ }
606
+ }
607
+ if (count === 0) return {
608
+ cx: 0,
609
+ cy: 0
610
+ };
611
+ return {
612
+ cx: sumX / count,
613
+ cy: sumY / count
614
+ };
615
+ }
616
+ const VisionWebGPURenderer = async (args) => {
617
+ const { ctx, encoder, pass, targetView, targetWidth, targetHeight, props, drawChild } = args;
618
+ const { virtualMedia } = props;
619
+ const rawOp = virtualMedia?.operation;
620
+ if (rawOp?.op !== "Vision") return;
621
+ const op = rawOp;
622
+ pass.end();
623
+ const childMedia = virtualMedia?.children?.[0];
624
+ if (!childMedia) return;
625
+ const effectKey = op.effect;
626
+ const hasEffectKey = effectKey !== void 0 && effectKey !== null;
627
+ const nodeKeyStr = virtualMedia?.id ?? rawOp?.id ?? `${targetWidth}x${targetHeight}_${op.mode}_${op.variant}_${op.keyBackground}_${op.backgroundKeyThreshold}`;
628
+ const fallbackKey = `vision-${nodeKeyStr}`;
629
+ let childTex = hasEffectKey ? childTexturesByEffect.get(effectKey) : childTexturesByKey.get(fallbackKey);
630
+ if (childTex && (childTex.width !== targetWidth || childTex.height !== targetHeight)) {
631
+ childTex.destroy();
632
+ childTex = void 0;
633
+ if (hasEffectKey) childTexturesByEffect.delete(effectKey);
634
+ else childTexturesByKey.delete(fallbackKey);
635
+ }
636
+ const hasPreviousFrame = childTex !== void 0;
637
+ if (!childTex) {
638
+ childTex = ctx.device.createTexture({
639
+ size: [targetWidth, targetHeight],
640
+ format: ctx.renderer.format,
641
+ usage: GPUTextureUsage.RENDER_ATTACHMENT | GPUTextureUsage.TEXTURE_BINDING | GPUTextureUsage.COPY_SRC,
642
+ label: "vision_child_persistent"
643
+ });
644
+ if (hasEffectKey) childTexturesByEffect.set(effectKey, childTex);
645
+ else childTexturesByKey.set(fallbackKey, childTex);
646
+ if (childTexturesByKey.size > 4) {
647
+ const oldest = childTexturesByKey.keys().next().value;
648
+ if (oldest && oldest !== fallbackKey) {
649
+ childTexturesByKey.get(oldest)?.destroy();
650
+ childTexturesByKey.delete(oldest);
651
+ }
652
+ }
653
+ }
654
+ const frameIdx = props.frame ?? 0;
655
+ const fps = props.fps ?? 24;
656
+ const mode = op.mode ?? "passthrough";
657
+ const isMatteMode = mode === "mask" || mode === "matte" || mode === "crop";
658
+ const visionBundle = op.visionBundle ?? virtualMedia.visionBundle;
659
+ const wantsDetection = op.enableDetection !== false || mode === "boxes" || mode === "tracking";
660
+ const wantsSegmentation = op.enableSegmentation === true || isMatteMode && op.matteSource !== "selfie";
661
+ const wantsSelfie = op.enableMatte === true || isMatteMode && op.matteSource === "selfie";
662
+ const wantsPose = op.enablePose === true || mode === "skeleton";
663
+ const cacheable = !wantsSelfie && !wantsPose && (wantsDetection || wantsSegmentation);
664
+ const source = cacheable ? cacheSource(nodeKeyStr, childMedia, op, targetWidth, targetHeight) : "";
665
+ const keyThresholdAt = (frame) => op.keyBackground === true ? backgroundKeyThreshold(op, frame, fps) : void 0;
666
+ const lookup = (frame) => {
667
+ if (!cacheable || frame === void 0) return void 0;
668
+ const hit = frameCache.get(source, frame);
669
+ if (!hit || hit.keyThreshold !== keyThresholdAt(frame)) return void 0;
670
+ if (wantsSegmentation && visionBundle && !hit.masks) return void 0;
671
+ return hit;
672
+ };
673
+ const heldFrame = hasPreviousFrame ? heldFrameByNode.get(nodeKeyStr) : void 0;
674
+ let cached = lookup(frameIdx);
675
+ const exact = cached !== void 0;
676
+ if (!cached) cached = lookup(heldFrame);
677
+ let heldTex;
678
+ if (isMatteMode && hasPreviousFrame && !exact) {
679
+ heldTex = heldTexture(ctx.device, nodeKeyStr, childTex);
680
+ encoder.copyTextureToTexture({ texture: childTex }, { texture: heldTex }, [
681
+ targetWidth,
682
+ targetHeight,
683
+ 1
684
+ ]);
685
+ }
686
+ let framePixels = null;
687
+ if (hasPreviousFrame && !cached) {
688
+ const unalignedBytesPerRow = targetWidth * 4;
689
+ const bytesPerRow = Math.ceil(unalignedBytesPerRow / 256) * 256;
690
+ const bufferSize = bytesPerRow * targetHeight;
691
+ const stagingBuffer = ctx.device.createBuffer({
692
+ size: bufferSize,
693
+ usage: GPUBufferUsage.COPY_DST | GPUBufferUsage.MAP_READ,
694
+ label: "vision_frame_staging"
695
+ });
696
+ const readbackEncoder = ctx.device.createCommandEncoder({ label: "vision_readback_encoder" });
697
+ readbackEncoder.copyTextureToBuffer({ texture: childTex }, {
698
+ buffer: stagingBuffer,
699
+ bytesPerRow,
700
+ rowsPerImage: targetHeight
701
+ }, [
702
+ targetWidth,
703
+ targetHeight,
704
+ 1
705
+ ]);
706
+ ctx.device.queue.submit([readbackEncoder.finish()]);
707
+ await stagingBuffer.mapAsync(GPUMapMode.READ);
708
+ const mappedBytes = new Uint8Array(stagingBuffer.getMappedRange());
709
+ framePixels = new Uint8ClampedArray(targetWidth * targetHeight * 4);
710
+ if (bytesPerRow === unalignedBytesPerRow) framePixels.set(mappedBytes.subarray(0, framePixels.length));
711
+ else for (let y = 0; y < targetHeight; y++) framePixels.set(mappedBytes.subarray(y * bytesPerRow, y * bytesPerRow + unalignedBytesPerRow), y * unalignedBytesPerRow);
712
+ stagingBuffer.unmap();
713
+ stagingBuffer.destroy();
714
+ }
715
+ const childView = childTex.createView();
716
+ ctx.renderer.beginFrame(encoder, childView, {
717
+ r: 0,
718
+ g: 0,
719
+ b: 0,
720
+ a: 0
721
+ }, targetWidth, targetHeight, "clear").end();
722
+ await drawChild(childMedia, { ...props }, childView, childTex, targetWidth, targetHeight);
723
+ heldFrameByNode.set(nodeKeyStr, frameIdx);
724
+ let frameObjects = [];
725
+ let frameMasks = [];
726
+ let personMatte;
727
+ let currentPoseRes;
728
+ let subject;
729
+ if (cached) {
730
+ frameObjects = cached.objects;
731
+ if (wantsDetection) visionBundle?.setObjectResult(frameIdx, {
732
+ objects: cached.objects,
733
+ rawDetections: cached.detections
734
+ });
735
+ if (wantsSegmentation && visionBundle && cached.masks) {
736
+ frameMasks = cached.masks.map(fromCachedMask);
737
+ visionBundle.setMaskResult(frameIdx, frameMasks);
738
+ }
739
+ if (cached.subject) {
740
+ subject = fromCachedMask(cached.subject.mask);
741
+ lastSubjectByNode.set(nodeKeyStr, {
742
+ mask: subject,
743
+ frame: frameIdx
744
+ });
745
+ }
746
+ } else if (framePixels) {
747
+ const runner = getSharedRunner(op);
748
+ const image = {
749
+ data: framePixels,
750
+ width: targetWidth,
751
+ height: targetHeight
752
+ };
753
+ let detections = [];
754
+ if (wantsDetection) {
755
+ detections = await runner.detect(image);
756
+ const tracker = getObjectTracker(nodeKeyStr, op);
757
+ if (frameIdx === 0) tracker.reset();
758
+ frameObjects = tracker.update(detections, frameIdx, fps);
759
+ visionBundle?.setObjectResult(frameIdx, {
760
+ objects: frameObjects,
761
+ rawDetections: detections
762
+ });
763
+ }
764
+ if (wantsSegmentation) {
765
+ frameMasks = assignMasksToTracks((await runner.segment(image)).masks, frameObjects);
766
+ visionBundle?.setMaskResult(frameIdx, frameMasks);
767
+ }
768
+ if (wantsSelfie) {
769
+ personMatte = await runner.matte(image);
770
+ visionBundle?.setMatteResult(frameIdx, personMatte);
771
+ }
772
+ if (wantsPose) {
773
+ currentPoseRes = { people: matchPoseToTracks(normalizePoseKeypoints(await runner.pose(image), image.width, image.height).people, frameObjects) };
774
+ visionBundle?.setPoseResult(frameIdx, currentPoseRes);
775
+ }
776
+ if (isMatteMode) {
777
+ subject = selectSubject(op.matteSource === "selfie" ? personMatteAsMask(personMatte) : mergeSubjectMask(frameMasks.length > 0 ? frameMasks : visionBundle?.getMaskResult(frameIdx) ?? []), nodeKeyStr, frameIdx);
778
+ const threshold = keyThresholdAt(heldFrame ?? frameIdx);
779
+ if (subject && threshold !== void 0) subject = {
780
+ ...subject,
781
+ mask: growMaskIntoForeground(subject.mask, subject.width, subject.height, framePixels, threshold, op.featherRadius)
782
+ };
783
+ }
784
+ if (cacheable && heldFrame !== void 0) frameCache.set(source, heldFrame, {
785
+ detections,
786
+ objects: frameObjects,
787
+ masks: visionBundle ? frameMasks.map(toCachedMask) : null,
788
+ subject: subject ? { mask: toCachedMask(subject) } : null,
789
+ keyThreshold: keyThresholdAt(heldFrame)
790
+ });
791
+ }
792
+ if (isMatteMode) {
793
+ const picture = exact ? childTex : heldTex;
794
+ if (!subject || !picture) {
795
+ ctx.renderer.beginFrame(encoder, targetView, {
796
+ r: 0,
797
+ g: 0,
798
+ b: 0,
799
+ a: 0
800
+ }, targetWidth, targetHeight, "clear").end();
801
+ return;
802
+ }
803
+ if (!sharedCompositor || sharedCompositorFormat !== ctx.renderer.format) {
804
+ sharedCompositor = new MatteCompositor(ctx.device, ctx.renderer.format);
805
+ sharedCompositorFormat = ctx.renderer.format;
806
+ }
807
+ const maskTex = sharedCompositor.uploadMask(nodeKeyStr, subject.mask, subject.width, subject.height);
808
+ const matteTex = matteTexture(ctx.device, nodeKeyStr, childTex);
809
+ sharedCompositor.draw(encoder, matteTex.createView(), picture, maskTex, mode, mode === "crop" ? maskBounds(subject) ?? void 0 : void 0);
810
+ visionBundle?.setStencilTexture(matteTex);
811
+ const outPass$1 = ctx.renderer.beginFrame(encoder, targetView, {
812
+ r: 0,
813
+ g: 0,
814
+ b: 0,
815
+ a: 0
816
+ }, targetWidth, targetHeight, "clear");
817
+ ctx.renderer.drawTexture(outPass$1, matteTex, {
818
+ x: 0,
819
+ y: 0,
820
+ width: targetWidth,
821
+ height: targetHeight
822
+ });
823
+ outPass$1.end();
824
+ return;
825
+ }
826
+ if (mode === "skeleton") {
827
+ if (!sharedSkeletonRenderer) sharedSkeletonRenderer = new PoseSkeletonRenderer(ctx.device);
828
+ const person = (visionBundle?.getPoseResult(frameIdx) ?? currentPoseRes)?.people[0];
829
+ if (person && person.keypoints.length > 0) {
830
+ const skelTex = sharedSkeletonRenderer.renderToTexture(person.keypoints, {
831
+ width: targetWidth,
832
+ height: targetHeight
833
+ });
834
+ const skelPass = ctx.renderer.beginFrame(encoder, targetView, {
835
+ r: 0,
836
+ g: 0,
837
+ b: 0,
838
+ a: 0
839
+ }, targetWidth, targetHeight, "clear");
840
+ ctx.renderer.drawTexture(skelPass, skelTex, {
841
+ x: 0,
842
+ y: 0,
843
+ width: targetWidth,
844
+ height: targetHeight
845
+ });
846
+ skelPass.end();
847
+ return;
848
+ }
849
+ }
850
+ if (mode === "boxes" || mode === "tracking") {
851
+ const outPass$1 = ctx.renderer.beginFrame(encoder, targetView, {
852
+ r: 0,
853
+ g: 0,
854
+ b: 0,
855
+ a: 0
856
+ }, targetWidth, targetHeight, "clear");
857
+ ctx.renderer.drawTexture(outPass$1, childTex, {
858
+ x: 0,
859
+ y: 0,
860
+ width: targetWidth,
861
+ height: targetHeight
862
+ });
863
+ const objects = frameObjects.length > 0 ? frameObjects : visionBundle?.getObjectResult(frameIdx).objects ?? [];
864
+ if (objects.length > 0) {
865
+ const boxColor = "#38bdf8";
866
+ const stroke = 2;
867
+ for (const obj of objects) {
868
+ if (!obj.active) continue;
869
+ const b = obj.boundingBox;
870
+ ctx.renderer.drawRect(outPass$1, {
871
+ x: b.originX,
872
+ y: b.originY,
873
+ width: b.width,
874
+ height: stroke
875
+ }, boxColor);
876
+ ctx.renderer.drawRect(outPass$1, {
877
+ x: b.originX,
878
+ y: b.originY + b.height - stroke,
879
+ width: b.width,
880
+ height: stroke
881
+ }, boxColor);
882
+ ctx.renderer.drawRect(outPass$1, {
883
+ x: b.originX,
884
+ y: b.originY,
885
+ width: stroke,
886
+ height: b.height
887
+ }, boxColor);
888
+ ctx.renderer.drawRect(outPass$1, {
889
+ x: b.originX + b.width - stroke,
890
+ y: b.originY,
891
+ width: stroke,
892
+ height: b.height
893
+ }, boxColor);
894
+ const cornerLen = Math.min(16, b.width / 4, b.height / 4);
895
+ ctx.renderer.drawRect(outPass$1, {
896
+ x: b.originX,
897
+ y: b.originY,
898
+ width: cornerLen,
899
+ height: stroke * 2
900
+ }, "#ffffff");
901
+ ctx.renderer.drawRect(outPass$1, {
902
+ x: b.originX,
903
+ y: b.originY,
904
+ width: stroke * 2,
905
+ height: cornerLen
906
+ }, "#ffffff");
907
+ if (mode === "tracking") ctx.renderer.drawRect(outPass$1, {
908
+ x: obj.centerX - 3,
909
+ y: obj.centerY - 3,
910
+ width: 6,
911
+ height: 6
912
+ }, "#ef4444", 3);
913
+ }
914
+ }
915
+ outPass$1.end();
916
+ return;
917
+ }
918
+ const outPass = ctx.renderer.beginFrame(encoder, targetView, {
919
+ r: 0,
920
+ g: 0,
921
+ b: 0,
922
+ a: 0
923
+ }, targetWidth, targetHeight, "clear");
924
+ ctx.renderer.drawTexture(outPass, childTex, {
925
+ x: 0,
926
+ y: 0,
927
+ width: targetWidth,
928
+ height: targetHeight
929
+ });
930
+ outPass.end();
931
+ };
932
+ /** Adapts a Selfie Segmenter matte to the instance-mask shape the composite path takes. */
933
+ function personMatteAsMask(matte) {
934
+ if (!matte || matte.coverage <= 0) return void 0;
935
+ const area = Math.round(matte.coverage * matte.width * matte.height);
936
+ return {
937
+ category: "person",
938
+ mask: matte.mask,
939
+ width: matte.width,
940
+ height: matte.height,
941
+ area,
942
+ coverage: matte.coverage,
943
+ detectionIndex: -1
944
+ };
945
+ }
946
+
947
+ //#endregion
948
+ //#region src/renderers/index.ts
949
+ var renderers_default = defineRenderer({ WebGPURenderer: VisionWebGPURenderer });
950
+
951
+ //#endregion
952
+ export { renderers_default as t };
953
+ //# sourceMappingURL=renderers-B24hckhO.mjs.map