@framefields/node-vision 2.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.turbo/turbo-build.log +20 -0
- package/CHANGELOG.md +23 -0
- package/LICENCE +202 -0
- package/dist/index.d.mts +123 -0
- package/dist/index.d.mts.map +1 -0
- package/dist/index.mjs +73 -0
- package/dist/index.mjs.map +1 -0
- package/dist/renderer.d.mts +7 -0
- package/dist/renderer.d.mts.map +1 -0
- package/dist/renderer.mjs +3 -0
- package/dist/renderers-B24hckhO.mjs +953 -0
- package/dist/renderers-B24hckhO.mjs.map +1 -0
- package/package.json +48 -0
- package/src/index.ts +2 -0
- package/src/renderers/frame-cache.ts +149 -0
- package/src/renderers/index.ts +6 -0
- package/src/renderers/matte-compositor.ts +189 -0
- package/src/renderers/webgpu-renderer.ts +997 -0
- package/src/shared/config.ts +88 -0
- package/src/shared/index.ts +1 -0
- package/tsconfig.json +11 -0
- package/tsdown.config.ts +13 -0
|
@@ -0,0 +1,953 @@
|
|
|
1
|
+
import { defineRenderer } from "@framefields/node-sdk/renderer";
|
|
2
|
+
import { PoseSkeletonRenderer, TemporalObjectTracker, VisionRunner, maskBounds, matchPoseToTracks, mergeSubjectMask } from "@framefields/vision";
|
|
3
|
+
|
|
4
|
+
//#region src/renderers/frame-cache.ts
|
|
5
|
+
let scratchValues = new Uint8Array(0);
|
|
6
|
+
let scratchLengths = new Uint32Array(0);
|
|
7
|
+
function encodeMask(mask, width, height) {
|
|
8
|
+
if (scratchValues.length < mask.length) {
|
|
9
|
+
scratchValues = new Uint8Array(mask.length);
|
|
10
|
+
scratchLengths = new Uint32Array(mask.length);
|
|
11
|
+
}
|
|
12
|
+
const values = scratchValues;
|
|
13
|
+
const lengths = scratchLengths;
|
|
14
|
+
let runs = 0;
|
|
15
|
+
let start = 0;
|
|
16
|
+
for (let i = 1; i <= mask.length; i++) if (i === mask.length || mask[i] !== mask[start]) {
|
|
17
|
+
values[runs] = mask[start];
|
|
18
|
+
lengths[runs] = i - start;
|
|
19
|
+
runs++;
|
|
20
|
+
start = i;
|
|
21
|
+
}
|
|
22
|
+
return {
|
|
23
|
+
width,
|
|
24
|
+
height,
|
|
25
|
+
values: values.slice(0, runs),
|
|
26
|
+
lengths: lengths.slice(0, runs)
|
|
27
|
+
};
|
|
28
|
+
}
|
|
29
|
+
function decodeMask(rle) {
|
|
30
|
+
const out = new Uint8Array(rle.width * rle.height);
|
|
31
|
+
let at = 0;
|
|
32
|
+
for (let r = 0; r < rle.values.length; r++) {
|
|
33
|
+
const end = at + rle.lengths[r];
|
|
34
|
+
if (rle.values[r] !== 0) out.fill(rle.values[r], at, end);
|
|
35
|
+
at = end;
|
|
36
|
+
}
|
|
37
|
+
return out;
|
|
38
|
+
}
|
|
39
|
+
function rleBytes(rle) {
|
|
40
|
+
return rle.values.byteLength + rle.lengths.byteLength;
|
|
41
|
+
}
|
|
42
|
+
function toCachedMask(mask) {
|
|
43
|
+
const { mask: data, ...rest } = mask;
|
|
44
|
+
return {
|
|
45
|
+
...rest,
|
|
46
|
+
rle: encodeMask(data, mask.width, mask.height)
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
function fromCachedMask(mask) {
|
|
50
|
+
const { rle, ...rest } = mask;
|
|
51
|
+
return {
|
|
52
|
+
...rest,
|
|
53
|
+
mask: decodeMask(rle)
|
|
54
|
+
};
|
|
55
|
+
}
|
|
56
|
+
function resultBytes(result) {
|
|
57
|
+
let bytes = 512 + result.detections.length * 128 + result.objects.length * 256;
|
|
58
|
+
for (const m of result.masks ?? []) bytes += rleBytes(m.rle);
|
|
59
|
+
if (result.subject) bytes += rleBytes(result.subject.mask.rle);
|
|
60
|
+
return bytes;
|
|
61
|
+
}
|
|
62
|
+
/** Results by source (node + inference settings) and frame, least recently used out first. */
|
|
63
|
+
var VisionFrameCache = class {
|
|
64
|
+
entries = /* @__PURE__ */ new Map();
|
|
65
|
+
bytes = 0;
|
|
66
|
+
constructor(maxBytes = 256 * 1024 * 1024) {
|
|
67
|
+
this.maxBytes = maxBytes;
|
|
68
|
+
}
|
|
69
|
+
get(source, frame) {
|
|
70
|
+
const key = `${source}#${frame}`;
|
|
71
|
+
const hit = this.entries.get(key);
|
|
72
|
+
if (hit) {
|
|
73
|
+
this.entries.delete(key);
|
|
74
|
+
this.entries.set(key, hit);
|
|
75
|
+
}
|
|
76
|
+
return hit;
|
|
77
|
+
}
|
|
78
|
+
set(source, frame, result) {
|
|
79
|
+
const key = `${source}#${frame}`;
|
|
80
|
+
const old = this.entries.get(key);
|
|
81
|
+
if (old) {
|
|
82
|
+
this.bytes -= resultBytes(old);
|
|
83
|
+
this.entries.delete(key);
|
|
84
|
+
}
|
|
85
|
+
this.entries.set(key, result);
|
|
86
|
+
this.bytes += resultBytes(result);
|
|
87
|
+
for (const [k, v] of this.entries) {
|
|
88
|
+
if (this.bytes <= this.maxBytes) break;
|
|
89
|
+
this.entries.delete(k);
|
|
90
|
+
this.bytes -= resultBytes(v);
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
clear() {
|
|
94
|
+
this.entries.clear();
|
|
95
|
+
this.bytes = 0;
|
|
96
|
+
}
|
|
97
|
+
get size() {
|
|
98
|
+
return this.entries.size;
|
|
99
|
+
}
|
|
100
|
+
};
|
|
101
|
+
|
|
102
|
+
//#endregion
|
|
103
|
+
//#region src/renderers/matte-compositor.ts
|
|
104
|
+
/**
|
|
105
|
+
* GPU composite for the mask-family modes: the picture times the subject mask,
|
|
106
|
+
* in one full-screen pass. Replaces building the RGBA result on the CPU and
|
|
107
|
+
* uploading it; only the 8-bit mask goes to the GPU.
|
|
108
|
+
*
|
|
109
|
+
* - matte: picture × mask (premultiplied, so the background is transparent)
|
|
110
|
+
* - mask: white silhouette (rgb = mask, a = 1)
|
|
111
|
+
* - crop: matte, with the subject's bounding box stretched to the frame
|
|
112
|
+
*/
|
|
113
|
+
const WGSL = `
|
|
114
|
+
struct Params {
|
|
115
|
+
mode: u32, // 0 matte, 1 mask, 2 crop
|
|
116
|
+
_pad: u32,
|
|
117
|
+
cropMin: vec2<f32>, // uv of the subject bounds (crop)
|
|
118
|
+
cropMax: vec2<f32>,
|
|
119
|
+
_pad2: vec2<f32>,
|
|
120
|
+
};
|
|
121
|
+
|
|
122
|
+
@group(0) @binding(0) var<uniform> params: Params;
|
|
123
|
+
@group(0) @binding(1) var picture: texture_2d<f32>;
|
|
124
|
+
@group(0) @binding(2) var mask: texture_2d<f32>;
|
|
125
|
+
@group(0) @binding(3) var samp: sampler;
|
|
126
|
+
|
|
127
|
+
struct VSOut {
|
|
128
|
+
@builtin(position) pos: vec4<f32>,
|
|
129
|
+
@location(0) uv: vec2<f32>,
|
|
130
|
+
};
|
|
131
|
+
|
|
132
|
+
@vertex fn vs(@builtin(vertex_index) i: u32) -> VSOut {
|
|
133
|
+
var p = array<vec2<f32>, 3>(vec2(-1.0, -1.0), vec2(3.0, -1.0), vec2(-1.0, 3.0));
|
|
134
|
+
var out: VSOut;
|
|
135
|
+
out.pos = vec4(p[i], 0.0, 1.0);
|
|
136
|
+
out.uv = vec2(p[i].x * 0.5 + 0.5, 0.5 - p[i].y * 0.5);
|
|
137
|
+
return out;
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
@fragment fn fs(in: VSOut) -> @location(0) vec4<f32> {
|
|
141
|
+
var uv = in.uv;
|
|
142
|
+
if (params.mode == 2u) {
|
|
143
|
+
uv = mix(params.cropMin, params.cropMax, uv);
|
|
144
|
+
}
|
|
145
|
+
let m = textureSampleLevel(mask, samp, uv, 0.0).r;
|
|
146
|
+
if (params.mode == 1u) {
|
|
147
|
+
return vec4(m, m, m, 1.0);
|
|
148
|
+
}
|
|
149
|
+
return textureSampleLevel(picture, samp, uv, 0.0) * m;
|
|
150
|
+
}
|
|
151
|
+
`;
|
|
152
|
+
var MatteCompositor = class {
|
|
153
|
+
pipeline;
|
|
154
|
+
sampler;
|
|
155
|
+
uniforms = [];
|
|
156
|
+
uniformIndex = 0;
|
|
157
|
+
/** One 8-bit mask texture per node, rewritten each frame. */
|
|
158
|
+
masks = /* @__PURE__ */ new Map();
|
|
159
|
+
constructor(device, format) {
|
|
160
|
+
this.device = device;
|
|
161
|
+
const module = device.createShaderModule({
|
|
162
|
+
label: "vision_matte_composite",
|
|
163
|
+
code: WGSL
|
|
164
|
+
});
|
|
165
|
+
this.pipeline = device.createRenderPipeline({
|
|
166
|
+
label: "vision_matte_composite",
|
|
167
|
+
layout: "auto",
|
|
168
|
+
vertex: {
|
|
169
|
+
module,
|
|
170
|
+
entryPoint: "vs"
|
|
171
|
+
},
|
|
172
|
+
fragment: {
|
|
173
|
+
module,
|
|
174
|
+
entryPoint: "fs",
|
|
175
|
+
targets: [{ format }]
|
|
176
|
+
},
|
|
177
|
+
primitive: { topology: "triangle-list" }
|
|
178
|
+
});
|
|
179
|
+
this.sampler = device.createSampler({
|
|
180
|
+
magFilter: "linear",
|
|
181
|
+
minFilter: "linear"
|
|
182
|
+
});
|
|
183
|
+
}
|
|
184
|
+
/** Uploads a frame-sized 0..255 mask for `key` and returns its texture. */
|
|
185
|
+
uploadMask(key, mask, width, height) {
|
|
186
|
+
let tex = this.masks.get(key);
|
|
187
|
+
if (!tex || tex.width !== width || tex.height !== height) {
|
|
188
|
+
tex?.destroy();
|
|
189
|
+
tex = this.device.createTexture({
|
|
190
|
+
label: `vision_mask_${key}`,
|
|
191
|
+
size: [width, height],
|
|
192
|
+
format: "r8unorm",
|
|
193
|
+
usage: GPUTextureUsage.TEXTURE_BINDING | GPUTextureUsage.COPY_DST
|
|
194
|
+
});
|
|
195
|
+
this.masks.set(key, tex);
|
|
196
|
+
}
|
|
197
|
+
const aligned = Math.ceil(width / 256) * 256;
|
|
198
|
+
let bytes = mask;
|
|
199
|
+
if (aligned !== width) {
|
|
200
|
+
bytes = new Uint8Array(aligned * height);
|
|
201
|
+
for (let y = 0; y < height; y++) bytes.set(mask.subarray(y * width, (y + 1) * width), y * aligned);
|
|
202
|
+
}
|
|
203
|
+
this.device.queue.writeTexture({ texture: tex }, bytes.buffer, {
|
|
204
|
+
offset: bytes.byteOffset,
|
|
205
|
+
bytesPerRow: aligned,
|
|
206
|
+
rowsPerImage: height
|
|
207
|
+
}, {
|
|
208
|
+
width,
|
|
209
|
+
height
|
|
210
|
+
});
|
|
211
|
+
return tex;
|
|
212
|
+
}
|
|
213
|
+
/** Clears `target` and draws `picture` through `mask` into it. */
|
|
214
|
+
draw(encoder, target, picture, mask, mode, bounds) {
|
|
215
|
+
const w = mask.width;
|
|
216
|
+
const h = mask.height;
|
|
217
|
+
const data = new Float32Array(8);
|
|
218
|
+
const view = new Uint32Array(data.buffer);
|
|
219
|
+
view[0] = mode === "matte" ? 0 : mode === "mask" ? 1 : 2;
|
|
220
|
+
if (mode === "crop" && bounds) {
|
|
221
|
+
data[2] = bounds.x0 / w;
|
|
222
|
+
data[3] = bounds.y0 / h;
|
|
223
|
+
data[4] = (bounds.x1 + 1) / w;
|
|
224
|
+
data[5] = (bounds.y1 + 1) / h;
|
|
225
|
+
} else {
|
|
226
|
+
data[2] = 0;
|
|
227
|
+
data[3] = 0;
|
|
228
|
+
data[4] = 1;
|
|
229
|
+
data[5] = 1;
|
|
230
|
+
if (mode === "crop") view[0] = 0;
|
|
231
|
+
}
|
|
232
|
+
if (this.uniformIndex >= this.uniforms.length) this.uniforms.push(this.device.createBuffer({
|
|
233
|
+
size: data.byteLength,
|
|
234
|
+
usage: GPUBufferUsage.UNIFORM | GPUBufferUsage.COPY_DST
|
|
235
|
+
}));
|
|
236
|
+
const uniform = this.uniforms[this.uniformIndex];
|
|
237
|
+
this.uniformIndex = (this.uniformIndex + 1) % 16;
|
|
238
|
+
this.device.queue.writeBuffer(uniform, 0, data);
|
|
239
|
+
const bindGroup = this.device.createBindGroup({
|
|
240
|
+
layout: this.pipeline.getBindGroupLayout(0),
|
|
241
|
+
entries: [
|
|
242
|
+
{
|
|
243
|
+
binding: 0,
|
|
244
|
+
resource: { buffer: uniform }
|
|
245
|
+
},
|
|
246
|
+
{
|
|
247
|
+
binding: 1,
|
|
248
|
+
resource: picture.createView()
|
|
249
|
+
},
|
|
250
|
+
{
|
|
251
|
+
binding: 2,
|
|
252
|
+
resource: mask.createView()
|
|
253
|
+
},
|
|
254
|
+
{
|
|
255
|
+
binding: 3,
|
|
256
|
+
resource: this.sampler
|
|
257
|
+
}
|
|
258
|
+
]
|
|
259
|
+
});
|
|
260
|
+
const pass = encoder.beginRenderPass({
|
|
261
|
+
label: "vision_matte_composite",
|
|
262
|
+
colorAttachments: [{
|
|
263
|
+
view: target,
|
|
264
|
+
loadOp: "clear",
|
|
265
|
+
storeOp: "store",
|
|
266
|
+
clearValue: {
|
|
267
|
+
r: 0,
|
|
268
|
+
g: 0,
|
|
269
|
+
b: 0,
|
|
270
|
+
a: 0
|
|
271
|
+
}
|
|
272
|
+
}]
|
|
273
|
+
});
|
|
274
|
+
pass.setPipeline(this.pipeline);
|
|
275
|
+
pass.setBindGroup(0, bindGroup);
|
|
276
|
+
pass.draw(3);
|
|
277
|
+
pass.end();
|
|
278
|
+
}
|
|
279
|
+
};
|
|
280
|
+
|
|
281
|
+
//#endregion
|
|
282
|
+
//#region src/renderers/webgpu-renderer.ts
|
|
283
|
+
let sharedRunner = null;
|
|
284
|
+
let sharedSkeletonRenderer = null;
|
|
285
|
+
let sharedCompositor = null;
|
|
286
|
+
let sharedCompositorFormat = null;
|
|
287
|
+
/** Analysed frames, shared by every vision node in the process. */
|
|
288
|
+
const frameCache = new VisionFrameCache();
|
|
289
|
+
/** The frame each node's child texture holds (drawn on its last call). */
|
|
290
|
+
const heldFrameByNode = /* @__PURE__ */ new Map();
|
|
291
|
+
/** Per node: a copy of the held frame, and the composited matte. */
|
|
292
|
+
const heldTextures = /* @__PURE__ */ new Map();
|
|
293
|
+
const matteTextures = /* @__PURE__ */ new Map();
|
|
294
|
+
/** One tracker per vision node — track ids must not mix across nodes/sources. */
|
|
295
|
+
const objectTrackers = /* @__PURE__ */ new Map();
|
|
296
|
+
/**
|
|
297
|
+
* Persistent per-node child textures. The shared frame encoder is submitted at
|
|
298
|
+
* the END of the frame, so a node renderer cannot read back the child it just
|
|
299
|
+
* drew. We keep the texture around and read back the PREVIOUS frame's pixels at
|
|
300
|
+
* the start of the next call — a stable one-frame delay instead of the
|
|
301
|
+
* unpredictable multi-frame lag you get from pooled textures.
|
|
302
|
+
*
|
|
303
|
+
* Keyed on the stable Vision `Effect` instance (render ids change every frame),
|
|
304
|
+
* with a size-keyed fallback for raw operation nodes.
|
|
305
|
+
*/
|
|
306
|
+
const childTexturesByEffect = /* @__PURE__ */ new WeakMap();
|
|
307
|
+
const childTexturesByKey = /* @__PURE__ */ new Map();
|
|
308
|
+
/**
|
|
309
|
+
* Last successful subject mask, reused for a few frames when the model misses —
|
|
310
|
+
* a transient miss then holds the silhouette instead of flashing the raw plate.
|
|
311
|
+
* Keyed per node/effect so multiple vision nodes do not collide.
|
|
312
|
+
*/
|
|
313
|
+
const lastSubjectByNode = /* @__PURE__ */ new Map();
|
|
314
|
+
const SUBJECT_HOLD_FRAMES = 3;
|
|
315
|
+
/** A per-node texture shaped like `like`, recreated when the size changes. */
|
|
316
|
+
function nodeTexture(textures, device, key, like, usage, label) {
|
|
317
|
+
let tex = textures.get(key);
|
|
318
|
+
if (!tex || tex.width !== like.width || tex.height !== like.height || tex.format !== like.format) {
|
|
319
|
+
tex?.destroy();
|
|
320
|
+
tex = device.createTexture({
|
|
321
|
+
size: [like.width, like.height],
|
|
322
|
+
format: like.format,
|
|
323
|
+
usage,
|
|
324
|
+
label
|
|
325
|
+
});
|
|
326
|
+
textures.set(key, tex);
|
|
327
|
+
}
|
|
328
|
+
return tex;
|
|
329
|
+
}
|
|
330
|
+
const heldTexture = (device, key, like) => nodeTexture(heldTextures, device, key, like, GPUTextureUsage.COPY_DST | GPUTextureUsage.TEXTURE_BINDING, "vision_held_frame");
|
|
331
|
+
const matteTexture = (device, key, like) => nodeTexture(matteTextures, device, key, like, GPUTextureUsage.RENDER_ATTACHMENT | GPUTextureUsage.TEXTURE_BINDING | GPUTextureUsage.COPY_SRC, "vision_matte");
|
|
332
|
+
/** Everything that decides a frame's results besides the frame itself. */
|
|
333
|
+
function cacheSource(nodeKey, child, op, width, height) {
|
|
334
|
+
const c = child;
|
|
335
|
+
return JSON.stringify([
|
|
336
|
+
nodeKey,
|
|
337
|
+
c?.operation?.id ?? c?.id ?? null,
|
|
338
|
+
width,
|
|
339
|
+
height,
|
|
340
|
+
op.mode ?? "passthrough",
|
|
341
|
+
op.variant,
|
|
342
|
+
op.confidence,
|
|
343
|
+
op.classes,
|
|
344
|
+
op.maskThreshold,
|
|
345
|
+
op.featherRadius,
|
|
346
|
+
op.matteSource,
|
|
347
|
+
op.keyBackground === true,
|
|
348
|
+
op.maxMissedFrames
|
|
349
|
+
]);
|
|
350
|
+
}
|
|
351
|
+
/** The background-key threshold at `frame` (it may be animated). */
|
|
352
|
+
function backgroundKeyThreshold(op, frame, fps) {
|
|
353
|
+
const raw = op.backgroundKeyThreshold;
|
|
354
|
+
if (typeof raw === "number") return raw;
|
|
355
|
+
if (typeof raw?.get === "function") return Number(raw.get({
|
|
356
|
+
frame,
|
|
357
|
+
fps
|
|
358
|
+
}));
|
|
359
|
+
if (typeof raw?._value === "number") return Number(raw._value);
|
|
360
|
+
return 70;
|
|
361
|
+
}
|
|
362
|
+
/**
|
|
363
|
+
* The model's subject, or the last one through brief misses: a transient miss
|
|
364
|
+
* then holds the silhouette instead of flashing the raw plate.
|
|
365
|
+
*/
|
|
366
|
+
function selectSubject(selected, nodeKey, frame) {
|
|
367
|
+
if (selected) {
|
|
368
|
+
lastSubjectByNode.set(nodeKey, {
|
|
369
|
+
mask: selected,
|
|
370
|
+
frame
|
|
371
|
+
});
|
|
372
|
+
return selected;
|
|
373
|
+
}
|
|
374
|
+
const last = lastSubjectByNode.get(nodeKey);
|
|
375
|
+
if (last && frame - last.frame <= SUBJECT_HOLD_FRAMES) return last.mask;
|
|
376
|
+
}
|
|
377
|
+
/**
|
|
378
|
+
* Lazy shared runner — `create()` is a pure constructor (zero I/O); the first frame that
|
|
379
|
+
* requests a task triggers that task's model download at inference time (never at init).
|
|
380
|
+
* Recreated only when an option that changes inference changes.
|
|
381
|
+
*/
|
|
382
|
+
function getSharedRunner(op) {
|
|
383
|
+
const options = {
|
|
384
|
+
variant: op.variant,
|
|
385
|
+
confidence: op.confidence,
|
|
386
|
+
classes: op.classes,
|
|
387
|
+
maskThreshold: op.maskThreshold,
|
|
388
|
+
featherRadius: op.featherRadius,
|
|
389
|
+
modelsDir: op.modelsDir,
|
|
390
|
+
baseUrl: op.baseUrl
|
|
391
|
+
};
|
|
392
|
+
const key = JSON.stringify(options);
|
|
393
|
+
if (sharedRunner?.key !== key) {
|
|
394
|
+
sharedRunner?.runner.close();
|
|
395
|
+
sharedRunner = {
|
|
396
|
+
key,
|
|
397
|
+
runner: VisionRunner.create(options)
|
|
398
|
+
};
|
|
399
|
+
}
|
|
400
|
+
return sharedRunner.runner;
|
|
401
|
+
}
|
|
402
|
+
function getObjectTracker(nodeKey, op) {
|
|
403
|
+
let tracker = objectTrackers.get(nodeKey);
|
|
404
|
+
if (!tracker) {
|
|
405
|
+
tracker = new TemporalObjectTracker({ maxMissedFrames: op.maxMissedFrames ?? 15 });
|
|
406
|
+
objectTrackers.set(nodeKey, tracker);
|
|
407
|
+
}
|
|
408
|
+
return tracker;
|
|
409
|
+
}
|
|
410
|
+
/**
|
|
411
|
+
* Normalizes the runner's pose keypoints (plate pixel space) to [0, 1] landmarks, which is
|
|
412
|
+
* the contract the signal bundle and the skeleton pipeline expect.
|
|
413
|
+
*/
|
|
414
|
+
function normalizePoseKeypoints(result, width, height) {
|
|
415
|
+
if (width <= 0 || height <= 0) return result;
|
|
416
|
+
return { people: result.people.map((person) => ({
|
|
417
|
+
...person,
|
|
418
|
+
keypoints: person.keypoints.map((k) => ({
|
|
419
|
+
x: k.x / width,
|
|
420
|
+
y: k.y / height,
|
|
421
|
+
visibility: k.visibility
|
|
422
|
+
}))
|
|
423
|
+
})) };
|
|
424
|
+
}
|
|
425
|
+
/** Assigns each decoded instance mask to the temporal track whose box center is nearest the mask's centroid. */
|
|
426
|
+
function assignMasksToTracks(masks, tracked) {
|
|
427
|
+
return masks.map((mask) => {
|
|
428
|
+
const { cx, cy } = maskCentroid(mask);
|
|
429
|
+
let best = null;
|
|
430
|
+
for (const obj of tracked) {
|
|
431
|
+
const dx = obj.centerX - cx;
|
|
432
|
+
const dy = obj.centerY - cy;
|
|
433
|
+
const distance = dx * dx + dy * dy;
|
|
434
|
+
if (!best || distance < best.distance) best = {
|
|
435
|
+
trackId: obj.trackId,
|
|
436
|
+
distance
|
|
437
|
+
};
|
|
438
|
+
}
|
|
439
|
+
return best ? {
|
|
440
|
+
...mask,
|
|
441
|
+
trackId: best.trackId
|
|
442
|
+
} : mask;
|
|
443
|
+
});
|
|
444
|
+
}
|
|
445
|
+
/**
|
|
446
|
+
* Grows a subject mask into connected foreground pixels — seeding a flood fill
|
|
447
|
+
* from the subject through pixels that differ from the sampled backdrop recovers
|
|
448
|
+
* thin or fast-moving edges (hair, fabric) without adding a near-uniform studio
|
|
449
|
+
* wall or its soft grey shadow.
|
|
450
|
+
*/
|
|
451
|
+
function fillInternalHoles(mask, width, height) {
|
|
452
|
+
const n = width * height;
|
|
453
|
+
const exterior = new Uint8Array(n);
|
|
454
|
+
const queue = new Int32Array(n);
|
|
455
|
+
let tail = 0;
|
|
456
|
+
const seed = (i) => {
|
|
457
|
+
if (exterior[i] || mask[i] > 0) return;
|
|
458
|
+
exterior[i] = 1;
|
|
459
|
+
queue[tail++] = i;
|
|
460
|
+
};
|
|
461
|
+
for (let x = 0; x < width; x++) {
|
|
462
|
+
seed(x);
|
|
463
|
+
seed((height - 1) * width + x);
|
|
464
|
+
}
|
|
465
|
+
for (let y = 0; y < height; y++) {
|
|
466
|
+
seed(y * width);
|
|
467
|
+
seed(y * width + width - 1);
|
|
468
|
+
}
|
|
469
|
+
let head = 0;
|
|
470
|
+
while (head < tail) {
|
|
471
|
+
const i = queue[head++];
|
|
472
|
+
const x = i % width;
|
|
473
|
+
let j = i + 1;
|
|
474
|
+
if (x + 1 < width && !exterior[j] && mask[j] === 0) {
|
|
475
|
+
exterior[j] = 1;
|
|
476
|
+
queue[tail++] = j;
|
|
477
|
+
}
|
|
478
|
+
j = i - 1;
|
|
479
|
+
if (x > 0 && !exterior[j] && mask[j] === 0) {
|
|
480
|
+
exterior[j] = 1;
|
|
481
|
+
queue[tail++] = j;
|
|
482
|
+
}
|
|
483
|
+
j = i + width;
|
|
484
|
+
if (j < n && !exterior[j] && mask[j] === 0) {
|
|
485
|
+
exterior[j] = 1;
|
|
486
|
+
queue[tail++] = j;
|
|
487
|
+
}
|
|
488
|
+
j = i - width;
|
|
489
|
+
if (j >= 0 && !exterior[j] && mask[j] === 0) {
|
|
490
|
+
exterior[j] = 1;
|
|
491
|
+
queue[tail++] = j;
|
|
492
|
+
}
|
|
493
|
+
}
|
|
494
|
+
for (let i = 0; i < n; i++) if (mask[i] === 0 && exterior[i] === 0) mask[i] = 255;
|
|
495
|
+
}
|
|
496
|
+
function smoothMaskBoundary(mask, width, height, radius = 2) {
|
|
497
|
+
if (radius <= 0) return mask;
|
|
498
|
+
const temp = new Uint8Array(width * height);
|
|
499
|
+
const out = new Uint8Array(width * height);
|
|
500
|
+
const div = 2 * radius + 1;
|
|
501
|
+
for (let y = 0; y < height; y++) {
|
|
502
|
+
const row = y * width;
|
|
503
|
+
let sum = 0;
|
|
504
|
+
for (let k = -radius; k <= radius; k++) {
|
|
505
|
+
const x = Math.max(0, Math.min(width - 1, k));
|
|
506
|
+
sum += mask[row + x];
|
|
507
|
+
}
|
|
508
|
+
for (let x = 0; x < width; x++) {
|
|
509
|
+
temp[row + x] = Math.round(sum / div);
|
|
510
|
+
const xOut = Math.max(0, x - radius);
|
|
511
|
+
const xIn = Math.min(width - 1, x + radius + 1);
|
|
512
|
+
sum += mask[row + xIn] - mask[row + xOut];
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
const sums = new Float64Array(width);
|
|
516
|
+
for (let k = -radius; k <= radius; k++) {
|
|
517
|
+
const row = Math.max(0, Math.min(height - 1, k)) * width;
|
|
518
|
+
for (let x = 0; x < width; x++) sums[x] += temp[row + x];
|
|
519
|
+
}
|
|
520
|
+
for (let y = 0; y < height; y++) {
|
|
521
|
+
const row = y * width;
|
|
522
|
+
const rowOut = Math.max(0, y - radius) * width;
|
|
523
|
+
const rowIn = Math.min(height - 1, y + radius + 1) * width;
|
|
524
|
+
for (let x = 0; x < width; x++) {
|
|
525
|
+
out[row + x] = Math.round(sums[x] / div);
|
|
526
|
+
sums[x] += temp[rowIn + x] - temp[rowOut + x];
|
|
527
|
+
}
|
|
528
|
+
}
|
|
529
|
+
return out;
|
|
530
|
+
}
|
|
531
|
+
function growMaskIntoForeground(mask, width, height, pixels, threshold, featherRadius) {
|
|
532
|
+
let br = 0;
|
|
533
|
+
let bg = 0;
|
|
534
|
+
let bb = 0;
|
|
535
|
+
let count = 0;
|
|
536
|
+
const sample = (x, y) => {
|
|
537
|
+
const i = y * width + x;
|
|
538
|
+
if (mask[i]) return;
|
|
539
|
+
const p = i * 4;
|
|
540
|
+
br += pixels[p];
|
|
541
|
+
bg += pixels[p + 1];
|
|
542
|
+
bb += pixels[p + 2];
|
|
543
|
+
count++;
|
|
544
|
+
};
|
|
545
|
+
for (let x = 0; x < width; x += 4) {
|
|
546
|
+
sample(x, 0);
|
|
547
|
+
sample(x, height - 1);
|
|
548
|
+
}
|
|
549
|
+
for (let y = 0; y < height; y += 4) {
|
|
550
|
+
sample(0, y);
|
|
551
|
+
sample(width - 1, y);
|
|
552
|
+
}
|
|
553
|
+
if (count === 0) return mask;
|
|
554
|
+
br /= count;
|
|
555
|
+
bg /= count;
|
|
556
|
+
bb /= count;
|
|
557
|
+
const n = width * height;
|
|
558
|
+
const fg = new Uint8Array(n);
|
|
559
|
+
for (let i = 0, p = 0; i < n; i++, p += 4) fg[i] = Math.max(Math.abs(pixels[p] - br), Math.abs(pixels[p + 1] - bg), Math.abs(pixels[p + 2] - bb)) > threshold ? 1 : 0;
|
|
560
|
+
const queue = new Int32Array(n);
|
|
561
|
+
let tail = 0;
|
|
562
|
+
for (let i = 0; i < n; i++) if (mask[i] > 0) {
|
|
563
|
+
if (fg[i]) mask[i] = 255;
|
|
564
|
+
queue[tail++] = i;
|
|
565
|
+
}
|
|
566
|
+
let head = 0;
|
|
567
|
+
while (head < tail) {
|
|
568
|
+
const i = queue[head++];
|
|
569
|
+
const x = i % width;
|
|
570
|
+
let j = i + 1;
|
|
571
|
+
if (x + 1 < width && mask[j] !== 255 && fg[j]) {
|
|
572
|
+
mask[j] = 255;
|
|
573
|
+
queue[tail++] = j;
|
|
574
|
+
}
|
|
575
|
+
j = i - 1;
|
|
576
|
+
if (x > 0 && mask[j] !== 255 && fg[j]) {
|
|
577
|
+
mask[j] = 255;
|
|
578
|
+
queue[tail++] = j;
|
|
579
|
+
}
|
|
580
|
+
j = i + width;
|
|
581
|
+
if (j < n && mask[j] !== 255 && fg[j]) {
|
|
582
|
+
mask[j] = 255;
|
|
583
|
+
queue[tail++] = j;
|
|
584
|
+
}
|
|
585
|
+
j = i - width;
|
|
586
|
+
if (j >= 0 && mask[j] !== 255 && fg[j]) {
|
|
587
|
+
mask[j] = 255;
|
|
588
|
+
queue[tail++] = j;
|
|
589
|
+
}
|
|
590
|
+
}
|
|
591
|
+
fillInternalHoles(mask, width, height);
|
|
592
|
+
return smoothMaskBoundary(mask, width, height, featherRadius !== void 0 && featherRadius > 0 ? Math.max(1, Math.min(8, Math.round(featherRadius * 50))) : 2);
|
|
593
|
+
}
|
|
594
|
+
function maskCentroid(mask) {
|
|
595
|
+
const { mask: data, width, height } = mask;
|
|
596
|
+
let sumX = 0;
|
|
597
|
+
let sumY = 0;
|
|
598
|
+
let count = 0;
|
|
599
|
+
for (let y = 0; y < height; y++) {
|
|
600
|
+
const row = y * width;
|
|
601
|
+
for (let x = 0; x < width; x++) if (data[row + x] > 0) {
|
|
602
|
+
sumX += x;
|
|
603
|
+
sumY += y;
|
|
604
|
+
count++;
|
|
605
|
+
}
|
|
606
|
+
}
|
|
607
|
+
if (count === 0) return {
|
|
608
|
+
cx: 0,
|
|
609
|
+
cy: 0
|
|
610
|
+
};
|
|
611
|
+
return {
|
|
612
|
+
cx: sumX / count,
|
|
613
|
+
cy: sumY / count
|
|
614
|
+
};
|
|
615
|
+
}
|
|
616
|
+
const VisionWebGPURenderer = async (args) => {
|
|
617
|
+
const { ctx, encoder, pass, targetView, targetWidth, targetHeight, props, drawChild } = args;
|
|
618
|
+
const { virtualMedia } = props;
|
|
619
|
+
const rawOp = virtualMedia?.operation;
|
|
620
|
+
if (rawOp?.op !== "Vision") return;
|
|
621
|
+
const op = rawOp;
|
|
622
|
+
pass.end();
|
|
623
|
+
const childMedia = virtualMedia?.children?.[0];
|
|
624
|
+
if (!childMedia) return;
|
|
625
|
+
const effectKey = op.effect;
|
|
626
|
+
const hasEffectKey = effectKey !== void 0 && effectKey !== null;
|
|
627
|
+
const nodeKeyStr = virtualMedia?.id ?? rawOp?.id ?? `${targetWidth}x${targetHeight}_${op.mode}_${op.variant}_${op.keyBackground}_${op.backgroundKeyThreshold}`;
|
|
628
|
+
const fallbackKey = `vision-${nodeKeyStr}`;
|
|
629
|
+
let childTex = hasEffectKey ? childTexturesByEffect.get(effectKey) : childTexturesByKey.get(fallbackKey);
|
|
630
|
+
if (childTex && (childTex.width !== targetWidth || childTex.height !== targetHeight)) {
|
|
631
|
+
childTex.destroy();
|
|
632
|
+
childTex = void 0;
|
|
633
|
+
if (hasEffectKey) childTexturesByEffect.delete(effectKey);
|
|
634
|
+
else childTexturesByKey.delete(fallbackKey);
|
|
635
|
+
}
|
|
636
|
+
const hasPreviousFrame = childTex !== void 0;
|
|
637
|
+
if (!childTex) {
|
|
638
|
+
childTex = ctx.device.createTexture({
|
|
639
|
+
size: [targetWidth, targetHeight],
|
|
640
|
+
format: ctx.renderer.format,
|
|
641
|
+
usage: GPUTextureUsage.RENDER_ATTACHMENT | GPUTextureUsage.TEXTURE_BINDING | GPUTextureUsage.COPY_SRC,
|
|
642
|
+
label: "vision_child_persistent"
|
|
643
|
+
});
|
|
644
|
+
if (hasEffectKey) childTexturesByEffect.set(effectKey, childTex);
|
|
645
|
+
else childTexturesByKey.set(fallbackKey, childTex);
|
|
646
|
+
if (childTexturesByKey.size > 4) {
|
|
647
|
+
const oldest = childTexturesByKey.keys().next().value;
|
|
648
|
+
if (oldest && oldest !== fallbackKey) {
|
|
649
|
+
childTexturesByKey.get(oldest)?.destroy();
|
|
650
|
+
childTexturesByKey.delete(oldest);
|
|
651
|
+
}
|
|
652
|
+
}
|
|
653
|
+
}
|
|
654
|
+
const frameIdx = props.frame ?? 0;
|
|
655
|
+
const fps = props.fps ?? 24;
|
|
656
|
+
const mode = op.mode ?? "passthrough";
|
|
657
|
+
const isMatteMode = mode === "mask" || mode === "matte" || mode === "crop";
|
|
658
|
+
const visionBundle = op.visionBundle ?? virtualMedia.visionBundle;
|
|
659
|
+
const wantsDetection = op.enableDetection !== false || mode === "boxes" || mode === "tracking";
|
|
660
|
+
const wantsSegmentation = op.enableSegmentation === true || isMatteMode && op.matteSource !== "selfie";
|
|
661
|
+
const wantsSelfie = op.enableMatte === true || isMatteMode && op.matteSource === "selfie";
|
|
662
|
+
const wantsPose = op.enablePose === true || mode === "skeleton";
|
|
663
|
+
const cacheable = !wantsSelfie && !wantsPose && (wantsDetection || wantsSegmentation);
|
|
664
|
+
const source = cacheable ? cacheSource(nodeKeyStr, childMedia, op, targetWidth, targetHeight) : "";
|
|
665
|
+
const keyThresholdAt = (frame) => op.keyBackground === true ? backgroundKeyThreshold(op, frame, fps) : void 0;
|
|
666
|
+
const lookup = (frame) => {
|
|
667
|
+
if (!cacheable || frame === void 0) return void 0;
|
|
668
|
+
const hit = frameCache.get(source, frame);
|
|
669
|
+
if (!hit || hit.keyThreshold !== keyThresholdAt(frame)) return void 0;
|
|
670
|
+
if (wantsSegmentation && visionBundle && !hit.masks) return void 0;
|
|
671
|
+
return hit;
|
|
672
|
+
};
|
|
673
|
+
const heldFrame = hasPreviousFrame ? heldFrameByNode.get(nodeKeyStr) : void 0;
|
|
674
|
+
let cached = lookup(frameIdx);
|
|
675
|
+
const exact = cached !== void 0;
|
|
676
|
+
if (!cached) cached = lookup(heldFrame);
|
|
677
|
+
let heldTex;
|
|
678
|
+
if (isMatteMode && hasPreviousFrame && !exact) {
|
|
679
|
+
heldTex = heldTexture(ctx.device, nodeKeyStr, childTex);
|
|
680
|
+
encoder.copyTextureToTexture({ texture: childTex }, { texture: heldTex }, [
|
|
681
|
+
targetWidth,
|
|
682
|
+
targetHeight,
|
|
683
|
+
1
|
|
684
|
+
]);
|
|
685
|
+
}
|
|
686
|
+
let framePixels = null;
|
|
687
|
+
if (hasPreviousFrame && !cached) {
|
|
688
|
+
const unalignedBytesPerRow = targetWidth * 4;
|
|
689
|
+
const bytesPerRow = Math.ceil(unalignedBytesPerRow / 256) * 256;
|
|
690
|
+
const bufferSize = bytesPerRow * targetHeight;
|
|
691
|
+
const stagingBuffer = ctx.device.createBuffer({
|
|
692
|
+
size: bufferSize,
|
|
693
|
+
usage: GPUBufferUsage.COPY_DST | GPUBufferUsage.MAP_READ,
|
|
694
|
+
label: "vision_frame_staging"
|
|
695
|
+
});
|
|
696
|
+
const readbackEncoder = ctx.device.createCommandEncoder({ label: "vision_readback_encoder" });
|
|
697
|
+
readbackEncoder.copyTextureToBuffer({ texture: childTex }, {
|
|
698
|
+
buffer: stagingBuffer,
|
|
699
|
+
bytesPerRow,
|
|
700
|
+
rowsPerImage: targetHeight
|
|
701
|
+
}, [
|
|
702
|
+
targetWidth,
|
|
703
|
+
targetHeight,
|
|
704
|
+
1
|
|
705
|
+
]);
|
|
706
|
+
ctx.device.queue.submit([readbackEncoder.finish()]);
|
|
707
|
+
await stagingBuffer.mapAsync(GPUMapMode.READ);
|
|
708
|
+
const mappedBytes = new Uint8Array(stagingBuffer.getMappedRange());
|
|
709
|
+
framePixels = new Uint8ClampedArray(targetWidth * targetHeight * 4);
|
|
710
|
+
if (bytesPerRow === unalignedBytesPerRow) framePixels.set(mappedBytes.subarray(0, framePixels.length));
|
|
711
|
+
else for (let y = 0; y < targetHeight; y++) framePixels.set(mappedBytes.subarray(y * bytesPerRow, y * bytesPerRow + unalignedBytesPerRow), y * unalignedBytesPerRow);
|
|
712
|
+
stagingBuffer.unmap();
|
|
713
|
+
stagingBuffer.destroy();
|
|
714
|
+
}
|
|
715
|
+
const childView = childTex.createView();
|
|
716
|
+
ctx.renderer.beginFrame(encoder, childView, {
|
|
717
|
+
r: 0,
|
|
718
|
+
g: 0,
|
|
719
|
+
b: 0,
|
|
720
|
+
a: 0
|
|
721
|
+
}, targetWidth, targetHeight, "clear").end();
|
|
722
|
+
await drawChild(childMedia, { ...props }, childView, childTex, targetWidth, targetHeight);
|
|
723
|
+
heldFrameByNode.set(nodeKeyStr, frameIdx);
|
|
724
|
+
let frameObjects = [];
|
|
725
|
+
let frameMasks = [];
|
|
726
|
+
let personMatte;
|
|
727
|
+
let currentPoseRes;
|
|
728
|
+
let subject;
|
|
729
|
+
if (cached) {
|
|
730
|
+
frameObjects = cached.objects;
|
|
731
|
+
if (wantsDetection) visionBundle?.setObjectResult(frameIdx, {
|
|
732
|
+
objects: cached.objects,
|
|
733
|
+
rawDetections: cached.detections
|
|
734
|
+
});
|
|
735
|
+
if (wantsSegmentation && visionBundle && cached.masks) {
|
|
736
|
+
frameMasks = cached.masks.map(fromCachedMask);
|
|
737
|
+
visionBundle.setMaskResult(frameIdx, frameMasks);
|
|
738
|
+
}
|
|
739
|
+
if (cached.subject) {
|
|
740
|
+
subject = fromCachedMask(cached.subject.mask);
|
|
741
|
+
lastSubjectByNode.set(nodeKeyStr, {
|
|
742
|
+
mask: subject,
|
|
743
|
+
frame: frameIdx
|
|
744
|
+
});
|
|
745
|
+
}
|
|
746
|
+
} else if (framePixels) {
|
|
747
|
+
const runner = getSharedRunner(op);
|
|
748
|
+
const image = {
|
|
749
|
+
data: framePixels,
|
|
750
|
+
width: targetWidth,
|
|
751
|
+
height: targetHeight
|
|
752
|
+
};
|
|
753
|
+
let detections = [];
|
|
754
|
+
if (wantsDetection) {
|
|
755
|
+
detections = await runner.detect(image);
|
|
756
|
+
const tracker = getObjectTracker(nodeKeyStr, op);
|
|
757
|
+
if (frameIdx === 0) tracker.reset();
|
|
758
|
+
frameObjects = tracker.update(detections, frameIdx, fps);
|
|
759
|
+
visionBundle?.setObjectResult(frameIdx, {
|
|
760
|
+
objects: frameObjects,
|
|
761
|
+
rawDetections: detections
|
|
762
|
+
});
|
|
763
|
+
}
|
|
764
|
+
if (wantsSegmentation) {
|
|
765
|
+
frameMasks = assignMasksToTracks((await runner.segment(image)).masks, frameObjects);
|
|
766
|
+
visionBundle?.setMaskResult(frameIdx, frameMasks);
|
|
767
|
+
}
|
|
768
|
+
if (wantsSelfie) {
|
|
769
|
+
personMatte = await runner.matte(image);
|
|
770
|
+
visionBundle?.setMatteResult(frameIdx, personMatte);
|
|
771
|
+
}
|
|
772
|
+
if (wantsPose) {
|
|
773
|
+
currentPoseRes = { people: matchPoseToTracks(normalizePoseKeypoints(await runner.pose(image), image.width, image.height).people, frameObjects) };
|
|
774
|
+
visionBundle?.setPoseResult(frameIdx, currentPoseRes);
|
|
775
|
+
}
|
|
776
|
+
if (isMatteMode) {
|
|
777
|
+
subject = selectSubject(op.matteSource === "selfie" ? personMatteAsMask(personMatte) : mergeSubjectMask(frameMasks.length > 0 ? frameMasks : visionBundle?.getMaskResult(frameIdx) ?? []), nodeKeyStr, frameIdx);
|
|
778
|
+
const threshold = keyThresholdAt(heldFrame ?? frameIdx);
|
|
779
|
+
if (subject && threshold !== void 0) subject = {
|
|
780
|
+
...subject,
|
|
781
|
+
mask: growMaskIntoForeground(subject.mask, subject.width, subject.height, framePixels, threshold, op.featherRadius)
|
|
782
|
+
};
|
|
783
|
+
}
|
|
784
|
+
if (cacheable && heldFrame !== void 0) frameCache.set(source, heldFrame, {
|
|
785
|
+
detections,
|
|
786
|
+
objects: frameObjects,
|
|
787
|
+
masks: visionBundle ? frameMasks.map(toCachedMask) : null,
|
|
788
|
+
subject: subject ? { mask: toCachedMask(subject) } : null,
|
|
789
|
+
keyThreshold: keyThresholdAt(heldFrame)
|
|
790
|
+
});
|
|
791
|
+
}
|
|
792
|
+
if (isMatteMode) {
|
|
793
|
+
const picture = exact ? childTex : heldTex;
|
|
794
|
+
if (!subject || !picture) {
|
|
795
|
+
ctx.renderer.beginFrame(encoder, targetView, {
|
|
796
|
+
r: 0,
|
|
797
|
+
g: 0,
|
|
798
|
+
b: 0,
|
|
799
|
+
a: 0
|
|
800
|
+
}, targetWidth, targetHeight, "clear").end();
|
|
801
|
+
return;
|
|
802
|
+
}
|
|
803
|
+
if (!sharedCompositor || sharedCompositorFormat !== ctx.renderer.format) {
|
|
804
|
+
sharedCompositor = new MatteCompositor(ctx.device, ctx.renderer.format);
|
|
805
|
+
sharedCompositorFormat = ctx.renderer.format;
|
|
806
|
+
}
|
|
807
|
+
const maskTex = sharedCompositor.uploadMask(nodeKeyStr, subject.mask, subject.width, subject.height);
|
|
808
|
+
const matteTex = matteTexture(ctx.device, nodeKeyStr, childTex);
|
|
809
|
+
sharedCompositor.draw(encoder, matteTex.createView(), picture, maskTex, mode, mode === "crop" ? maskBounds(subject) ?? void 0 : void 0);
|
|
810
|
+
visionBundle?.setStencilTexture(matteTex);
|
|
811
|
+
const outPass$1 = ctx.renderer.beginFrame(encoder, targetView, {
|
|
812
|
+
r: 0,
|
|
813
|
+
g: 0,
|
|
814
|
+
b: 0,
|
|
815
|
+
a: 0
|
|
816
|
+
}, targetWidth, targetHeight, "clear");
|
|
817
|
+
ctx.renderer.drawTexture(outPass$1, matteTex, {
|
|
818
|
+
x: 0,
|
|
819
|
+
y: 0,
|
|
820
|
+
width: targetWidth,
|
|
821
|
+
height: targetHeight
|
|
822
|
+
});
|
|
823
|
+
outPass$1.end();
|
|
824
|
+
return;
|
|
825
|
+
}
|
|
826
|
+
if (mode === "skeleton") {
|
|
827
|
+
if (!sharedSkeletonRenderer) sharedSkeletonRenderer = new PoseSkeletonRenderer(ctx.device);
|
|
828
|
+
const person = (visionBundle?.getPoseResult(frameIdx) ?? currentPoseRes)?.people[0];
|
|
829
|
+
if (person && person.keypoints.length > 0) {
|
|
830
|
+
const skelTex = sharedSkeletonRenderer.renderToTexture(person.keypoints, {
|
|
831
|
+
width: targetWidth,
|
|
832
|
+
height: targetHeight
|
|
833
|
+
});
|
|
834
|
+
const skelPass = ctx.renderer.beginFrame(encoder, targetView, {
|
|
835
|
+
r: 0,
|
|
836
|
+
g: 0,
|
|
837
|
+
b: 0,
|
|
838
|
+
a: 0
|
|
839
|
+
}, targetWidth, targetHeight, "clear");
|
|
840
|
+
ctx.renderer.drawTexture(skelPass, skelTex, {
|
|
841
|
+
x: 0,
|
|
842
|
+
y: 0,
|
|
843
|
+
width: targetWidth,
|
|
844
|
+
height: targetHeight
|
|
845
|
+
});
|
|
846
|
+
skelPass.end();
|
|
847
|
+
return;
|
|
848
|
+
}
|
|
849
|
+
}
|
|
850
|
+
if (mode === "boxes" || mode === "tracking") {
|
|
851
|
+
const outPass$1 = ctx.renderer.beginFrame(encoder, targetView, {
|
|
852
|
+
r: 0,
|
|
853
|
+
g: 0,
|
|
854
|
+
b: 0,
|
|
855
|
+
a: 0
|
|
856
|
+
}, targetWidth, targetHeight, "clear");
|
|
857
|
+
ctx.renderer.drawTexture(outPass$1, childTex, {
|
|
858
|
+
x: 0,
|
|
859
|
+
y: 0,
|
|
860
|
+
width: targetWidth,
|
|
861
|
+
height: targetHeight
|
|
862
|
+
});
|
|
863
|
+
const objects = frameObjects.length > 0 ? frameObjects : visionBundle?.getObjectResult(frameIdx).objects ?? [];
|
|
864
|
+
if (objects.length > 0) {
|
|
865
|
+
const boxColor = "#38bdf8";
|
|
866
|
+
const stroke = 2;
|
|
867
|
+
for (const obj of objects) {
|
|
868
|
+
if (!obj.active) continue;
|
|
869
|
+
const b = obj.boundingBox;
|
|
870
|
+
ctx.renderer.drawRect(outPass$1, {
|
|
871
|
+
x: b.originX,
|
|
872
|
+
y: b.originY,
|
|
873
|
+
width: b.width,
|
|
874
|
+
height: stroke
|
|
875
|
+
}, boxColor);
|
|
876
|
+
ctx.renderer.drawRect(outPass$1, {
|
|
877
|
+
x: b.originX,
|
|
878
|
+
y: b.originY + b.height - stroke,
|
|
879
|
+
width: b.width,
|
|
880
|
+
height: stroke
|
|
881
|
+
}, boxColor);
|
|
882
|
+
ctx.renderer.drawRect(outPass$1, {
|
|
883
|
+
x: b.originX,
|
|
884
|
+
y: b.originY,
|
|
885
|
+
width: stroke,
|
|
886
|
+
height: b.height
|
|
887
|
+
}, boxColor);
|
|
888
|
+
ctx.renderer.drawRect(outPass$1, {
|
|
889
|
+
x: b.originX + b.width - stroke,
|
|
890
|
+
y: b.originY,
|
|
891
|
+
width: stroke,
|
|
892
|
+
height: b.height
|
|
893
|
+
}, boxColor);
|
|
894
|
+
const cornerLen = Math.min(16, b.width / 4, b.height / 4);
|
|
895
|
+
ctx.renderer.drawRect(outPass$1, {
|
|
896
|
+
x: b.originX,
|
|
897
|
+
y: b.originY,
|
|
898
|
+
width: cornerLen,
|
|
899
|
+
height: stroke * 2
|
|
900
|
+
}, "#ffffff");
|
|
901
|
+
ctx.renderer.drawRect(outPass$1, {
|
|
902
|
+
x: b.originX,
|
|
903
|
+
y: b.originY,
|
|
904
|
+
width: stroke * 2,
|
|
905
|
+
height: cornerLen
|
|
906
|
+
}, "#ffffff");
|
|
907
|
+
if (mode === "tracking") ctx.renderer.drawRect(outPass$1, {
|
|
908
|
+
x: obj.centerX - 3,
|
|
909
|
+
y: obj.centerY - 3,
|
|
910
|
+
width: 6,
|
|
911
|
+
height: 6
|
|
912
|
+
}, "#ef4444", 3);
|
|
913
|
+
}
|
|
914
|
+
}
|
|
915
|
+
outPass$1.end();
|
|
916
|
+
return;
|
|
917
|
+
}
|
|
918
|
+
const outPass = ctx.renderer.beginFrame(encoder, targetView, {
|
|
919
|
+
r: 0,
|
|
920
|
+
g: 0,
|
|
921
|
+
b: 0,
|
|
922
|
+
a: 0
|
|
923
|
+
}, targetWidth, targetHeight, "clear");
|
|
924
|
+
ctx.renderer.drawTexture(outPass, childTex, {
|
|
925
|
+
x: 0,
|
|
926
|
+
y: 0,
|
|
927
|
+
width: targetWidth,
|
|
928
|
+
height: targetHeight
|
|
929
|
+
});
|
|
930
|
+
outPass.end();
|
|
931
|
+
};
|
|
932
|
+
/** Adapts a Selfie Segmenter matte to the instance-mask shape the composite path takes. */
|
|
933
|
+
function personMatteAsMask(matte) {
|
|
934
|
+
if (!matte || matte.coverage <= 0) return void 0;
|
|
935
|
+
const area = Math.round(matte.coverage * matte.width * matte.height);
|
|
936
|
+
return {
|
|
937
|
+
category: "person",
|
|
938
|
+
mask: matte.mask,
|
|
939
|
+
width: matte.width,
|
|
940
|
+
height: matte.height,
|
|
941
|
+
area,
|
|
942
|
+
coverage: matte.coverage,
|
|
943
|
+
detectionIndex: -1
|
|
944
|
+
};
|
|
945
|
+
}
|
|
946
|
+
|
|
947
|
+
//#endregion
|
|
948
|
+
//#region src/renderers/index.ts
|
|
949
|
+
var renderers_default = defineRenderer({ WebGPURenderer: VisionWebGPURenderer });
|
|
950
|
+
|
|
951
|
+
//#endregion
|
|
952
|
+
export { renderers_default as t };
|
|
953
|
+
//# sourceMappingURL=renderers-B24hckhO.mjs.map
|