@framefields/vision 0.0.0-stage → 2.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENCE +202 -0
- package/dist/chunk-CFhWmker.mjs +36 -0
- package/dist/index-CoAN0xxG.d.mts +1198 -0
- package/dist/index-CoAN0xxG.d.mts.map +1 -0
- package/dist/index.d.mts +3 -0
- package/dist/index.mjs +6 -0
- package/dist/registry-CSFimt7C.mjs +196 -0
- package/dist/registry-CSFimt7C.mjs.map +1 -0
- package/dist/schemas.d.mts +198 -0
- package/dist/schemas.d.mts.map +1 -0
- package/dist/schemas.mjs +118 -0
- package/dist/schemas.mjs.map +1 -0
- package/dist/src-C2sA_wDT.mjs +2973 -0
- package/dist/src-C2sA_wDT.mjs.map +1 -0
- package/dist/web.d.mts +11 -0
- package/dist/web.d.mts.map +1 -0
- package/dist/web.mjs +64 -0
- package/dist/web.mjs.map +1 -0
- package/package.json +64 -3
- package/README.md +0 -3
|
@@ -0,0 +1,2973 @@
|
|
|
1
|
+
import { n as __reExport, t as __exportAll } from "./chunk-CFhWmker.mjs";
|
|
2
|
+
import { a as modelKeyFor, i as VISION_VARIANTS, n as VISION_MODELS, r as VISION_TASKS, t as COCO_CLASSES } from "./registry-CSFimt7C.mjs";
|
|
3
|
+
import { PoseSkeletonComputePipeline } from "@framefields/tensor-webgpu";
|
|
4
|
+
import { createHash } from "node:crypto";
|
|
5
|
+
import { existsSync, mkdirSync, statSync } from "node:fs";
|
|
6
|
+
import { readFile, rename, unlink, writeFile } from "node:fs/promises";
|
|
7
|
+
import { homedir } from "node:os";
|
|
8
|
+
import { dirname, join, resolve } from "node:path";
|
|
9
|
+
import { frameSignal, programmaticSignal } from "@framefields/core";
|
|
10
|
+
|
|
11
|
+
//#region src/tracking/temporal-object-tracker.ts
|
|
12
|
+
function computeIoU(boxA, boxB) {
|
|
13
|
+
const xA = Math.max(boxA.originX, boxB.originX);
|
|
14
|
+
const yA = Math.max(boxA.originY, boxB.originY);
|
|
15
|
+
const xB = Math.min(boxA.originX + boxA.width, boxB.originX + boxB.width);
|
|
16
|
+
const yB = Math.min(boxA.originY + boxA.height, boxB.originY + boxB.height);
|
|
17
|
+
const interArea = Math.max(0, xB - xA) * Math.max(0, yB - yA);
|
|
18
|
+
if (interArea <= 0) return 0;
|
|
19
|
+
const unionArea = boxA.width * boxA.height + boxB.width * boxB.height - interArea;
|
|
20
|
+
return unionArea > 0 ? interArea / unionArea : 0;
|
|
21
|
+
}
|
|
22
|
+
var TemporalObjectTracker = class {
|
|
23
|
+
_nextTrackId = 1;
|
|
24
|
+
_tracks = [];
|
|
25
|
+
_iouThreshold;
|
|
26
|
+
_maxMissedFrames;
|
|
27
|
+
_minHits;
|
|
28
|
+
_positionSmoothing;
|
|
29
|
+
_activeDuringCoast;
|
|
30
|
+
constructor(options = {}) {
|
|
31
|
+
this._iouThreshold = options.iouThreshold ?? .25;
|
|
32
|
+
this._maxMissedFrames = options.maxMissedFrames ?? 15;
|
|
33
|
+
this._minHits = options.minHits ?? 1;
|
|
34
|
+
this._positionSmoothing = Math.max(.1, Math.min(1, options.positionSmoothing ?? .75));
|
|
35
|
+
this._activeDuringCoast = options.activeDuringCoast !== false;
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* Updates the multi-object tracker with new detections for the current frame.
|
|
39
|
+
*
|
|
40
|
+
* @param detections Raw detections found on the current frame
|
|
41
|
+
* @param frame Current frame index
|
|
42
|
+
* @param fps Video framerate (used for physical velocity estimation)
|
|
43
|
+
* @returns Stable, temporal list of TrackedObjects
|
|
44
|
+
*/
|
|
45
|
+
update(detections, _frame, fps = 24) {
|
|
46
|
+
const dt = fps > 0 ? 1 / fps : .0416;
|
|
47
|
+
for (const track of this._tracks) {
|
|
48
|
+
track.age++;
|
|
49
|
+
track.timeSinceUpdate++;
|
|
50
|
+
if (track.timeSinceUpdate > 0) {
|
|
51
|
+
const predCenterX = track.centerX + track.vx * dt;
|
|
52
|
+
const predCenterY = track.centerY + track.vy * dt;
|
|
53
|
+
track.centerX = predCenterX;
|
|
54
|
+
track.centerY = predCenterY;
|
|
55
|
+
const w = track.box.width;
|
|
56
|
+
const h = track.box.height;
|
|
57
|
+
const normW = track.box.normalizedWidth;
|
|
58
|
+
const normH = track.box.normalizedHeight;
|
|
59
|
+
const normCenterX = track.box.normalizedX + normW / 2 + track.vx * dt / Math.max(1, track.box.width / normW);
|
|
60
|
+
const normCenterY = track.box.normalizedY + normH / 2 + track.vy * dt / Math.max(1, track.box.height / normH);
|
|
61
|
+
track.box = {
|
|
62
|
+
originX: predCenterX - w / 2,
|
|
63
|
+
originY: predCenterY - h / 2,
|
|
64
|
+
width: w,
|
|
65
|
+
height: h,
|
|
66
|
+
normalizedX: normCenterX - normW / 2,
|
|
67
|
+
normalizedY: normCenterY - normH / 2,
|
|
68
|
+
normalizedWidth: normW,
|
|
69
|
+
normalizedHeight: normH
|
|
70
|
+
};
|
|
71
|
+
track.vx *= .92;
|
|
72
|
+
track.vy *= .92;
|
|
73
|
+
track.isCoasting = true;
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
const matchedDetections = /* @__PURE__ */ new Set();
|
|
77
|
+
const matchedTracks = /* @__PURE__ */ new Set();
|
|
78
|
+
const candidateMatches = [];
|
|
79
|
+
for (let t = 0; t < this._tracks.length; t++) {
|
|
80
|
+
const trk = this._tracks[t];
|
|
81
|
+
for (let d = 0; d < detections.length; d++) {
|
|
82
|
+
const det = detections[d];
|
|
83
|
+
if (trk.category !== det.category) continue;
|
|
84
|
+
const iou = computeIoU(trk.box, det.boundingBox);
|
|
85
|
+
if (iou >= this._iouThreshold) candidateMatches.push({
|
|
86
|
+
trackIdx: t,
|
|
87
|
+
detIdx: d,
|
|
88
|
+
iou
|
|
89
|
+
});
|
|
90
|
+
}
|
|
91
|
+
}
|
|
92
|
+
candidateMatches.sort((a, b) => b.iou - a.iou);
|
|
93
|
+
for (const match of candidateMatches) {
|
|
94
|
+
if (matchedTracks.has(match.trackIdx) || matchedDetections.has(match.detIdx)) continue;
|
|
95
|
+
matchedTracks.add(match.trackIdx);
|
|
96
|
+
matchedDetections.add(match.detIdx);
|
|
97
|
+
const trk = this._tracks[match.trackIdx];
|
|
98
|
+
const det = detections[match.detIdx];
|
|
99
|
+
const detBox = det.boundingBox;
|
|
100
|
+
const detCenterX = detBox.originX + detBox.width / 2;
|
|
101
|
+
const detCenterY = detBox.originY + detBox.height / 2;
|
|
102
|
+
const instVx = (detCenterX - trk.centerX) / dt;
|
|
103
|
+
const instVy = (detCenterY - trk.centerY) / dt;
|
|
104
|
+
trk.vx = trk.vx * .4 + instVx * .6;
|
|
105
|
+
trk.vy = trk.vy * .4 + instVy * .6;
|
|
106
|
+
const alpha = this._positionSmoothing;
|
|
107
|
+
const smoothW = trk.box.width * (1 - alpha) + detBox.width * alpha;
|
|
108
|
+
const smoothH = trk.box.height * (1 - alpha) + detBox.height * alpha;
|
|
109
|
+
const smoothCenterX = trk.centerX * (1 - alpha) + detCenterX * alpha;
|
|
110
|
+
const smoothCenterY = trk.centerY * (1 - alpha) + detCenterY * alpha;
|
|
111
|
+
trk.centerX = smoothCenterX;
|
|
112
|
+
trk.centerY = smoothCenterY;
|
|
113
|
+
trk.score = det.score;
|
|
114
|
+
trk.timeSinceUpdate = 0;
|
|
115
|
+
trk.hits++;
|
|
116
|
+
trk.isCoasting = false;
|
|
117
|
+
if (trk.hits >= this._minHits) trk.active = true;
|
|
118
|
+
trk.box = {
|
|
119
|
+
originX: smoothCenterX - smoothW / 2,
|
|
120
|
+
originY: smoothCenterY - smoothH / 2,
|
|
121
|
+
width: smoothW,
|
|
122
|
+
height: smoothH,
|
|
123
|
+
normalizedX: detBox.normalizedX,
|
|
124
|
+
normalizedY: detBox.normalizedY,
|
|
125
|
+
normalizedWidth: detBox.normalizedWidth,
|
|
126
|
+
normalizedHeight: detBox.normalizedHeight
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
for (let d = 0; d < detections.length; d++) {
|
|
130
|
+
if (matchedDetections.has(d)) continue;
|
|
131
|
+
const det = detections[d];
|
|
132
|
+
const detBox = det.boundingBox;
|
|
133
|
+
const centerX = detBox.originX + detBox.width / 2;
|
|
134
|
+
const centerY = detBox.originY + detBox.height / 2;
|
|
135
|
+
this._tracks.push({
|
|
136
|
+
trackId: this._nextTrackId++,
|
|
137
|
+
category: det.category,
|
|
138
|
+
score: det.score,
|
|
139
|
+
box: detBox,
|
|
140
|
+
centerX,
|
|
141
|
+
centerY,
|
|
142
|
+
vx: 0,
|
|
143
|
+
vy: 0,
|
|
144
|
+
age: 1,
|
|
145
|
+
hits: 1,
|
|
146
|
+
timeSinceUpdate: 0,
|
|
147
|
+
active: this._minHits <= 1,
|
|
148
|
+
isCoasting: false
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
this._tracks = this._tracks.filter((t) => t.timeSinceUpdate <= this._maxMissedFrames);
|
|
152
|
+
return this._tracks.filter((t) => t.active).map((t) => {
|
|
153
|
+
const normW = t.box.normalizedWidth;
|
|
154
|
+
const normH = t.box.normalizedHeight;
|
|
155
|
+
const normCenterX = t.box.normalizedX + normW / 2;
|
|
156
|
+
const normCenterY = t.box.normalizedY + normH / 2;
|
|
157
|
+
const speed = Math.sqrt(t.vx * t.vx + t.vy * t.vy);
|
|
158
|
+
return {
|
|
159
|
+
trackId: t.trackId,
|
|
160
|
+
category: t.category,
|
|
161
|
+
score: t.score,
|
|
162
|
+
boundingBox: t.box,
|
|
163
|
+
centerX: t.centerX,
|
|
164
|
+
centerY: t.centerY,
|
|
165
|
+
normalizedCenterX: normCenterX,
|
|
166
|
+
normalizedCenterY: normCenterY,
|
|
167
|
+
velocity: {
|
|
168
|
+
vx: t.vx,
|
|
169
|
+
vy: t.vy
|
|
170
|
+
},
|
|
171
|
+
speed,
|
|
172
|
+
age: t.age,
|
|
173
|
+
hits: t.hits,
|
|
174
|
+
active: this._activeDuringCoast ? true : t.timeSinceUpdate === 0,
|
|
175
|
+
isCoasting: t.isCoasting
|
|
176
|
+
};
|
|
177
|
+
});
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Tracks objects across a full sequence of frames with gap interpolation,
|
|
181
|
+
* boundary extrapolation (ensuring frame 0 through end are tracked),
|
|
182
|
+
* and temporal Gaussian smoothing.
|
|
183
|
+
*
|
|
184
|
+
* Guarantees zero flicker, zero drift, and frame-accurate stability across the entire video.
|
|
185
|
+
*/
|
|
186
|
+
trackSequence(perFrameDetections, totalFrames, fps = 24) {
|
|
187
|
+
this.reset();
|
|
188
|
+
const dt = fps > 0 ? 1 / fps : .0416;
|
|
189
|
+
const trackHistory = /* @__PURE__ */ new Map();
|
|
190
|
+
for (let f = 0; f < totalFrames; f++) {
|
|
191
|
+
const dets = perFrameDetections[f] ?? [];
|
|
192
|
+
const stepTracks = this.update(dets, f, fps);
|
|
193
|
+
for (const trk of stepTracks) {
|
|
194
|
+
if (!trackHistory.has(trk.trackId)) trackHistory.set(trk.trackId, {
|
|
195
|
+
category: trk.category,
|
|
196
|
+
observations: []
|
|
197
|
+
});
|
|
198
|
+
if (!trk.isCoasting) trackHistory.get(trk.trackId).observations.push({
|
|
199
|
+
frame: f,
|
|
200
|
+
box: trk.boundingBox,
|
|
201
|
+
score: trk.score
|
|
202
|
+
});
|
|
203
|
+
}
|
|
204
|
+
}
|
|
205
|
+
const fullTrajectories = /* @__PURE__ */ new Map();
|
|
206
|
+
for (const [trackId, info] of trackHistory.entries()) {
|
|
207
|
+
const obs = info.observations;
|
|
208
|
+
if (obs.length === 0) continue;
|
|
209
|
+
obs.sort((a, b) => a.frame - b.frame);
|
|
210
|
+
const frameData = new Array(totalFrames).fill(null);
|
|
211
|
+
for (const o of obs) frameData[o.frame] = {
|
|
212
|
+
box: o.box,
|
|
213
|
+
score: o.score
|
|
214
|
+
};
|
|
215
|
+
const firstObs = obs[0];
|
|
216
|
+
for (let f = 0; f < firstObs.frame; f++) frameData[f] = {
|
|
217
|
+
box: { ...firstObs.box },
|
|
218
|
+
score: firstObs.score * .9
|
|
219
|
+
};
|
|
220
|
+
for (let i = 0; i < obs.length - 1; i++) {
|
|
221
|
+
const startObs = obs[i];
|
|
222
|
+
const endObs = obs[i + 1];
|
|
223
|
+
const gap = endObs.frame - startObs.frame;
|
|
224
|
+
if (gap <= 1) continue;
|
|
225
|
+
for (let f = startObs.frame + 1; f < endObs.frame; f++) {
|
|
226
|
+
const t = (f - startObs.frame) / gap;
|
|
227
|
+
frameData[f] = {
|
|
228
|
+
box: {
|
|
229
|
+
originX: startObs.box.originX + (endObs.box.originX - startObs.box.originX) * t,
|
|
230
|
+
originY: startObs.box.originY + (endObs.box.originY - startObs.box.originY) * t,
|
|
231
|
+
width: startObs.box.width + (endObs.box.width - startObs.box.width) * t,
|
|
232
|
+
height: startObs.box.height + (endObs.box.height - startObs.box.height) * t,
|
|
233
|
+
normalizedX: startObs.box.normalizedX + (endObs.box.normalizedX - startObs.box.normalizedX) * t,
|
|
234
|
+
normalizedY: startObs.box.normalizedY + (endObs.box.normalizedY - startObs.box.normalizedY) * t,
|
|
235
|
+
normalizedWidth: startObs.box.normalizedWidth + (endObs.box.normalizedWidth - startObs.box.normalizedWidth) * t,
|
|
236
|
+
normalizedHeight: startObs.box.normalizedHeight + (endObs.box.normalizedHeight - startObs.box.normalizedHeight) * t
|
|
237
|
+
},
|
|
238
|
+
score: startObs.score * (1 - t) + endObs.score * t
|
|
239
|
+
};
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
const lastObs = obs[obs.length - 1];
|
|
243
|
+
for (let f = lastObs.frame + 1; f < totalFrames; f++) frameData[f] = {
|
|
244
|
+
box: { ...lastObs.box },
|
|
245
|
+
score: lastObs.score * .9
|
|
246
|
+
};
|
|
247
|
+
const smoothedFrames = [];
|
|
248
|
+
const radius = 2;
|
|
249
|
+
const weights = [
|
|
250
|
+
.06136,
|
|
251
|
+
.24477,
|
|
252
|
+
.38774,
|
|
253
|
+
.24477,
|
|
254
|
+
.06136
|
|
255
|
+
];
|
|
256
|
+
for (let f = 0; f < totalFrames; f++) {
|
|
257
|
+
let sumWeight = 0;
|
|
258
|
+
let sumX = 0;
|
|
259
|
+
let sumY = 0;
|
|
260
|
+
let sumW = 0;
|
|
261
|
+
let sumH = 0;
|
|
262
|
+
let sumNormX = 0;
|
|
263
|
+
let sumNormY = 0;
|
|
264
|
+
let sumNormW = 0;
|
|
265
|
+
let sumNormH = 0;
|
|
266
|
+
let sumScore = 0;
|
|
267
|
+
for (let r = -radius; r <= radius; r++) {
|
|
268
|
+
const pt = frameData[Math.max(0, Math.min(totalFrames - 1, f + r))];
|
|
269
|
+
if (pt) {
|
|
270
|
+
const w = weights[r + radius];
|
|
271
|
+
sumWeight += w;
|
|
272
|
+
sumX += pt.box.originX * w;
|
|
273
|
+
sumY += pt.box.originY * w;
|
|
274
|
+
sumW += pt.box.width * w;
|
|
275
|
+
sumH += pt.box.height * w;
|
|
276
|
+
sumNormX += pt.box.normalizedX * w;
|
|
277
|
+
sumNormY += pt.box.normalizedY * w;
|
|
278
|
+
sumNormW += pt.box.normalizedWidth * w;
|
|
279
|
+
sumNormH += pt.box.normalizedHeight * w;
|
|
280
|
+
sumScore += pt.score * w;
|
|
281
|
+
}
|
|
282
|
+
}
|
|
283
|
+
const invW = sumWeight > 0 ? 1 / sumWeight : 1;
|
|
284
|
+
smoothedFrames.push({
|
|
285
|
+
box: {
|
|
286
|
+
originX: sumX * invW,
|
|
287
|
+
originY: sumY * invW,
|
|
288
|
+
width: sumW * invW,
|
|
289
|
+
height: sumH * invW,
|
|
290
|
+
normalizedX: sumNormX * invW,
|
|
291
|
+
normalizedY: sumNormY * invW,
|
|
292
|
+
normalizedWidth: sumNormW * invW,
|
|
293
|
+
normalizedHeight: sumNormH * invW
|
|
294
|
+
},
|
|
295
|
+
score: sumScore * invW
|
|
296
|
+
});
|
|
297
|
+
}
|
|
298
|
+
fullTrajectories.set(trackId, {
|
|
299
|
+
category: info.category,
|
|
300
|
+
frames: smoothedFrames
|
|
301
|
+
});
|
|
302
|
+
}
|
|
303
|
+
const result = [];
|
|
304
|
+
for (let f = 0; f < totalFrames; f++) {
|
|
305
|
+
const frameObjects = [];
|
|
306
|
+
for (const [trackId, traj] of fullTrajectories.entries()) {
|
|
307
|
+
const current = traj.frames[f];
|
|
308
|
+
if (!current) continue;
|
|
309
|
+
const prev = traj.frames[Math.max(0, f - 1)] ?? current;
|
|
310
|
+
const next = traj.frames[Math.min(totalFrames - 1, f + 1)] ?? current;
|
|
311
|
+
const curCenterX = current.box.originX + current.box.width / 2;
|
|
312
|
+
const curCenterY = current.box.originY + current.box.height / 2;
|
|
313
|
+
const prevCenterX = prev.box.originX + prev.box.width / 2;
|
|
314
|
+
const prevCenterY = prev.box.originY + prev.box.height / 2;
|
|
315
|
+
const nextCenterX = next.box.originX + next.box.width / 2;
|
|
316
|
+
const nextCenterY = next.box.originY + next.box.height / 2;
|
|
317
|
+
const timeDelta = f === 0 || f === totalFrames - 1 ? dt : 2 * dt;
|
|
318
|
+
const vx = (nextCenterX - prevCenterX) / timeDelta;
|
|
319
|
+
const vy = (nextCenterY - prevCenterY) / timeDelta;
|
|
320
|
+
const speed = Math.sqrt(vx * vx + vy * vy);
|
|
321
|
+
const normW = current.box.normalizedWidth;
|
|
322
|
+
const normH = current.box.normalizedHeight;
|
|
323
|
+
frameObjects.push({
|
|
324
|
+
trackId,
|
|
325
|
+
category: traj.category,
|
|
326
|
+
score: current.score,
|
|
327
|
+
boundingBox: current.box,
|
|
328
|
+
centerX: curCenterX,
|
|
329
|
+
centerY: curCenterY,
|
|
330
|
+
normalizedCenterX: current.box.normalizedX + normW / 2,
|
|
331
|
+
normalizedCenterY: current.box.normalizedY + normH / 2,
|
|
332
|
+
velocity: {
|
|
333
|
+
vx,
|
|
334
|
+
vy
|
|
335
|
+
},
|
|
336
|
+
speed,
|
|
337
|
+
age: f + 1,
|
|
338
|
+
hits: 10,
|
|
339
|
+
active: true,
|
|
340
|
+
isCoasting: false
|
|
341
|
+
});
|
|
342
|
+
}
|
|
343
|
+
result.push(frameObjects);
|
|
344
|
+
}
|
|
345
|
+
return result;
|
|
346
|
+
}
|
|
347
|
+
/**
|
|
348
|
+
* Resets all internal track state.
|
|
349
|
+
*/
|
|
350
|
+
reset() {
|
|
351
|
+
this._tracks = [];
|
|
352
|
+
this._nextTrackId = 1;
|
|
353
|
+
}
|
|
354
|
+
};
|
|
355
|
+
|
|
356
|
+
//#endregion
|
|
357
|
+
//#region src/analysis/analyze-sequence.ts
|
|
358
|
+
const now = () => typeof performance !== "undefined" && typeof performance.now === "function" ? performance.now() : Date.now();
|
|
359
|
+
async function analyzeSequence(frames, options) {
|
|
360
|
+
const { runner } = options;
|
|
361
|
+
const fps = options.fps ?? 24;
|
|
362
|
+
const tasks = options.tasks && options.tasks.length > 0 ? options.tasks : ["detect"];
|
|
363
|
+
const taskSet = new Set(tasks);
|
|
364
|
+
const categoryFilter = options.categories ? new Set(options.categories.map((c) => c.toLowerCase())) : null;
|
|
365
|
+
const includeMasks = options.includeMasks === true && taskSet.has("segment");
|
|
366
|
+
const tracksObjects = taskSet.has("detect") || taskSet.has("segment");
|
|
367
|
+
const perFrameDetections = [];
|
|
368
|
+
const modelDownloads = {};
|
|
369
|
+
const maskCoverage = /* @__PURE__ */ new Map();
|
|
370
|
+
let inferenceMs = 0;
|
|
371
|
+
let frameIndex = 0;
|
|
372
|
+
const accepts = (category) => !categoryFilter || categoryFilter.has(category.toLowerCase());
|
|
373
|
+
const runTask = async (task, frame, fn) => {
|
|
374
|
+
const key = runner.modelKeyFor(task, frame.width / frame.height);
|
|
375
|
+
const wasReady = runner.downloadStatus.get(key) === "ready";
|
|
376
|
+
const start = now();
|
|
377
|
+
const result = await fn();
|
|
378
|
+
const elapsed = now() - start;
|
|
379
|
+
inferenceMs += elapsed;
|
|
380
|
+
if (!wasReady && !modelDownloads[key]) modelDownloads[key] = {
|
|
381
|
+
bytes: VISION_MODELS[key].bytes,
|
|
382
|
+
ms: elapsed
|
|
383
|
+
};
|
|
384
|
+
return result;
|
|
385
|
+
};
|
|
386
|
+
const recordMasks = (masks$1) => {
|
|
387
|
+
for (const m of masks$1) {
|
|
388
|
+
if (!accepts(m.category)) continue;
|
|
389
|
+
const entry = maskCoverage.get(m.category) ?? {
|
|
390
|
+
sum: 0,
|
|
391
|
+
count: 0
|
|
392
|
+
};
|
|
393
|
+
entry.sum += m.coverage;
|
|
394
|
+
entry.count++;
|
|
395
|
+
maskCoverage.set(m.category, entry);
|
|
396
|
+
}
|
|
397
|
+
};
|
|
398
|
+
for await (const frame of frames) {
|
|
399
|
+
let detections = [];
|
|
400
|
+
if (includeMasks) {
|
|
401
|
+
const seg = await runTask("segment", frame, () => runner.segment(frame));
|
|
402
|
+
recordMasks(seg.masks);
|
|
403
|
+
detections = seg.detections;
|
|
404
|
+
} else if (tracksObjects) detections = await runTask("detect", frame, () => runner.detect(frame));
|
|
405
|
+
if (taskSet.has("pose")) await runTask("pose", frame, () => runner.pose(frame));
|
|
406
|
+
if (taskSet.has("matte")) await runTask("matte", frame, () => runner.matte(frame));
|
|
407
|
+
perFrameDetections.push(categoryFilter ? detections.filter((d) => accepts(d.category)) : detections);
|
|
408
|
+
frameIndex++;
|
|
409
|
+
}
|
|
410
|
+
const totalFrames = frameIndex;
|
|
411
|
+
const trackedPerFrame = new TemporalObjectTracker({
|
|
412
|
+
iouThreshold: options.iouThreshold ?? .25,
|
|
413
|
+
maxMissedFrames: options.maxMissedFrames ?? 15
|
|
414
|
+
}).trackSequence(perFrameDetections, totalFrames, fps);
|
|
415
|
+
const pathStride = options.pathStride ?? Math.max(1, Math.ceil(totalFrames / 50));
|
|
416
|
+
const tracks = /* @__PURE__ */ new Map();
|
|
417
|
+
const classPresentFrames = /* @__PURE__ */ new Map();
|
|
418
|
+
const classMaxConfidence = /* @__PURE__ */ new Map();
|
|
419
|
+
const classTrackIds = /* @__PURE__ */ new Map();
|
|
420
|
+
const getOrCreate = (map, key, make) => {
|
|
421
|
+
let value = map.get(key);
|
|
422
|
+
if (value === void 0) {
|
|
423
|
+
value = make();
|
|
424
|
+
map.set(key, value);
|
|
425
|
+
}
|
|
426
|
+
return value;
|
|
427
|
+
};
|
|
428
|
+
for (let f = 0; f < totalFrames; f++) for (const obj of trackedPerFrame[f] ?? []) {
|
|
429
|
+
if (!accepts(obj.category)) continue;
|
|
430
|
+
let acc = tracks.get(obj.trackId);
|
|
431
|
+
if (!acc) {
|
|
432
|
+
acc = {
|
|
433
|
+
category: obj.category,
|
|
434
|
+
firstFrame: f,
|
|
435
|
+
lastFrame: f,
|
|
436
|
+
speedSum: 0,
|
|
437
|
+
speedCount: 0,
|
|
438
|
+
centers: []
|
|
439
|
+
};
|
|
440
|
+
tracks.set(obj.trackId, acc);
|
|
441
|
+
}
|
|
442
|
+
acc.lastFrame = f;
|
|
443
|
+
if (Number.isFinite(obj.speed)) {
|
|
444
|
+
acc.speedSum += obj.speed;
|
|
445
|
+
acc.speedCount++;
|
|
446
|
+
}
|
|
447
|
+
if ((f - acc.firstFrame) % pathStride === 0) acc.centers.push({
|
|
448
|
+
frame: f,
|
|
449
|
+
x: obj.centerX,
|
|
450
|
+
y: obj.centerY
|
|
451
|
+
});
|
|
452
|
+
getOrCreate(classPresentFrames, obj.category, () => /* @__PURE__ */ new Set()).add(f);
|
|
453
|
+
classMaxConfidence.set(obj.category, Math.max(classMaxConfidence.get(obj.category) ?? 0, obj.score));
|
|
454
|
+
getOrCreate(classTrackIds, obj.category, () => /* @__PURE__ */ new Set()).add(obj.trackId);
|
|
455
|
+
}
|
|
456
|
+
const trackSummaries = Array.from(tracks.entries()).map(([trackId, acc]) => ({
|
|
457
|
+
trackId,
|
|
458
|
+
category: acc.category,
|
|
459
|
+
frames: [acc.firstFrame, acc.lastFrame],
|
|
460
|
+
speedMeanPxS: acc.speedCount > 0 ? acc.speedSum / acc.speedCount : 0,
|
|
461
|
+
centerPath: acc.centers.map((c) => [c.x, c.y])
|
|
462
|
+
})).sort((a, b) => a.trackId - b.trackId);
|
|
463
|
+
const classes = {};
|
|
464
|
+
for (const [category, frameSet] of classPresentFrames) classes[category] = {
|
|
465
|
+
framesPresent: frameSet.size,
|
|
466
|
+
totalFrames,
|
|
467
|
+
maxConfidence: classMaxConfidence.get(category) ?? 0,
|
|
468
|
+
tracks: classTrackIds.get(category)?.size ?? 0
|
|
469
|
+
};
|
|
470
|
+
const masks = {};
|
|
471
|
+
for (const [category, entry] of maskCoverage) masks[category] = { meanCoverage: entry.count > 0 ? entry.sum / entry.count : 0 };
|
|
472
|
+
return {
|
|
473
|
+
source: options.source ?? "frames",
|
|
474
|
+
totalFrames,
|
|
475
|
+
fps,
|
|
476
|
+
durationSec: fps > 0 ? totalFrames / fps : 0,
|
|
477
|
+
tasks: Array.from(taskSet),
|
|
478
|
+
modelDownloads,
|
|
479
|
+
inferenceMs,
|
|
480
|
+
trackCount: trackSummaries.length,
|
|
481
|
+
classes,
|
|
482
|
+
tracks: trackSummaries,
|
|
483
|
+
masks
|
|
484
|
+
};
|
|
485
|
+
}
|
|
486
|
+
|
|
487
|
+
//#endregion
|
|
488
|
+
//#region src/decode/preprocess.ts
|
|
489
|
+
const IMAGENET_BGR_MEAN = [
|
|
490
|
+
103.53,
|
|
491
|
+
116.28,
|
|
492
|
+
123.675
|
|
493
|
+
];
|
|
494
|
+
const IMAGENET_BGR_STD = [
|
|
495
|
+
57.375,
|
|
496
|
+
57.12,
|
|
497
|
+
58.395
|
|
498
|
+
];
|
|
499
|
+
/** Per-family normalization; the input size always comes from the model's registry entry. */
|
|
500
|
+
const PREPROCESS_BY_FAMILY = {
|
|
501
|
+
"rtmdet-ins": {
|
|
502
|
+
fit: "topLeft",
|
|
503
|
+
channels: "bgr",
|
|
504
|
+
mean: IMAGENET_BGR_MEAN,
|
|
505
|
+
std: IMAGENET_BGR_STD,
|
|
506
|
+
padValue: 114
|
|
507
|
+
},
|
|
508
|
+
rtmo: {
|
|
509
|
+
fit: "center",
|
|
510
|
+
channels: "bgr",
|
|
511
|
+
mean: [
|
|
512
|
+
0,
|
|
513
|
+
0,
|
|
514
|
+
0
|
|
515
|
+
],
|
|
516
|
+
std: [
|
|
517
|
+
1,
|
|
518
|
+
1,
|
|
519
|
+
1
|
|
520
|
+
],
|
|
521
|
+
padValue: 114
|
|
522
|
+
},
|
|
523
|
+
selfie: {
|
|
524
|
+
fit: "stretch",
|
|
525
|
+
channels: "rgb",
|
|
526
|
+
mean: [
|
|
527
|
+
0,
|
|
528
|
+
0,
|
|
529
|
+
0
|
|
530
|
+
],
|
|
531
|
+
std: [
|
|
532
|
+
255,
|
|
533
|
+
255,
|
|
534
|
+
255
|
|
535
|
+
],
|
|
536
|
+
padValue: 0
|
|
537
|
+
}
|
|
538
|
+
};
|
|
539
|
+
function preprocessSpec(family, input) {
|
|
540
|
+
return {
|
|
541
|
+
...PREPROCESS_BY_FAMILY[family],
|
|
542
|
+
width: input[0],
|
|
543
|
+
height: input[1]
|
|
544
|
+
};
|
|
545
|
+
}
|
|
546
|
+
function computeInputTransform(srcW, srcH, spec) {
|
|
547
|
+
if (spec.fit === "stretch") return {
|
|
548
|
+
scaleX: spec.width / srcW,
|
|
549
|
+
scaleY: spec.height / srcH,
|
|
550
|
+
offsetX: 0,
|
|
551
|
+
offsetY: 0,
|
|
552
|
+
inputWidth: spec.width,
|
|
553
|
+
inputHeight: spec.height
|
|
554
|
+
};
|
|
555
|
+
const scale = Math.min(spec.width / srcW, spec.height / srcH);
|
|
556
|
+
const contentW = Math.round(srcW * scale);
|
|
557
|
+
const contentH = Math.round(srcH * scale);
|
|
558
|
+
const center = spec.fit === "center";
|
|
559
|
+
return {
|
|
560
|
+
scaleX: scale,
|
|
561
|
+
scaleY: scale,
|
|
562
|
+
offsetX: center ? (spec.width - contentW) / 2 : 0,
|
|
563
|
+
offsetY: center ? (spec.height - contentH) / 2 : 0,
|
|
564
|
+
inputWidth: spec.width,
|
|
565
|
+
inputHeight: spec.height
|
|
566
|
+
};
|
|
567
|
+
}
|
|
568
|
+
/** Model-input point → source pixel point. */
|
|
569
|
+
function toSourcePoint(x, y, t) {
|
|
570
|
+
return {
|
|
571
|
+
x: (x - t.offsetX) / t.scaleX,
|
|
572
|
+
y: (y - t.offsetY) / t.scaleY
|
|
573
|
+
};
|
|
574
|
+
}
|
|
575
|
+
/**
|
|
576
|
+
* Resizes RGBA → NCHW float32 `[1, 3, height, width]` with bilinear sampling, padding,
|
|
577
|
+
* channel reordering and normalization per `spec`. Reuses `out` when large enough
|
|
578
|
+
* (hot-path frame loop).
|
|
579
|
+
*/
|
|
580
|
+
function imageToTensor(image, spec, out) {
|
|
581
|
+
const { width: srcW, height: srcH, data: src } = image;
|
|
582
|
+
const { width: W, height: H } = spec;
|
|
583
|
+
const transform = computeInputTransform(srcW, srcH, spec);
|
|
584
|
+
const { scaleX, scaleY, offsetX, offsetY } = transform;
|
|
585
|
+
const plane = W * H;
|
|
586
|
+
if (!out || out.length < 3 * plane) out = new Float32Array(3 * plane);
|
|
587
|
+
const order = spec.channels === "bgr" ? [
|
|
588
|
+
2,
|
|
589
|
+
1,
|
|
590
|
+
0
|
|
591
|
+
] : [
|
|
592
|
+
0,
|
|
593
|
+
1,
|
|
594
|
+
2
|
|
595
|
+
];
|
|
596
|
+
const [m0, m1, m2] = spec.mean;
|
|
597
|
+
const inv0 = 1 / spec.std[0];
|
|
598
|
+
const inv1 = 1 / spec.std[1];
|
|
599
|
+
const inv2 = 1 / spec.std[2];
|
|
600
|
+
const pad0 = (spec.padValue - m0) * inv0;
|
|
601
|
+
const pad1 = (spec.padValue - m1) * inv1;
|
|
602
|
+
const pad2 = (spec.padValue - m2) * inv2;
|
|
603
|
+
const [c0, c1, c2] = order;
|
|
604
|
+
const x0 = Math.ceil(offsetX);
|
|
605
|
+
const y0 = Math.ceil(offsetY);
|
|
606
|
+
const x1 = Math.min(W, Math.floor(offsetX + srcW * scaleX)) - 1;
|
|
607
|
+
const y1 = Math.min(H, Math.floor(offsetY + srcH * scaleY)) - 1;
|
|
608
|
+
for (let y = 0; y < H; y++) {
|
|
609
|
+
const rowBase = y * W;
|
|
610
|
+
const paddedRow = y < y0 || y > y1;
|
|
611
|
+
const sy = Math.max(0, (y + .5 - offsetY) / scaleY - .5);
|
|
612
|
+
const iy0 = Math.min(srcH - 1, Math.floor(sy));
|
|
613
|
+
const iy1 = Math.min(srcH - 1, iy0 + 1);
|
|
614
|
+
const fy = sy - iy0;
|
|
615
|
+
for (let x = 0; x < W; x++) {
|
|
616
|
+
const o = rowBase + x;
|
|
617
|
+
if (paddedRow || x < x0 || x > x1) {
|
|
618
|
+
out[o] = pad0;
|
|
619
|
+
out[plane + o] = pad1;
|
|
620
|
+
out[2 * plane + o] = pad2;
|
|
621
|
+
continue;
|
|
622
|
+
}
|
|
623
|
+
const sx = Math.max(0, (x + .5 - offsetX) / scaleX - .5);
|
|
624
|
+
const ix0 = Math.min(srcW - 1, Math.floor(sx));
|
|
625
|
+
const ix1 = Math.min(srcW - 1, ix0 + 1);
|
|
626
|
+
const fx = sx - ix0;
|
|
627
|
+
const i00 = (iy0 * srcW + ix0) * 4;
|
|
628
|
+
const i10 = (iy0 * srcW + ix1) * 4;
|
|
629
|
+
const i01 = (iy1 * srcW + ix0) * 4;
|
|
630
|
+
const i11 = (iy1 * srcW + ix1) * 4;
|
|
631
|
+
const w00 = (1 - fx) * (1 - fy);
|
|
632
|
+
const w10 = fx * (1 - fy);
|
|
633
|
+
const w01 = (1 - fx) * fy;
|
|
634
|
+
const w11 = fx * fy;
|
|
635
|
+
out[o] = (src[i00 + c0] * w00 + src[i10 + c0] * w10 + src[i01 + c0] * w01 + src[i11 + c0] * w11 - m0) * inv0;
|
|
636
|
+
out[plane + o] = (src[i00 + c1] * w00 + src[i10 + c1] * w10 + src[i01 + c1] * w01 + src[i11 + c1] * w11 - m1) * inv1;
|
|
637
|
+
out[2 * plane + o] = (src[i00 + c2] * w00 + src[i10 + c2] * w10 + src[i01 + c2] * w01 + src[i11 + c2] * w11 - m2) * inv2;
|
|
638
|
+
}
|
|
639
|
+
}
|
|
640
|
+
return {
|
|
641
|
+
tensor: out,
|
|
642
|
+
transform
|
|
643
|
+
};
|
|
644
|
+
}
|
|
645
|
+
|
|
646
|
+
//#endregion
|
|
647
|
+
//#region src/decode/rtmdet-ins.ts
|
|
648
|
+
/** Source pixels a mask may extend past its box (instance masks are not box-clipped). */
|
|
649
|
+
const MASK_BOX_PAD_PX = 8;
|
|
650
|
+
function decodeRtmdetIns(out, opts) {
|
|
651
|
+
const { sourceWidth: sw, sourceHeight: sh, transform } = opts;
|
|
652
|
+
const classFilter = opts.classes ? new Set(opts.classes.map((c) => c.toLowerCase())) : null;
|
|
653
|
+
const kept = [];
|
|
654
|
+
for (let i = 0; i < out.count; i++) {
|
|
655
|
+
const score = out.dets[i * 5 + 4];
|
|
656
|
+
if (!(score >= opts.confidence)) continue;
|
|
657
|
+
const classIndex = Number(out.labels[i]);
|
|
658
|
+
const category = opts.classNames[classIndex] ?? `class_${classIndex}`;
|
|
659
|
+
if (classFilter && !classFilter.has(category.toLowerCase())) continue;
|
|
660
|
+
const p0 = toSourcePoint(out.dets[i * 5], out.dets[i * 5 + 1], transform);
|
|
661
|
+
const p1 = toSourcePoint(out.dets[i * 5 + 2], out.dets[i * 5 + 3], transform);
|
|
662
|
+
const x0 = clamp$1(p0.x, 0, sw);
|
|
663
|
+
const y0 = clamp$1(p0.y, 0, sh);
|
|
664
|
+
const width = clamp$1(p1.x, 0, sw) - x0;
|
|
665
|
+
const height = clamp$1(p1.y, 0, sh) - y0;
|
|
666
|
+
if (width <= 0 || height <= 0) continue;
|
|
667
|
+
kept.push({
|
|
668
|
+
index: i,
|
|
669
|
+
detection: {
|
|
670
|
+
category,
|
|
671
|
+
score,
|
|
672
|
+
boundingBox: {
|
|
673
|
+
originX: x0,
|
|
674
|
+
originY: y0,
|
|
675
|
+
width,
|
|
676
|
+
height,
|
|
677
|
+
normalizedX: x0 / sw,
|
|
678
|
+
normalizedY: y0 / sh,
|
|
679
|
+
normalizedWidth: width / sw,
|
|
680
|
+
normalizedHeight: height / sh
|
|
681
|
+
}
|
|
682
|
+
}
|
|
683
|
+
});
|
|
684
|
+
}
|
|
685
|
+
kept.sort((a, b) => b.detection.score - a.detection.score);
|
|
686
|
+
const detections = kept.map((k) => k.detection);
|
|
687
|
+
const masks = [];
|
|
688
|
+
if (!out.masks) return {
|
|
689
|
+
detections,
|
|
690
|
+
masks
|
|
691
|
+
};
|
|
692
|
+
const alphaOf = probabilityToAlpha(opts.maskThreshold ?? .5, opts.featherRadius ?? .05);
|
|
693
|
+
const { maskWidth: mw, maskHeight: mh } = out;
|
|
694
|
+
const plane = mw * mh;
|
|
695
|
+
const gx = mw / transform.inputWidth;
|
|
696
|
+
const gy = mh / transform.inputHeight;
|
|
697
|
+
for (let d = 0; d < kept.length; d++) {
|
|
698
|
+
const { index, detection } = kept[d];
|
|
699
|
+
const box = detection.boundingBox;
|
|
700
|
+
const base = index * plane;
|
|
701
|
+
const bx0 = Math.max(0, Math.floor(box.originX) - MASK_BOX_PAD_PX);
|
|
702
|
+
const by0 = Math.max(0, Math.floor(box.originY) - MASK_BOX_PAD_PX);
|
|
703
|
+
const bx1 = Math.min(sw - 1, Math.ceil(box.originX + box.width) + MASK_BOX_PAD_PX);
|
|
704
|
+
const by1 = Math.min(sh - 1, Math.ceil(box.originY + box.height) + MASK_BOX_PAD_PX);
|
|
705
|
+
const mask = new Uint8Array(sw * sh);
|
|
706
|
+
let area = 0;
|
|
707
|
+
for (let sy = by0; sy <= by1; sy++) {
|
|
708
|
+
const my = ((sy + .5) * transform.scaleY + transform.offsetY) * gy - .5;
|
|
709
|
+
const my0 = clamp$1(Math.floor(my), 0, mh - 1);
|
|
710
|
+
const my1 = Math.min(mh - 1, my0 + 1);
|
|
711
|
+
const fy = clamp$1(my - my0, 0, 1);
|
|
712
|
+
const row = sy * sw;
|
|
713
|
+
for (let sx = bx0; sx <= bx1; sx++) {
|
|
714
|
+
const mx = ((sx + .5) * transform.scaleX + transform.offsetX) * gx - .5;
|
|
715
|
+
const mx0 = clamp$1(Math.floor(mx), 0, mw - 1);
|
|
716
|
+
const mx1 = Math.min(mw - 1, mx0 + 1);
|
|
717
|
+
const fx = clamp$1(mx - mx0, 0, 1);
|
|
718
|
+
const alpha = alphaOf(out.masks[base + my0 * mw + mx0] * (1 - fx) * (1 - fy) + out.masks[base + my0 * mw + mx1] * fx * (1 - fy) + out.masks[base + my1 * mw + mx0] * (1 - fx) * fy + out.masks[base + my1 * mw + mx1] * fx * fy);
|
|
719
|
+
if (alpha > 0) {
|
|
720
|
+
mask[row + sx] = alpha;
|
|
721
|
+
area += alpha / 255;
|
|
722
|
+
}
|
|
723
|
+
}
|
|
724
|
+
}
|
|
725
|
+
area = Math.round(area);
|
|
726
|
+
if (area === 0) continue;
|
|
727
|
+
masks.push({
|
|
728
|
+
category: detection.category,
|
|
729
|
+
mask,
|
|
730
|
+
width: sw,
|
|
731
|
+
height: sh,
|
|
732
|
+
area,
|
|
733
|
+
coverage: area / (sw * sh),
|
|
734
|
+
detectionIndex: d
|
|
735
|
+
});
|
|
736
|
+
}
|
|
737
|
+
return {
|
|
738
|
+
detections,
|
|
739
|
+
masks
|
|
740
|
+
};
|
|
741
|
+
}
|
|
742
|
+
/**
|
|
743
|
+
* Probability → 0..255 alpha. A hard threshold when `feather` is 0, otherwise a linear ramp
|
|
744
|
+
* across `[threshold − feather, threshold + feather]` for anti-aliased edges.
|
|
745
|
+
*/
|
|
746
|
+
function probabilityToAlpha(threshold, feather) {
|
|
747
|
+
if (feather <= 0) return (p) => p > threshold ? 255 : 0;
|
|
748
|
+
const lo = threshold - feather;
|
|
749
|
+
const inv = 255 / (2 * feather);
|
|
750
|
+
return (p) => p <= lo ? 0 : p >= threshold + feather ? 255 : (p - lo) * inv;
|
|
751
|
+
}
|
|
752
|
+
function clamp$1(v, lo, hi) {
|
|
753
|
+
return v < lo ? lo : v > hi ? hi : v;
|
|
754
|
+
}
|
|
755
|
+
|
|
756
|
+
//#endregion
|
|
757
|
+
//#region src/decode/rtmo.ts
|
|
758
|
+
const COCO17_KEYPOINT_COUNT = 17;
|
|
759
|
+
/** Keypoint visibility at which a joint counts as seen. */
|
|
760
|
+
const VISIBLE = .3;
|
|
761
|
+
function decodeRtmo(out, opts) {
|
|
762
|
+
const { sourceWidth: sw, sourceHeight: sh, transform } = opts;
|
|
763
|
+
const stride = out.keypointStride;
|
|
764
|
+
const people = [];
|
|
765
|
+
for (let i = 0; i < out.count; i++) {
|
|
766
|
+
const score = out.dets[i * 5 + 4];
|
|
767
|
+
if (!(score >= opts.confidence)) continue;
|
|
768
|
+
const p0 = toSourcePoint(out.dets[i * 5], out.dets[i * 5 + 1], transform);
|
|
769
|
+
const p1 = toSourcePoint(out.dets[i * 5 + 2], out.dets[i * 5 + 3], transform);
|
|
770
|
+
const x0 = clamp(p0.x, 0, sw);
|
|
771
|
+
const y0 = clamp(p0.y, 0, sh);
|
|
772
|
+
const width = clamp(p1.x, 0, sw) - x0;
|
|
773
|
+
const height = clamp(p1.y, 0, sh) - y0;
|
|
774
|
+
if (width <= 0 || height <= 0) continue;
|
|
775
|
+
const keypoints = [];
|
|
776
|
+
const base = i * COCO17_KEYPOINT_COUNT * stride;
|
|
777
|
+
for (let k = 0; k < COCO17_KEYPOINT_COUNT; k++) {
|
|
778
|
+
const o = base + k * stride;
|
|
779
|
+
const p = toSourcePoint(out.keypoints[o], out.keypoints[o + 1], transform);
|
|
780
|
+
keypoints.push({
|
|
781
|
+
x: clamp(p.x, 0, sw),
|
|
782
|
+
y: clamp(p.y, 0, sh),
|
|
783
|
+
visibility: clamp(out.keypoints[o + 2], 0, 1)
|
|
784
|
+
});
|
|
785
|
+
}
|
|
786
|
+
let visible = 0;
|
|
787
|
+
for (const k of keypoints) if (k.visibility >= VISIBLE) visible++;
|
|
788
|
+
if (visible < (opts.minVisibleKeypoints ?? 3)) continue;
|
|
789
|
+
people.push({
|
|
790
|
+
score,
|
|
791
|
+
boundingBox: {
|
|
792
|
+
originX: x0,
|
|
793
|
+
originY: y0,
|
|
794
|
+
width,
|
|
795
|
+
height,
|
|
796
|
+
normalizedX: x0 / sw,
|
|
797
|
+
normalizedY: y0 / sh,
|
|
798
|
+
normalizedWidth: width / sw,
|
|
799
|
+
normalizedHeight: height / sh
|
|
800
|
+
},
|
|
801
|
+
keypoints
|
|
802
|
+
});
|
|
803
|
+
}
|
|
804
|
+
people.sort((a, b) => b.score - a.score);
|
|
805
|
+
const iou = opts.iouThreshold ?? .6;
|
|
806
|
+
const kept = [];
|
|
807
|
+
for (const p of people) if (!kept.some((k) => isDuplicate(k, p, iou))) kept.push(p);
|
|
808
|
+
return { people: kept };
|
|
809
|
+
}
|
|
810
|
+
function clamp(v, lo, hi) {
|
|
811
|
+
return v < lo ? lo : v > hi ? hi : v;
|
|
812
|
+
}
|
|
813
|
+
/**
|
|
814
|
+
* Same person twice: boxes overlap above `iou`, or the joints both detections can see sit
|
|
815
|
+
* within 10% of the stronger detection's box size of each other (boxes may differ when one
|
|
816
|
+
* of them also covers a flowing garment or a shadow).
|
|
817
|
+
*/
|
|
818
|
+
function isDuplicate(a, b, iou) {
|
|
819
|
+
if (computeIoU(a.boundingBox, b.boundingBox) > iou) return true;
|
|
820
|
+
const scale = Math.sqrt(a.boundingBox.width * a.boundingBox.height);
|
|
821
|
+
let sum = 0;
|
|
822
|
+
let n = 0;
|
|
823
|
+
for (let k = 0; k < a.keypoints.length; k++) {
|
|
824
|
+
const ka = a.keypoints[k];
|
|
825
|
+
const kb = b.keypoints[k];
|
|
826
|
+
if (ka.visibility < VISIBLE || kb.visibility < VISIBLE) continue;
|
|
827
|
+
sum += Math.hypot(ka.x - kb.x, ka.y - kb.y);
|
|
828
|
+
n++;
|
|
829
|
+
}
|
|
830
|
+
return n >= 3 && sum / n < .1 * scale;
|
|
831
|
+
}
|
|
832
|
+
|
|
833
|
+
//#endregion
|
|
834
|
+
//#region src/decode/selfie.ts
|
|
835
|
+
function decodeSelfie(alphas, matteWidth, matteHeight, opts) {
|
|
836
|
+
const { sourceWidth: sw, sourceHeight: sh } = opts;
|
|
837
|
+
const alphaOf = probabilityToAlpha(opts.maskThreshold ?? .5, opts.featherRadius ?? .2);
|
|
838
|
+
const mask = new Uint8Array(sw * sh);
|
|
839
|
+
const kx = matteWidth / sw;
|
|
840
|
+
const ky = matteHeight / sh;
|
|
841
|
+
let sum = 0;
|
|
842
|
+
for (let y = 0; y < sh; y++) {
|
|
843
|
+
const my = Math.max(0, (y + .5) * ky - .5);
|
|
844
|
+
const my0 = Math.min(matteHeight - 1, Math.floor(my));
|
|
845
|
+
const my1 = Math.min(matteHeight - 1, my0 + 1);
|
|
846
|
+
const fy = my - my0;
|
|
847
|
+
for (let x = 0; x < sw; x++) {
|
|
848
|
+
const mx = Math.max(0, (x + .5) * kx - .5);
|
|
849
|
+
const mx0 = Math.min(matteWidth - 1, Math.floor(mx));
|
|
850
|
+
const mx1 = Math.min(matteWidth - 1, mx0 + 1);
|
|
851
|
+
const fx = mx - mx0;
|
|
852
|
+
const a = alphaOf(alphas[my0 * matteWidth + mx0] * (1 - fx) * (1 - fy) + alphas[my0 * matteWidth + mx1] * fx * (1 - fy) + alphas[my1 * matteWidth + mx0] * (1 - fx) * fy + alphas[my1 * matteWidth + mx1] * fx * fy);
|
|
853
|
+
mask[y * sw + x] = a;
|
|
854
|
+
sum += a;
|
|
855
|
+
}
|
|
856
|
+
}
|
|
857
|
+
return {
|
|
858
|
+
mask,
|
|
859
|
+
width: sw,
|
|
860
|
+
height: sh,
|
|
861
|
+
coverage: sw * sh > 0 ? sum / (255 * sw * sh) : 0
|
|
862
|
+
};
|
|
863
|
+
}
|
|
864
|
+
|
|
865
|
+
//#endregion
|
|
866
|
+
//#region src/pose/keypoints.ts
|
|
867
|
+
/** COCO-17 keypoint indices (RTMO output order). */
|
|
868
|
+
const COCO17_KEYPOINTS = {
|
|
869
|
+
NOSE: 0,
|
|
870
|
+
LEFT_EYE: 1,
|
|
871
|
+
RIGHT_EYE: 2,
|
|
872
|
+
LEFT_EAR: 3,
|
|
873
|
+
RIGHT_EAR: 4,
|
|
874
|
+
LEFT_SHOULDER: 5,
|
|
875
|
+
RIGHT_SHOULDER: 6,
|
|
876
|
+
LEFT_ELBOW: 7,
|
|
877
|
+
RIGHT_ELBOW: 8,
|
|
878
|
+
LEFT_WRIST: 9,
|
|
879
|
+
RIGHT_WRIST: 10,
|
|
880
|
+
LEFT_HIP: 11,
|
|
881
|
+
RIGHT_HIP: 12,
|
|
882
|
+
LEFT_KNEE: 13,
|
|
883
|
+
RIGHT_KNEE: 14,
|
|
884
|
+
LEFT_ANKLE: 15,
|
|
885
|
+
RIGHT_ANKLE: 16
|
|
886
|
+
};
|
|
887
|
+
const COCO17_KEYPOINT_NAMES = [
|
|
888
|
+
"nose",
|
|
889
|
+
"leftEye",
|
|
890
|
+
"rightEye",
|
|
891
|
+
"leftEar",
|
|
892
|
+
"rightEar",
|
|
893
|
+
"leftShoulder",
|
|
894
|
+
"rightShoulder",
|
|
895
|
+
"leftElbow",
|
|
896
|
+
"rightElbow",
|
|
897
|
+
"leftWrist",
|
|
898
|
+
"rightWrist",
|
|
899
|
+
"leftHip",
|
|
900
|
+
"rightHip",
|
|
901
|
+
"leftKnee",
|
|
902
|
+
"rightKnee",
|
|
903
|
+
"leftAnkle",
|
|
904
|
+
"rightAnkle"
|
|
905
|
+
];
|
|
906
|
+
/** COCO-17 skeleton bones for the WebGPU skeleton renderer (indices above). */
|
|
907
|
+
const COCO17_BONES = [
|
|
908
|
+
{
|
|
909
|
+
from: COCO17_KEYPOINTS.NOSE,
|
|
910
|
+
to: COCO17_KEYPOINTS.LEFT_SHOULDER,
|
|
911
|
+
color: [
|
|
912
|
+
0,
|
|
913
|
+
1,
|
|
914
|
+
1,
|
|
915
|
+
1
|
|
916
|
+
]
|
|
917
|
+
},
|
|
918
|
+
{
|
|
919
|
+
from: COCO17_KEYPOINTS.NOSE,
|
|
920
|
+
to: COCO17_KEYPOINTS.RIGHT_SHOULDER,
|
|
921
|
+
color: [
|
|
922
|
+
0,
|
|
923
|
+
1,
|
|
924
|
+
1,
|
|
925
|
+
1
|
|
926
|
+
]
|
|
927
|
+
},
|
|
928
|
+
{
|
|
929
|
+
from: COCO17_KEYPOINTS.LEFT_SHOULDER,
|
|
930
|
+
to: COCO17_KEYPOINTS.RIGHT_SHOULDER,
|
|
931
|
+
color: [
|
|
932
|
+
1,
|
|
933
|
+
0,
|
|
934
|
+
0,
|
|
935
|
+
1
|
|
936
|
+
]
|
|
937
|
+
},
|
|
938
|
+
{
|
|
939
|
+
from: COCO17_KEYPOINTS.LEFT_SHOULDER,
|
|
940
|
+
to: COCO17_KEYPOINTS.LEFT_ELBOW,
|
|
941
|
+
color: [
|
|
942
|
+
1,
|
|
943
|
+
.333,
|
|
944
|
+
0,
|
|
945
|
+
1
|
|
946
|
+
]
|
|
947
|
+
},
|
|
948
|
+
{
|
|
949
|
+
from: COCO17_KEYPOINTS.LEFT_ELBOW,
|
|
950
|
+
to: COCO17_KEYPOINTS.LEFT_WRIST,
|
|
951
|
+
color: [
|
|
952
|
+
1,
|
|
953
|
+
.667,
|
|
954
|
+
0,
|
|
955
|
+
1
|
|
956
|
+
]
|
|
957
|
+
},
|
|
958
|
+
{
|
|
959
|
+
from: COCO17_KEYPOINTS.RIGHT_SHOULDER,
|
|
960
|
+
to: COCO17_KEYPOINTS.RIGHT_ELBOW,
|
|
961
|
+
color: [
|
|
962
|
+
1,
|
|
963
|
+
1,
|
|
964
|
+
0,
|
|
965
|
+
1
|
|
966
|
+
]
|
|
967
|
+
},
|
|
968
|
+
{
|
|
969
|
+
from: COCO17_KEYPOINTS.RIGHT_ELBOW,
|
|
970
|
+
to: COCO17_KEYPOINTS.RIGHT_WRIST,
|
|
971
|
+
color: [
|
|
972
|
+
.667,
|
|
973
|
+
1,
|
|
974
|
+
0,
|
|
975
|
+
1
|
|
976
|
+
]
|
|
977
|
+
},
|
|
978
|
+
{
|
|
979
|
+
from: COCO17_KEYPOINTS.LEFT_SHOULDER,
|
|
980
|
+
to: COCO17_KEYPOINTS.LEFT_HIP,
|
|
981
|
+
color: [
|
|
982
|
+
.333,
|
|
983
|
+
1,
|
|
984
|
+
0,
|
|
985
|
+
1
|
|
986
|
+
]
|
|
987
|
+
},
|
|
988
|
+
{
|
|
989
|
+
from: COCO17_KEYPOINTS.RIGHT_SHOULDER,
|
|
990
|
+
to: COCO17_KEYPOINTS.RIGHT_HIP,
|
|
991
|
+
color: [
|
|
992
|
+
0,
|
|
993
|
+
1,
|
|
994
|
+
0,
|
|
995
|
+
1
|
|
996
|
+
]
|
|
997
|
+
},
|
|
998
|
+
{
|
|
999
|
+
from: COCO17_KEYPOINTS.LEFT_HIP,
|
|
1000
|
+
to: COCO17_KEYPOINTS.RIGHT_HIP,
|
|
1001
|
+
color: [
|
|
1002
|
+
0,
|
|
1003
|
+
1,
|
|
1004
|
+
.333,
|
|
1005
|
+
1
|
|
1006
|
+
]
|
|
1007
|
+
},
|
|
1008
|
+
{
|
|
1009
|
+
from: COCO17_KEYPOINTS.LEFT_HIP,
|
|
1010
|
+
to: COCO17_KEYPOINTS.LEFT_KNEE,
|
|
1011
|
+
color: [
|
|
1012
|
+
0,
|
|
1013
|
+
1,
|
|
1014
|
+
.667,
|
|
1015
|
+
1
|
|
1016
|
+
]
|
|
1017
|
+
},
|
|
1018
|
+
{
|
|
1019
|
+
from: COCO17_KEYPOINTS.LEFT_KNEE,
|
|
1020
|
+
to: COCO17_KEYPOINTS.LEFT_ANKLE,
|
|
1021
|
+
color: [
|
|
1022
|
+
0,
|
|
1023
|
+
1,
|
|
1024
|
+
1,
|
|
1025
|
+
1
|
|
1026
|
+
]
|
|
1027
|
+
},
|
|
1028
|
+
{
|
|
1029
|
+
from: COCO17_KEYPOINTS.RIGHT_HIP,
|
|
1030
|
+
to: COCO17_KEYPOINTS.RIGHT_KNEE,
|
|
1031
|
+
color: [
|
|
1032
|
+
0,
|
|
1033
|
+
.667,
|
|
1034
|
+
1,
|
|
1035
|
+
1
|
|
1036
|
+
]
|
|
1037
|
+
},
|
|
1038
|
+
{
|
|
1039
|
+
from: COCO17_KEYPOINTS.RIGHT_KNEE,
|
|
1040
|
+
to: COCO17_KEYPOINTS.RIGHT_ANKLE,
|
|
1041
|
+
color: [
|
|
1042
|
+
0,
|
|
1043
|
+
.333,
|
|
1044
|
+
1,
|
|
1045
|
+
1
|
|
1046
|
+
]
|
|
1047
|
+
}
|
|
1048
|
+
];
|
|
1049
|
+
|
|
1050
|
+
//#endregion
|
|
1051
|
+
//#region src/gpu/pose-skeleton-renderer.ts
|
|
1052
|
+
/**
|
|
1053
|
+
* WebGPU skeleton renderer for pose results.
|
|
1054
|
+
*
|
|
1055
|
+
* Rasterizes the primary person's COCO-17 keypoints into an OpenPose-style
|
|
1056
|
+
* conditioning texture natively in VRAM, using the COCO-17 bone set.
|
|
1057
|
+
*/
|
|
1058
|
+
var PoseSkeletonRenderer = class {
|
|
1059
|
+
pipeline;
|
|
1060
|
+
constructor(device) {
|
|
1061
|
+
this.pipeline = new PoseSkeletonComputePipeline(device);
|
|
1062
|
+
}
|
|
1063
|
+
renderToTexture(keypoints, options, bones = COCO17_BONES) {
|
|
1064
|
+
return this.pipeline.execute(keypoints, options, bones);
|
|
1065
|
+
}
|
|
1066
|
+
destroy() {
|
|
1067
|
+
this.pipeline.destroy();
|
|
1068
|
+
}
|
|
1069
|
+
};
|
|
1070
|
+
|
|
1071
|
+
//#endregion
|
|
1072
|
+
//#region src/gpu/segmentation-texture-pool.ts
|
|
1073
|
+
var SegmentationTexturePool = class {
|
|
1074
|
+
device;
|
|
1075
|
+
pool = /* @__PURE__ */ new Map();
|
|
1076
|
+
constructor(device) {
|
|
1077
|
+
this.device = device;
|
|
1078
|
+
}
|
|
1079
|
+
getOrCreateTexture(key, options) {
|
|
1080
|
+
const existing = this.pool.get(key);
|
|
1081
|
+
if (existing && existing.width === options.width && existing.height === options.height) return existing;
|
|
1082
|
+
if (existing) existing.destroy();
|
|
1083
|
+
const format = options.format ?? "rgba8unorm";
|
|
1084
|
+
const texture = this.device.createTexture({
|
|
1085
|
+
label: options.label ?? `vision_segmentation_${key}`,
|
|
1086
|
+
size: [
|
|
1087
|
+
options.width,
|
|
1088
|
+
options.height,
|
|
1089
|
+
1
|
|
1090
|
+
],
|
|
1091
|
+
format,
|
|
1092
|
+
usage: GPUTextureUsage.TEXTURE_BINDING | GPUTextureUsage.COPY_DST | GPUTextureUsage.RENDER_ATTACHMENT
|
|
1093
|
+
});
|
|
1094
|
+
this.pool.set(key, texture);
|
|
1095
|
+
return texture;
|
|
1096
|
+
}
|
|
1097
|
+
uploadMask(key, maskData, width, height) {
|
|
1098
|
+
const isRgba = maskData.length >= width * height * 4;
|
|
1099
|
+
const format = isRgba ? "rgba8unorm" : "r8unorm";
|
|
1100
|
+
const texture = this.getOrCreateTexture(key, {
|
|
1101
|
+
width,
|
|
1102
|
+
height,
|
|
1103
|
+
format
|
|
1104
|
+
});
|
|
1105
|
+
let uploadBytes;
|
|
1106
|
+
if (maskData instanceof Float32Array) {
|
|
1107
|
+
uploadBytes = new Uint8Array(maskData.length);
|
|
1108
|
+
for (let i = 0; i < maskData.length; i++) uploadBytes[i] = Math.round(Math.max(0, Math.min(1, maskData[i])) * 255);
|
|
1109
|
+
} else uploadBytes = maskData;
|
|
1110
|
+
const bytesPerRow = width * (isRgba ? 4 : 1);
|
|
1111
|
+
const alignedBytesPerRow = Math.ceil(bytesPerRow / 256) * 256;
|
|
1112
|
+
let finalBuffer;
|
|
1113
|
+
if (alignedBytesPerRow === bytesPerRow) finalBuffer = uploadBytes;
|
|
1114
|
+
else {
|
|
1115
|
+
finalBuffer = new Uint8Array(alignedBytesPerRow * height);
|
|
1116
|
+
for (let y = 0; y < height; y++) finalBuffer.set(uploadBytes.subarray(y * bytesPerRow, (y + 1) * bytesPerRow), y * alignedBytesPerRow);
|
|
1117
|
+
}
|
|
1118
|
+
this.device.queue.writeTexture({ texture }, finalBuffer.buffer, {
|
|
1119
|
+
offset: finalBuffer.byteOffset,
|
|
1120
|
+
bytesPerRow: alignedBytesPerRow,
|
|
1121
|
+
rowsPerImage: height
|
|
1122
|
+
}, {
|
|
1123
|
+
width,
|
|
1124
|
+
height
|
|
1125
|
+
});
|
|
1126
|
+
return texture;
|
|
1127
|
+
}
|
|
1128
|
+
destroy() {
|
|
1129
|
+
for (const tex of this.pool.values()) tex.destroy();
|
|
1130
|
+
this.pool.clear();
|
|
1131
|
+
}
|
|
1132
|
+
};
|
|
1133
|
+
|
|
1134
|
+
//#endregion
|
|
1135
|
+
//#region src/model/model-store.ts
|
|
1136
|
+
/**
|
|
1137
|
+
* Resolves the default model cache directory:
|
|
1138
|
+
* 1. `FRAMEFIELDS_MODELS_DIR`
|
|
1139
|
+
* 2. `~/.cache/framefields/models`
|
|
1140
|
+
*/
|
|
1141
|
+
function getDefaultModelsDir() {
|
|
1142
|
+
if (process.env.FRAMEFIELDS_MODELS_DIR) return resolve(process.env.FRAMEFIELDS_MODELS_DIR);
|
|
1143
|
+
return resolve(homedir(), ".cache/framefields/models");
|
|
1144
|
+
}
|
|
1145
|
+
/**
|
|
1146
|
+
* Lazy model store. The constructor performs ZERO I/O; the only entry that can hit the
|
|
1147
|
+
* network is `ensure(key)`, called by an inference function the first time it runs.
|
|
1148
|
+
* Concurrent callers of the same key share a single in-flight download (promise dedupe),
|
|
1149
|
+
* downloads are written atomically (temp + rename), and verified files on disk are reused
|
|
1150
|
+
* across processes.
|
|
1151
|
+
*/
|
|
1152
|
+
var VisionModelStore = class {
|
|
1153
|
+
_modelsDir;
|
|
1154
|
+
_baseUrl;
|
|
1155
|
+
_timeoutMs;
|
|
1156
|
+
_retries;
|
|
1157
|
+
_retryDelayMs;
|
|
1158
|
+
_verify;
|
|
1159
|
+
_fetch;
|
|
1160
|
+
_onProgress;
|
|
1161
|
+
_inflight = /* @__PURE__ */ new Map();
|
|
1162
|
+
_buffers = /* @__PURE__ */ new Map();
|
|
1163
|
+
_status = /* @__PURE__ */ new Map();
|
|
1164
|
+
constructor(options = {}) {
|
|
1165
|
+
this._modelsDir = options.modelsDir ? resolve(options.modelsDir) : getDefaultModelsDir();
|
|
1166
|
+
this._baseUrl = options.baseUrl ?? process.env.FRAMEFIELDS_MODELS_BASE_URL ?? void 0;
|
|
1167
|
+
this._timeoutMs = options.timeoutMs;
|
|
1168
|
+
this._retries = Math.max(0, options.retries ?? 2);
|
|
1169
|
+
this._retryDelayMs = options.retryDelayMs ?? 1e3;
|
|
1170
|
+
this._verify = options.verify !== false;
|
|
1171
|
+
this._fetch = options.fetchImpl ?? globalThis.fetch;
|
|
1172
|
+
this._onProgress = options.onProgress;
|
|
1173
|
+
}
|
|
1174
|
+
get modelsDir() {
|
|
1175
|
+
return this._modelsDir;
|
|
1176
|
+
}
|
|
1177
|
+
pathFor(key) {
|
|
1178
|
+
return join(this._modelsDir, VISION_MODELS[key].filename);
|
|
1179
|
+
}
|
|
1180
|
+
descriptor(key) {
|
|
1181
|
+
return VISION_MODELS[key];
|
|
1182
|
+
}
|
|
1183
|
+
/** The URL `ensure(key)` downloads from (mirror when `baseUrl` is set). */
|
|
1184
|
+
urlFor(key) {
|
|
1185
|
+
const desc = VISION_MODELS[key];
|
|
1186
|
+
if (!this._baseUrl) return desc.url;
|
|
1187
|
+
return `${this._baseUrl.replace(/\/+$/, "")}/${desc.filename}`;
|
|
1188
|
+
}
|
|
1189
|
+
/** True when a complete cached file exists (exact registry size when verifying). */
|
|
1190
|
+
has(key) {
|
|
1191
|
+
const targetPath = this.pathFor(key);
|
|
1192
|
+
if (!existsSync(targetPath)) return false;
|
|
1193
|
+
try {
|
|
1194
|
+
const size = statSync(targetPath).size;
|
|
1195
|
+
return this._verify ? size === VISION_MODELS[key].bytes : size > 0;
|
|
1196
|
+
} catch {
|
|
1197
|
+
return false;
|
|
1198
|
+
}
|
|
1199
|
+
}
|
|
1200
|
+
/** Drops a cached model (file + memory) so the next `ensure` re-downloads it. */
|
|
1201
|
+
async evict(key) {
|
|
1202
|
+
this._buffers.delete(key);
|
|
1203
|
+
this._status.set(key, "pending");
|
|
1204
|
+
await unlink(this.pathFor(key)).catch(() => void 0);
|
|
1205
|
+
}
|
|
1206
|
+
status(key) {
|
|
1207
|
+
return this._status.get(key) ?? "pending";
|
|
1208
|
+
}
|
|
1209
|
+
/** Snapshot of every tracked model's status (empty until anything is fetched). */
|
|
1210
|
+
statuses() {
|
|
1211
|
+
return new Map(this._status);
|
|
1212
|
+
}
|
|
1213
|
+
/** True when the model is resident in memory (loaded or downloaded this process). */
|
|
1214
|
+
isLoaded(key) {
|
|
1215
|
+
return this._buffers.has(key);
|
|
1216
|
+
}
|
|
1217
|
+
/**
|
|
1218
|
+
* THE only download entry point. Returns the model bytes from memory (cached),
|
|
1219
|
+
* loading from disk or fetching from the registry URL as needed.
|
|
1220
|
+
*/
|
|
1221
|
+
async ensure(key) {
|
|
1222
|
+
const inMemory = this._buffers.get(key);
|
|
1223
|
+
if (inMemory) return inMemory;
|
|
1224
|
+
const inflight = this._inflight.get(key);
|
|
1225
|
+
if (inflight) return inflight;
|
|
1226
|
+
const promise = this.load(key).finally(() => this._inflight.delete(key));
|
|
1227
|
+
this._inflight.set(key, promise);
|
|
1228
|
+
return promise;
|
|
1229
|
+
}
|
|
1230
|
+
/** Explicit warm-up — fetch multiple models ahead of use (agents, renderers). */
|
|
1231
|
+
async preload(keys) {
|
|
1232
|
+
await Promise.all([...new Set(keys)].map((key) => this.ensure(key)));
|
|
1233
|
+
}
|
|
1234
|
+
async load(key) {
|
|
1235
|
+
const targetPath = this.pathFor(key);
|
|
1236
|
+
if (this.has(key)) {
|
|
1237
|
+
this._status.set(key, "ready");
|
|
1238
|
+
return this.readBuffer(key, targetPath);
|
|
1239
|
+
}
|
|
1240
|
+
const downloadUrl = this.urlFor(key);
|
|
1241
|
+
this._status.set(key, "downloading");
|
|
1242
|
+
if (!existsSync(dirname(targetPath))) mkdirSync(dirname(targetPath), { recursive: true });
|
|
1243
|
+
const tempPath = `${targetPath}.tmp.${Date.now()}`;
|
|
1244
|
+
try {
|
|
1245
|
+
const { buffer, totalBytes } = await this.download(key, downloadUrl);
|
|
1246
|
+
if (buffer.byteLength === 0) throw new Error(`Downloaded vision model '${key}' is empty (0 bytes).`);
|
|
1247
|
+
if (this._verify) this.verifyBytes(key, buffer, downloadUrl);
|
|
1248
|
+
this._onProgress?.(key, buffer.byteLength, totalBytes || buffer.byteLength);
|
|
1249
|
+
await writeFile(tempPath, buffer);
|
|
1250
|
+
await rename(tempPath, targetPath);
|
|
1251
|
+
this._buffers.set(key, buffer);
|
|
1252
|
+
this._status.set(key, "ready");
|
|
1253
|
+
return buffer;
|
|
1254
|
+
} catch (error) {
|
|
1255
|
+
this._status.set(key, "error");
|
|
1256
|
+
if (existsSync(tempPath)) await unlink(tempPath).catch(() => void 0);
|
|
1257
|
+
throw error;
|
|
1258
|
+
}
|
|
1259
|
+
}
|
|
1260
|
+
/** Fetches with retries on transient failures (network errors, HTTP 429/5xx). */
|
|
1261
|
+
async download(key, url) {
|
|
1262
|
+
for (let attempt = 0;; attempt++) {
|
|
1263
|
+
let transient;
|
|
1264
|
+
try {
|
|
1265
|
+
const response = await this._fetch(url, { signal: this._timeoutMs ? AbortSignal.timeout(this._timeoutMs) : void 0 });
|
|
1266
|
+
if (response.ok) {
|
|
1267
|
+
const totalBytes = Number(response.headers.get("content-length") ?? 0);
|
|
1268
|
+
return {
|
|
1269
|
+
buffer: new Uint8Array(await response.arrayBuffer()),
|
|
1270
|
+
totalBytes
|
|
1271
|
+
};
|
|
1272
|
+
}
|
|
1273
|
+
const error = /* @__PURE__ */ new Error(`Failed to download vision model '${key}' from ${url}: HTTP ${response.status} ${response.statusText}`);
|
|
1274
|
+
if (response.status !== 429 && response.status < 500) throw error;
|
|
1275
|
+
transient = error;
|
|
1276
|
+
} catch (error) {
|
|
1277
|
+
if (!(error instanceof TypeError)) throw error;
|
|
1278
|
+
transient = error;
|
|
1279
|
+
}
|
|
1280
|
+
if (attempt >= this._retries) throw new Error(`Failed to download vision model '${key}' from ${url} after ${attempt + 1} attempts: ${String(transient)}`, { cause: transient });
|
|
1281
|
+
await new Promise((r) => setTimeout(r, this._retryDelayMs * 2 ** attempt));
|
|
1282
|
+
}
|
|
1283
|
+
}
|
|
1284
|
+
verifyBytes(key, buffer, url) {
|
|
1285
|
+
const desc = VISION_MODELS[key];
|
|
1286
|
+
if (buffer.byteLength !== desc.bytes) throw new Error(`Vision model '${key}' from ${url} is ${buffer.byteLength} bytes, expected ${desc.bytes}.`);
|
|
1287
|
+
const digest = createHash("sha256").update(buffer).digest("hex");
|
|
1288
|
+
if (digest !== desc.sha256) throw new Error(`Vision model '${key}' from ${url} failed its SHA-256 check (got ${digest}, expected ${desc.sha256}).`);
|
|
1289
|
+
}
|
|
1290
|
+
async readBuffer(key, targetPath) {
|
|
1291
|
+
const fileBuffer = await readFile(targetPath);
|
|
1292
|
+
const buffer = new Uint8Array(fileBuffer.buffer, fileBuffer.byteOffset, fileBuffer.byteLength);
|
|
1293
|
+
this._buffers.set(key, buffer);
|
|
1294
|
+
return buffer;
|
|
1295
|
+
}
|
|
1296
|
+
/** Drops in-memory buffers (model stays on disk for the next run). */
|
|
1297
|
+
clearMemoryCache() {
|
|
1298
|
+
this._buffers.clear();
|
|
1299
|
+
}
|
|
1300
|
+
};
|
|
1301
|
+
|
|
1302
|
+
//#endregion
|
|
1303
|
+
//#region src/runner/canonical-baselines.ts
|
|
1304
|
+
/** Neutral fallbacks so signal evaluators always have a well-formed frame to read. */
|
|
1305
|
+
function createNeutralObjectResult() {
|
|
1306
|
+
return {
|
|
1307
|
+
objects: [],
|
|
1308
|
+
rawDetections: []
|
|
1309
|
+
};
|
|
1310
|
+
}
|
|
1311
|
+
function createNeutralPoseResult() {
|
|
1312
|
+
return { people: [] };
|
|
1313
|
+
}
|
|
1314
|
+
/** 17 neutral COCO keypoints at frame center — used by pose signals when nobody is detected. */
|
|
1315
|
+
function createNeutralLandmarks() {
|
|
1316
|
+
const landmarks = [];
|
|
1317
|
+
for (let i = 0; i < 17; i++) landmarks.push({
|
|
1318
|
+
x: .5,
|
|
1319
|
+
y: .5,
|
|
1320
|
+
z: 0,
|
|
1321
|
+
visibility: 0
|
|
1322
|
+
});
|
|
1323
|
+
return landmarks;
|
|
1324
|
+
}
|
|
1325
|
+
|
|
1326
|
+
//#endregion
|
|
1327
|
+
//#region src/runtime/webgpu-provider.ts
|
|
1328
|
+
const ORT_WEB_SPECIFIER = "onnxruntime-web";
|
|
1329
|
+
const defaultLoader = () => import(
|
|
1330
|
+
/* @vite-ignore */
|
|
1331
|
+
ORT_WEB_SPECIFIER
|
|
1332
|
+
);
|
|
1333
|
+
var WebGPUProvider = class {
|
|
1334
|
+
kind = "webgpu";
|
|
1335
|
+
_loader;
|
|
1336
|
+
_providers;
|
|
1337
|
+
_wasmFallback;
|
|
1338
|
+
_ortPromise;
|
|
1339
|
+
constructor(options = {}) {
|
|
1340
|
+
this._loader = options.loader ?? defaultLoader;
|
|
1341
|
+
this._providers = options.executionProviders ?? ["webgpu", "wasm"];
|
|
1342
|
+
this._wasmFallback = options.wasmFallback !== false;
|
|
1343
|
+
}
|
|
1344
|
+
ort() {
|
|
1345
|
+
if (!this._ortPromise) this._ortPromise = this._loader().catch((err) => {
|
|
1346
|
+
this._ortPromise = void 0;
|
|
1347
|
+
throw new Error(`onnxruntime-web failed to load (${String(err)}). Install it in the browser bundle or pass a custom loader.`);
|
|
1348
|
+
});
|
|
1349
|
+
return this._ortPromise;
|
|
1350
|
+
}
|
|
1351
|
+
async createSession(modelBytes) {
|
|
1352
|
+
const ort = await this.ort();
|
|
1353
|
+
let session;
|
|
1354
|
+
try {
|
|
1355
|
+
session = await ort.InferenceSession.create(modelBytes, {
|
|
1356
|
+
executionProviders: [...this._providers],
|
|
1357
|
+
graphOptimizationLevel: "all",
|
|
1358
|
+
logSeverityLevel: 3
|
|
1359
|
+
});
|
|
1360
|
+
} catch (err) {
|
|
1361
|
+
if (!this._wasmFallback || !this._providers.includes("webgpu")) throw err;
|
|
1362
|
+
session = await ort.InferenceSession.create(modelBytes, {
|
|
1363
|
+
executionProviders: ["wasm"],
|
|
1364
|
+
graphOptimizationLevel: "all",
|
|
1365
|
+
logSeverityLevel: 3
|
|
1366
|
+
});
|
|
1367
|
+
}
|
|
1368
|
+
return {
|
|
1369
|
+
run: async (input) => {
|
|
1370
|
+
const inputName = session.inputNames.includes(input.name) ? input.name : session.inputNames[0] ?? input.name;
|
|
1371
|
+
const results = await session.run({ [inputName]: new ort.Tensor("float32", input.data, [...input.dims]) });
|
|
1372
|
+
const outputs = {};
|
|
1373
|
+
for (const [name, tensor] of Object.entries(results)) outputs[name] = {
|
|
1374
|
+
data: tensor.data,
|
|
1375
|
+
dims: [...tensor.dims]
|
|
1376
|
+
};
|
|
1377
|
+
return outputs;
|
|
1378
|
+
},
|
|
1379
|
+
release: () => {
|
|
1380
|
+
session.release();
|
|
1381
|
+
}
|
|
1382
|
+
};
|
|
1383
|
+
}
|
|
1384
|
+
};
|
|
1385
|
+
/**
|
|
1386
|
+
* Convenience factory: a `SessionProvider` type guard for the browser entry. Kept tiny so
|
|
1387
|
+
* `@framefields/vision/web` stays tree-shakeable.
|
|
1388
|
+
*/
|
|
1389
|
+
function createWebGPUProvider(options) {
|
|
1390
|
+
return new WebGPUProvider(options);
|
|
1391
|
+
}
|
|
1392
|
+
/** True when the environment exposes a WebGPU device (synchronous capability probe). */
|
|
1393
|
+
function hasWebGPU() {
|
|
1394
|
+
return typeof navigator !== "undefined" && "gpu" in navigator;
|
|
1395
|
+
}
|
|
1396
|
+
|
|
1397
|
+
//#endregion
|
|
1398
|
+
//#region src/runtime/node-webgpu-provider.ts
|
|
1399
|
+
const ORT_WEBGPU_SPECIFIER = "onnxruntime-web/webgpu";
|
|
1400
|
+
const WEBGPU_SPECIFIER = "webgpu";
|
|
1401
|
+
let nodeWebGPUReady;
|
|
1402
|
+
/**
|
|
1403
|
+
* Ensure `globalThis.navigator.gpu` exists. In the renderer this is already the
|
|
1404
|
+
* Dawn device; otherwise imports the `webgpu` (Dawn) package and installs its
|
|
1405
|
+
* globals plus an adapter, which is what onnxruntime-web's WebGPU EP needs.
|
|
1406
|
+
*/
|
|
1407
|
+
function ensureNodeWebGPU() {
|
|
1408
|
+
if (!nodeWebGPUReady) {
|
|
1409
|
+
nodeWebGPUReady = (async () => {
|
|
1410
|
+
const g = globalThis;
|
|
1411
|
+
if (g.navigator?.gpu) return;
|
|
1412
|
+
const { create, globals } = await import(
|
|
1413
|
+
/* @vite-ignore */
|
|
1414
|
+
WEBGPU_SPECIFIER
|
|
1415
|
+
);
|
|
1416
|
+
Object.assign(globalThis, globals);
|
|
1417
|
+
const nav = g.navigator ??= {};
|
|
1418
|
+
Object.defineProperty(nav, "gpu", {
|
|
1419
|
+
value: create([]),
|
|
1420
|
+
configurable: true,
|
|
1421
|
+
writable: true
|
|
1422
|
+
});
|
|
1423
|
+
})();
|
|
1424
|
+
nodeWebGPUReady.catch(() => {
|
|
1425
|
+
nodeWebGPUReady = void 0;
|
|
1426
|
+
});
|
|
1427
|
+
}
|
|
1428
|
+
return nodeWebGPUReady;
|
|
1429
|
+
}
|
|
1430
|
+
var NodeWebGPUProvider = class {
|
|
1431
|
+
kind = "webgpu";
|
|
1432
|
+
_inner;
|
|
1433
|
+
constructor(options = {}) {
|
|
1434
|
+
const loader = options.loader ?? (async () => {
|
|
1435
|
+
await ensureNodeWebGPU();
|
|
1436
|
+
const mod = await import(
|
|
1437
|
+
/* @vite-ignore */
|
|
1438
|
+
ORT_WEBGPU_SPECIFIER
|
|
1439
|
+
);
|
|
1440
|
+
if (mod.env?.wasm) mod.env.wasm.numThreads = 1;
|
|
1441
|
+
return mod;
|
|
1442
|
+
});
|
|
1443
|
+
this._inner = new WebGPUProvider({
|
|
1444
|
+
wasmFallback: false,
|
|
1445
|
+
...options,
|
|
1446
|
+
loader
|
|
1447
|
+
});
|
|
1448
|
+
}
|
|
1449
|
+
createSession(modelBytes) {
|
|
1450
|
+
return this._inner.createSession(modelBytes);
|
|
1451
|
+
}
|
|
1452
|
+
};
|
|
1453
|
+
function createNodeWebGPUProvider(options) {
|
|
1454
|
+
return new NodeWebGPUProvider(options);
|
|
1455
|
+
}
|
|
1456
|
+
|
|
1457
|
+
//#endregion
|
|
1458
|
+
//#region src/runtime/session-provider.ts
|
|
1459
|
+
let defaultProvider;
|
|
1460
|
+
/**
|
|
1461
|
+
* Sets the provider runners use when none is passed, for hosts that create
|
|
1462
|
+
* runners indirectly (e.g. a browser player rendering vision nodes). `undefined`
|
|
1463
|
+
* restores the default, onnxruntime-node.
|
|
1464
|
+
*/
|
|
1465
|
+
function setDefaultSessionProvider(factory) {
|
|
1466
|
+
defaultProvider = factory;
|
|
1467
|
+
}
|
|
1468
|
+
/**
|
|
1469
|
+
* Prefers the GPU (onnxruntime-web's WebGPU EP on the process's WebGPU device —
|
|
1470
|
+
* the Dawn device the renderer already creates, or one this ensures) and falls
|
|
1471
|
+
* back to the CPU provider (onnxruntime-node) when a WebGPU session cannot be
|
|
1472
|
+
* created: no device, no Dawn, or the EP unavailable. The choice is made on the
|
|
1473
|
+
* first session and reused for the runner's lifetime.
|
|
1474
|
+
*/
|
|
1475
|
+
var AutoSessionProvider = class {
|
|
1476
|
+
_resolved;
|
|
1477
|
+
get kind() {
|
|
1478
|
+
return this._resolved?.kind ?? "webgpu";
|
|
1479
|
+
}
|
|
1480
|
+
async createSession(modelBytes) {
|
|
1481
|
+
if (!this._resolved) {
|
|
1482
|
+
const webgpu = new NodeWebGPUProvider({ wasmFallback: false });
|
|
1483
|
+
try {
|
|
1484
|
+
const session = await webgpu.createSession(modelBytes);
|
|
1485
|
+
this._resolved = webgpu;
|
|
1486
|
+
return session;
|
|
1487
|
+
} catch {
|
|
1488
|
+
this._resolved = new NodeSessionProvider();
|
|
1489
|
+
}
|
|
1490
|
+
}
|
|
1491
|
+
return this._resolved.createSession(modelBytes);
|
|
1492
|
+
}
|
|
1493
|
+
};
|
|
1494
|
+
/**
|
|
1495
|
+
* The provider for a runner created without one: GPU if possible, otherwise CPU.
|
|
1496
|
+
*/
|
|
1497
|
+
function createDefaultSessionProvider() {
|
|
1498
|
+
return defaultProvider?.() ?? new AutoSessionProvider();
|
|
1499
|
+
}
|
|
1500
|
+
/** onnxruntime-node provider. The native module is imported lazily on first session. */
|
|
1501
|
+
var NodeSessionProvider = class {
|
|
1502
|
+
kind = "node";
|
|
1503
|
+
ortPromise;
|
|
1504
|
+
async ort() {
|
|
1505
|
+
if (!this.ortPromise) this.ortPromise = import("onnxruntime-node").catch((err) => {
|
|
1506
|
+
this.ortPromise = void 0;
|
|
1507
|
+
throw new Error(`onnxruntime-node failed to load (${String(err)}). Install it in the environment or provide a custom SessionProvider.`);
|
|
1508
|
+
});
|
|
1509
|
+
return this.ortPromise;
|
|
1510
|
+
}
|
|
1511
|
+
async createSession(modelBytes) {
|
|
1512
|
+
const ort = await this.ort();
|
|
1513
|
+
const session = await ort.InferenceSession.create(modelBytes, {
|
|
1514
|
+
executionProviders: ["cpu"],
|
|
1515
|
+
graphOptimizationLevel: "all"
|
|
1516
|
+
});
|
|
1517
|
+
return {
|
|
1518
|
+
run: async (input) => {
|
|
1519
|
+
const feeds = { [session.inputNames.includes(input.name) ? input.name : session.inputNames[0] ?? input.name]: new ort.Tensor("float32", input.data, [...input.dims]) };
|
|
1520
|
+
const results = await session.run(feeds);
|
|
1521
|
+
const outputs = {};
|
|
1522
|
+
for (const [name, tensor] of Object.entries(results)) outputs[name] = {
|
|
1523
|
+
data: tensor.data,
|
|
1524
|
+
dims: [...tensor.dims]
|
|
1525
|
+
};
|
|
1526
|
+
return outputs;
|
|
1527
|
+
},
|
|
1528
|
+
release: () => {
|
|
1529
|
+
session.release();
|
|
1530
|
+
}
|
|
1531
|
+
};
|
|
1532
|
+
}
|
|
1533
|
+
};
|
|
1534
|
+
|
|
1535
|
+
//#endregion
|
|
1536
|
+
//#region src/runner/vision-runner.ts
|
|
1537
|
+
/**
|
|
1538
|
+
* Lazy vision runner.
|
|
1539
|
+
*
|
|
1540
|
+
* `create()` performs ZERO I/O: no downloads, no sessions. Each inference method downloads
|
|
1541
|
+
* its model and warms its session on FIRST USE ONLY, then reuses them for the process
|
|
1542
|
+
* lifetime. `detect()` and `segment()` share one RTMDet-Ins forward pass per frame.
|
|
1543
|
+
* Call `close()` to release sessions.
|
|
1544
|
+
*/
|
|
1545
|
+
var VisionRunner = class VisionRunner {
|
|
1546
|
+
variant;
|
|
1547
|
+
confidence;
|
|
1548
|
+
classes;
|
|
1549
|
+
maskThreshold;
|
|
1550
|
+
featherRadius;
|
|
1551
|
+
_store;
|
|
1552
|
+
_provider;
|
|
1553
|
+
_sessions = /* @__PURE__ */ new Map();
|
|
1554
|
+
_tensorScratch = /* @__PURE__ */ new Map();
|
|
1555
|
+
/** One RTMDet-Ins pass per frame buffer, shared by detect() and segment(). */
|
|
1556
|
+
_instancePass;
|
|
1557
|
+
constructor(options) {
|
|
1558
|
+
this.variant = options.variant ?? "s";
|
|
1559
|
+
this.confidence = options.confidence ?? .3;
|
|
1560
|
+
this.classes = options.classes;
|
|
1561
|
+
this.maskThreshold = options.maskThreshold ?? .5;
|
|
1562
|
+
this.featherRadius = options.featherRadius;
|
|
1563
|
+
this._store = options.store ?? new VisionModelStore({
|
|
1564
|
+
modelsDir: options.modelsDir,
|
|
1565
|
+
baseUrl: options.baseUrl,
|
|
1566
|
+
timeoutMs: options.timeoutMs,
|
|
1567
|
+
onProgress: options.onProgress
|
|
1568
|
+
});
|
|
1569
|
+
this._provider = options.provider ?? createDefaultSessionProvider();
|
|
1570
|
+
}
|
|
1571
|
+
/** Cheap: no downloads, no sessions, no side effects. */
|
|
1572
|
+
static create(options = {}) {
|
|
1573
|
+
return new VisionRunner(options);
|
|
1574
|
+
}
|
|
1575
|
+
get modelsDir() {
|
|
1576
|
+
return this._store.modelsDir;
|
|
1577
|
+
}
|
|
1578
|
+
/** Every registry model keyed to its current download state (downloaded → "ready"). */
|
|
1579
|
+
get downloadStatus() {
|
|
1580
|
+
const statuses = /* @__PURE__ */ new Map();
|
|
1581
|
+
for (const key of Object.keys(VISION_MODELS)) statuses.set(key, "pending");
|
|
1582
|
+
for (const [key, status] of this._store.statuses()) statuses.set(key, status);
|
|
1583
|
+
return statuses;
|
|
1584
|
+
}
|
|
1585
|
+
/** The model serving `task` (matte depends on frame aspect; defaults to square). */
|
|
1586
|
+
modelKeyFor(task, aspect = 1) {
|
|
1587
|
+
return modelKeyFor(task, this.variant, aspect);
|
|
1588
|
+
}
|
|
1589
|
+
/**
|
|
1590
|
+
* Explicit warm-up: downloads + sessions for the given tasks, ahead of first use.
|
|
1591
|
+
* `matte` warms both aspect variants (the frame aspect is unknown until inference).
|
|
1592
|
+
*/
|
|
1593
|
+
async preload(tasks = ["detect"]) {
|
|
1594
|
+
const keys = /* @__PURE__ */ new Set();
|
|
1595
|
+
for (const task of tasks) if (task === "matte") {
|
|
1596
|
+
keys.add("selfie-square");
|
|
1597
|
+
keys.add("selfie-landscape");
|
|
1598
|
+
} else keys.add(this.modelKeyFor(task));
|
|
1599
|
+
await this._store.preload([...keys]);
|
|
1600
|
+
await Promise.all([...keys].map((key) => this.session(key)));
|
|
1601
|
+
}
|
|
1602
|
+
/** COCO-80 boxes (sorted by score, source pixels). Masks are not decoded. */
|
|
1603
|
+
async detect(image) {
|
|
1604
|
+
const pass = await this.instancePass(image);
|
|
1605
|
+
return decodeRtmdetIns({
|
|
1606
|
+
...pass.outputs,
|
|
1607
|
+
masks: void 0
|
|
1608
|
+
}, this.instanceDecodeOptions(image, pass.transform)).detections;
|
|
1609
|
+
}
|
|
1610
|
+
/** COCO-80 boxes plus a frame-aligned soft mask per instance. */
|
|
1611
|
+
async segment(image) {
|
|
1612
|
+
const pass = await this.instancePass(image);
|
|
1613
|
+
return decodeRtmdetIns(pass.outputs, this.instanceDecodeOptions(image, pass.transform));
|
|
1614
|
+
}
|
|
1615
|
+
/** COCO-17 keypoints per person (source pixels), highest score first. */
|
|
1616
|
+
async pose(image) {
|
|
1617
|
+
const key = this.modelKeyFor("pose");
|
|
1618
|
+
const { outputs, transform } = await this.infer(key, image);
|
|
1619
|
+
const dets = output(outputs, "dets", key);
|
|
1620
|
+
const keypoints = output(outputs, "keypoints", key);
|
|
1621
|
+
return decodeRtmo({
|
|
1622
|
+
dets: dets.data,
|
|
1623
|
+
keypoints: keypoints.data,
|
|
1624
|
+
count: dets.dims[1] ?? 0,
|
|
1625
|
+
keypointStride: keypoints.dims[3] ?? 3
|
|
1626
|
+
}, {
|
|
1627
|
+
confidence: this.confidence,
|
|
1628
|
+
transform,
|
|
1629
|
+
sourceWidth: image.width,
|
|
1630
|
+
sourceHeight: image.height
|
|
1631
|
+
});
|
|
1632
|
+
}
|
|
1633
|
+
/** Person-vs-background alpha for the whole frame (Selfie Segmenter). */
|
|
1634
|
+
async matte(image) {
|
|
1635
|
+
const key = this.modelKeyFor("matte", image.width / image.height);
|
|
1636
|
+
const [w, h] = VISION_MODELS[key].input;
|
|
1637
|
+
const { outputs } = await this.infer(key, image);
|
|
1638
|
+
return decodeSelfie(output(outputs, "alphas", key).data, w, h, {
|
|
1639
|
+
sourceWidth: image.width,
|
|
1640
|
+
sourceHeight: image.height,
|
|
1641
|
+
maskThreshold: this.maskThreshold,
|
|
1642
|
+
...this.featherRadius !== void 0 ? { featherRadius: this.featherRadius } : {}
|
|
1643
|
+
});
|
|
1644
|
+
}
|
|
1645
|
+
close() {
|
|
1646
|
+
for (const promise of this._sessions.values()) promise.then((session) => session.release()).catch(() => void 0);
|
|
1647
|
+
this._sessions.clear();
|
|
1648
|
+
this._instancePass = void 0;
|
|
1649
|
+
this._store.clearMemoryCache();
|
|
1650
|
+
}
|
|
1651
|
+
instancePass(image) {
|
|
1652
|
+
if (this._instancePass?.frame === image.data) return this._instancePass.result;
|
|
1653
|
+
const key = this.modelKeyFor("segment");
|
|
1654
|
+
const result = this.infer(key, image).then(({ outputs, transform }) => {
|
|
1655
|
+
const dets = output(outputs, "dets", key);
|
|
1656
|
+
const masks = output(outputs, "masks", key);
|
|
1657
|
+
return {
|
|
1658
|
+
transform,
|
|
1659
|
+
outputs: {
|
|
1660
|
+
dets: dets.data,
|
|
1661
|
+
labels: output(outputs, "labels", key).data,
|
|
1662
|
+
masks: masks.data,
|
|
1663
|
+
count: dets.dims[1] ?? 0,
|
|
1664
|
+
maskHeight: masks.dims[masks.dims.length - 2] ?? 0,
|
|
1665
|
+
maskWidth: masks.dims[masks.dims.length - 1] ?? 0
|
|
1666
|
+
}
|
|
1667
|
+
};
|
|
1668
|
+
});
|
|
1669
|
+
this._instancePass = {
|
|
1670
|
+
frame: image.data,
|
|
1671
|
+
result
|
|
1672
|
+
};
|
|
1673
|
+
result.catch(() => {
|
|
1674
|
+
if (this._instancePass?.result === result) this._instancePass = void 0;
|
|
1675
|
+
});
|
|
1676
|
+
return result;
|
|
1677
|
+
}
|
|
1678
|
+
instanceDecodeOptions(image, transform) {
|
|
1679
|
+
return {
|
|
1680
|
+
confidence: this.confidence,
|
|
1681
|
+
classes: this.classes,
|
|
1682
|
+
classNames: COCO_CLASSES,
|
|
1683
|
+
transform,
|
|
1684
|
+
sourceWidth: image.width,
|
|
1685
|
+
sourceHeight: image.height,
|
|
1686
|
+
maskThreshold: this.maskThreshold,
|
|
1687
|
+
...this.featherRadius !== void 0 ? { featherRadius: this.featherRadius } : {}
|
|
1688
|
+
};
|
|
1689
|
+
}
|
|
1690
|
+
async infer(key, image) {
|
|
1691
|
+
const session = await this.session(key);
|
|
1692
|
+
const desc = VISION_MODELS[key];
|
|
1693
|
+
const spec = preprocessSpec(desc.family, desc.input);
|
|
1694
|
+
const size = 3 * spec.width * spec.height;
|
|
1695
|
+
let scratch = this._tensorScratch.get(key);
|
|
1696
|
+
if (!scratch || scratch.length < size) {
|
|
1697
|
+
scratch = new Float32Array(size);
|
|
1698
|
+
this._tensorScratch.set(key, scratch);
|
|
1699
|
+
}
|
|
1700
|
+
const { tensor, transform } = imageToTensor(image, spec, scratch);
|
|
1701
|
+
return {
|
|
1702
|
+
outputs: await session.run({
|
|
1703
|
+
name: "input",
|
|
1704
|
+
data: tensor,
|
|
1705
|
+
dims: [
|
|
1706
|
+
1,
|
|
1707
|
+
3,
|
|
1708
|
+
spec.height,
|
|
1709
|
+
spec.width
|
|
1710
|
+
]
|
|
1711
|
+
}),
|
|
1712
|
+
transform
|
|
1713
|
+
};
|
|
1714
|
+
}
|
|
1715
|
+
session(key) {
|
|
1716
|
+
let promise = this._sessions.get(key);
|
|
1717
|
+
if (!promise) {
|
|
1718
|
+
promise = this._store.ensure(key).then((bytes) => this._provider.createSession(bytes)).catch((err) => {
|
|
1719
|
+
this._sessions.delete(key);
|
|
1720
|
+
this._store.evict(key);
|
|
1721
|
+
throw err;
|
|
1722
|
+
});
|
|
1723
|
+
this._sessions.set(key, promise);
|
|
1724
|
+
}
|
|
1725
|
+
return promise;
|
|
1726
|
+
}
|
|
1727
|
+
};
|
|
1728
|
+
function output(outputs, name, key) {
|
|
1729
|
+
const tensor = outputs[name];
|
|
1730
|
+
if (!tensor) throw new Error(`Vision model '${key}' returned no '${name}' output (got: ${Object.keys(outputs).join(", ") || "none"}). Expected the ${VISION_MODELS[key].source}.`);
|
|
1731
|
+
return tensor;
|
|
1732
|
+
}
|
|
1733
|
+
|
|
1734
|
+
//#endregion
|
|
1735
|
+
//#region src/segmentation/subject.ts
|
|
1736
|
+
/**
|
|
1737
|
+
* The frame's primary instance: the largest person when present, else the most confident
|
|
1738
|
+
* instance (lowest `detectionIndex` — detections are score-sorted). Size alone is a poor
|
|
1739
|
+
* signal without a person: the largest instance is usually a backdrop ("dining table").
|
|
1740
|
+
*/
|
|
1741
|
+
function selectSubjectMask(masks) {
|
|
1742
|
+
let best;
|
|
1743
|
+
for (const m of masks) {
|
|
1744
|
+
if (!best) {
|
|
1745
|
+
best = m;
|
|
1746
|
+
continue;
|
|
1747
|
+
}
|
|
1748
|
+
const mIsPerson = isPerson(m);
|
|
1749
|
+
if (mIsPerson !== isPerson(best)) {
|
|
1750
|
+
if (mIsPerson) best = m;
|
|
1751
|
+
} else if (mIsPerson ? m.area > best.area : m.detectionIndex < best.detectionIndex) best = m;
|
|
1752
|
+
}
|
|
1753
|
+
return best;
|
|
1754
|
+
}
|
|
1755
|
+
function isPerson(mask) {
|
|
1756
|
+
return mask.category.toLowerCase() === "person";
|
|
1757
|
+
}
|
|
1758
|
+
/** Tight pixel bounds of a mask's non-zero alpha, or null when empty. */
|
|
1759
|
+
function maskBounds(mask) {
|
|
1760
|
+
const { mask: data, width, height } = mask;
|
|
1761
|
+
let x0 = width;
|
|
1762
|
+
let y0 = height;
|
|
1763
|
+
let x1 = -1;
|
|
1764
|
+
let y1 = -1;
|
|
1765
|
+
for (let y = 0; y < height; y++) {
|
|
1766
|
+
const row = y * width;
|
|
1767
|
+
for (let x = 0; x < width; x++) {
|
|
1768
|
+
if (data[row + x] === 0) continue;
|
|
1769
|
+
if (x < x0) x0 = x;
|
|
1770
|
+
if (x > x1) x1 = x;
|
|
1771
|
+
if (y < y0) y0 = y;
|
|
1772
|
+
if (y > y1) y1 = y;
|
|
1773
|
+
}
|
|
1774
|
+
}
|
|
1775
|
+
return x1 < 0 ? null : {
|
|
1776
|
+
x0,
|
|
1777
|
+
y0,
|
|
1778
|
+
x1,
|
|
1779
|
+
y1
|
|
1780
|
+
};
|
|
1781
|
+
}
|
|
1782
|
+
/**
|
|
1783
|
+
* Builds the full subject silhouette: the primary instance plus every comparably sized
|
|
1784
|
+
* instance whose bounds sit mostly inside or across it (`overlap` = intersection / smaller
|
|
1785
|
+
* box area; `maxGrowth` caps a part's box area relative to the subject's).
|
|
1786
|
+
*
|
|
1787
|
+
* COCO has no "clothing" class, so a flowing dress, a held guitar or a ridden bike comes back
|
|
1788
|
+
* as its own instance (often mislabeled) — merging them keeps the whole figure in the cutout.
|
|
1789
|
+
* The size cap keeps containers out: a small figure inside a tunnel or window detected as a
|
|
1790
|
+
* huge "clock" must not drag the whole frame into the subject.
|
|
1791
|
+
*/
|
|
1792
|
+
function mergeSubjectMask(masks, overlap = .5, maxGrowth = 2) {
|
|
1793
|
+
const subject = selectSubjectMask(masks);
|
|
1794
|
+
if (!subject) return void 0;
|
|
1795
|
+
const sb = maskBounds(subject);
|
|
1796
|
+
if (!sb) return subject;
|
|
1797
|
+
const subjectArea = boundsArea(sb);
|
|
1798
|
+
const parts = masks.filter((m) => {
|
|
1799
|
+
if (m === subject) return false;
|
|
1800
|
+
const b = maskBounds(m);
|
|
1801
|
+
return b !== null && boundsArea(b) <= subjectArea * maxGrowth && boundsOverlap(sb, b) >= overlap;
|
|
1802
|
+
});
|
|
1803
|
+
if (parts.length === 0) return subject;
|
|
1804
|
+
const merged = subject.mask.slice();
|
|
1805
|
+
for (const part of parts) {
|
|
1806
|
+
const data = part.mask;
|
|
1807
|
+
for (let i = 0; i < merged.length; i++) if (data[i] > merged[i]) merged[i] = data[i];
|
|
1808
|
+
}
|
|
1809
|
+
let area = 0;
|
|
1810
|
+
for (let i = 0; i < merged.length; i++) area += merged[i];
|
|
1811
|
+
area = Math.round(area / 255);
|
|
1812
|
+
return {
|
|
1813
|
+
...subject,
|
|
1814
|
+
mask: merged,
|
|
1815
|
+
area,
|
|
1816
|
+
coverage: area / (subject.width * subject.height)
|
|
1817
|
+
};
|
|
1818
|
+
}
|
|
1819
|
+
function boundsArea(b) {
|
|
1820
|
+
return (b.x1 - b.x0 + 1) * (b.y1 - b.y0 + 1);
|
|
1821
|
+
}
|
|
1822
|
+
function boundsOverlap(a, b) {
|
|
1823
|
+
const iw = Math.min(a.x1, b.x1) - Math.max(a.x0, b.x0) + 1;
|
|
1824
|
+
const ih = Math.min(a.y1, b.y1) - Math.max(a.y0, b.y0) + 1;
|
|
1825
|
+
if (iw <= 0 || ih <= 0) return 0;
|
|
1826
|
+
return iw * ih / Math.min(boundsArea(a), boundsArea(b));
|
|
1827
|
+
}
|
|
1828
|
+
|
|
1829
|
+
//#endregion
|
|
1830
|
+
//#region src/spatial/camera-space-transformer.ts
|
|
1831
|
+
var SpatialLandmarkTransformer = class {
|
|
1832
|
+
width;
|
|
1833
|
+
height;
|
|
1834
|
+
fovRad;
|
|
1835
|
+
focalDistance;
|
|
1836
|
+
constructor(camera) {
|
|
1837
|
+
this.width = camera.width;
|
|
1838
|
+
this.height = camera.height;
|
|
1839
|
+
this.fovRad = (camera.fov ?? 60) * Math.PI / 180;
|
|
1840
|
+
this.focalDistance = camera.cameraDistance ?? this.height / 2 / Math.tan(this.fovRad / 2);
|
|
1841
|
+
}
|
|
1842
|
+
/**
|
|
1843
|
+
* Projects a normalized landmark [0, 1] into 2D/3D canvas coordinates.
|
|
1844
|
+
*/
|
|
1845
|
+
projectNormalizedLandmark(landmark, offsetZ = 0) {
|
|
1846
|
+
const screenX = landmark.x * this.width;
|
|
1847
|
+
const screenY = landmark.y * this.height;
|
|
1848
|
+
const rawZ = landmark.z * this.width + offsetZ;
|
|
1849
|
+
const effectiveZ = Math.min(rawZ, this.focalDistance - 10);
|
|
1850
|
+
const scale = this.focalDistance / (this.focalDistance - effectiveZ);
|
|
1851
|
+
return {
|
|
1852
|
+
x: screenX,
|
|
1853
|
+
y: screenY,
|
|
1854
|
+
z: rawZ,
|
|
1855
|
+
scale: Math.max(.1, scale)
|
|
1856
|
+
};
|
|
1857
|
+
}
|
|
1858
|
+
/**
|
|
1859
|
+
* Projects a metric world landmark [X_w, Y_w, Z_w] in meters into canvas space.
|
|
1860
|
+
*/
|
|
1861
|
+
projectWorldLandmark(worldLandmark, subjectDistanceMeters = 2, scaleMetersToPixels = 800) {
|
|
1862
|
+
const centerX = this.width / 2;
|
|
1863
|
+
const centerY = this.height / 2;
|
|
1864
|
+
const worldX = worldLandmark.x * scaleMetersToPixels;
|
|
1865
|
+
const worldY = worldLandmark.y * scaleMetersToPixels;
|
|
1866
|
+
const worldZ = (worldLandmark.z + subjectDistanceMeters) * scaleMetersToPixels;
|
|
1867
|
+
const effectiveZ = Math.min(worldZ, this.focalDistance - 10);
|
|
1868
|
+
const scale = this.focalDistance / (this.focalDistance - effectiveZ);
|
|
1869
|
+
return {
|
|
1870
|
+
x: centerX + worldX * scale,
|
|
1871
|
+
y: centerY + worldY * scale,
|
|
1872
|
+
z: worldZ,
|
|
1873
|
+
scale: Math.max(.1, scale)
|
|
1874
|
+
};
|
|
1875
|
+
}
|
|
1876
|
+
};
|
|
1877
|
+
|
|
1878
|
+
//#endregion
|
|
1879
|
+
//#region src/tracking/pose-track-matcher.ts
|
|
1880
|
+
function getPersonBoundingBox(person) {
|
|
1881
|
+
if (person.boundingBox && person.boundingBox.width > 0 && person.boundingBox.height > 0) return {
|
|
1882
|
+
originX: person.boundingBox.originX,
|
|
1883
|
+
originY: person.boundingBox.originY,
|
|
1884
|
+
width: person.boundingBox.width,
|
|
1885
|
+
height: person.boundingBox.height
|
|
1886
|
+
};
|
|
1887
|
+
let minX = Infinity;
|
|
1888
|
+
let minY = Infinity;
|
|
1889
|
+
let maxX = -Infinity;
|
|
1890
|
+
let maxY = -Infinity;
|
|
1891
|
+
let validCount = 0;
|
|
1892
|
+
for (const kp of person.keypoints) if (kp.visibility > .1) {
|
|
1893
|
+
validCount++;
|
|
1894
|
+
if (kp.x < minX) minX = kp.x;
|
|
1895
|
+
if (kp.y < minY) minY = kp.y;
|
|
1896
|
+
if (kp.x > maxX) maxX = kp.x;
|
|
1897
|
+
if (kp.y > maxY) maxY = kp.y;
|
|
1898
|
+
}
|
|
1899
|
+
if (validCount === 0 || minX >= maxX || minY >= maxY) return {
|
|
1900
|
+
originX: person.boundingBox?.originX ?? 0,
|
|
1901
|
+
originY: person.boundingBox?.originY ?? 0,
|
|
1902
|
+
width: person.boundingBox?.width ?? 0,
|
|
1903
|
+
height: person.boundingBox?.height ?? 0
|
|
1904
|
+
};
|
|
1905
|
+
return {
|
|
1906
|
+
originX: minX,
|
|
1907
|
+
originY: minY,
|
|
1908
|
+
width: maxX - minX,
|
|
1909
|
+
height: maxY - minY
|
|
1910
|
+
};
|
|
1911
|
+
}
|
|
1912
|
+
/**
|
|
1913
|
+
* Matches detected pose people with active person tracks using greedy bipartite IoU.
|
|
1914
|
+
*
|
|
1915
|
+
* Annotates each person with the matched trackId.
|
|
1916
|
+
*/
|
|
1917
|
+
function matchPoseToTracks(people, trackedObjects, minIoU = .15) {
|
|
1918
|
+
if (people.length === 0) return [];
|
|
1919
|
+
const personTracks = trackedObjects.filter((o) => o.category.toLowerCase() === "person" && o.active);
|
|
1920
|
+
if (personTracks.length === 0) return [...people];
|
|
1921
|
+
const candidateMatches = [];
|
|
1922
|
+
for (let p = 0; p < people.length; p++) {
|
|
1923
|
+
const person = people[p];
|
|
1924
|
+
if (person.trackId !== void 0) continue;
|
|
1925
|
+
const pBox = getPersonBoundingBox(person);
|
|
1926
|
+
const isNormalized = pBox.width <= 1.05 && pBox.height <= 1.05;
|
|
1927
|
+
for (const trk of personTracks) {
|
|
1928
|
+
const iou = computeIoU(pBox, isNormalized && trk.boundingBox.normalizedWidth > 0 ? {
|
|
1929
|
+
originX: trk.boundingBox.normalizedX,
|
|
1930
|
+
originY: trk.boundingBox.normalizedY,
|
|
1931
|
+
width: trk.boundingBox.normalizedWidth,
|
|
1932
|
+
height: trk.boundingBox.normalizedHeight
|
|
1933
|
+
} : trk.boundingBox);
|
|
1934
|
+
if (iou >= minIoU) candidateMatches.push({
|
|
1935
|
+
personIdx: p,
|
|
1936
|
+
trackId: trk.trackId,
|
|
1937
|
+
iou
|
|
1938
|
+
});
|
|
1939
|
+
}
|
|
1940
|
+
}
|
|
1941
|
+
candidateMatches.sort((a, b) => b.iou - a.iou);
|
|
1942
|
+
const assignedPeople = /* @__PURE__ */ new Set();
|
|
1943
|
+
const assignedTracks = /* @__PURE__ */ new Set();
|
|
1944
|
+
const personTrackMap = /* @__PURE__ */ new Map();
|
|
1945
|
+
for (const match of candidateMatches) {
|
|
1946
|
+
if (assignedPeople.has(match.personIdx) || assignedTracks.has(match.trackId)) continue;
|
|
1947
|
+
assignedPeople.add(match.personIdx);
|
|
1948
|
+
assignedTracks.add(match.trackId);
|
|
1949
|
+
personTrackMap.set(match.personIdx, match.trackId);
|
|
1950
|
+
}
|
|
1951
|
+
if (people.length === 1 && personTracks.length === 1 && !personTrackMap.has(0) && people[0].trackId === void 0) personTrackMap.set(0, personTracks[0].trackId);
|
|
1952
|
+
return people.map((person, idx) => {
|
|
1953
|
+
if (person.trackId !== void 0) return person;
|
|
1954
|
+
const trackId = personTrackMap.get(idx);
|
|
1955
|
+
return trackId !== void 0 ? {
|
|
1956
|
+
...person,
|
|
1957
|
+
trackId
|
|
1958
|
+
} : person;
|
|
1959
|
+
});
|
|
1960
|
+
}
|
|
1961
|
+
|
|
1962
|
+
//#endregion
|
|
1963
|
+
//#region src/signals/vision-bundle.ts
|
|
1964
|
+
var VisionBundle = class {
|
|
1965
|
+
objects;
|
|
1966
|
+
poseLandmarks;
|
|
1967
|
+
masks;
|
|
1968
|
+
segmentation;
|
|
1969
|
+
classes;
|
|
1970
|
+
poseLandmarksTensor;
|
|
1971
|
+
objectsTensor;
|
|
1972
|
+
masksTensor;
|
|
1973
|
+
/** Per-class detection histogram [nc] — top-level convenience mirror of `classes.histogram`. */
|
|
1974
|
+
histogramTensor;
|
|
1975
|
+
_totalFrames;
|
|
1976
|
+
_fps;
|
|
1977
|
+
_width;
|
|
1978
|
+
_height;
|
|
1979
|
+
_transformer;
|
|
1980
|
+
_stencilTexture;
|
|
1981
|
+
_poseCache = /* @__PURE__ */ new Map();
|
|
1982
|
+
_objectCache = /* @__PURE__ */ new Map();
|
|
1983
|
+
_maskCache = /* @__PURE__ */ new Map();
|
|
1984
|
+
_matteCache = /* @__PURE__ */ new Map();
|
|
1985
|
+
_poseLmCache = /* @__PURE__ */ new Map();
|
|
1986
|
+
_trackCache = /* @__PURE__ */ new Map();
|
|
1987
|
+
_categoryCache = /* @__PURE__ */ new Map();
|
|
1988
|
+
_classCache = /* @__PURE__ */ new Map();
|
|
1989
|
+
_maskTrackCache = /* @__PURE__ */ new Map();
|
|
1990
|
+
_registeredSignals = /* @__PURE__ */ new Set();
|
|
1991
|
+
constructor(options = {}) {
|
|
1992
|
+
this._totalFrames = options.totalFrames ?? 120;
|
|
1993
|
+
this._fps = options.fps ?? 24;
|
|
1994
|
+
this._width = options.width ?? 1920;
|
|
1995
|
+
this._height = options.height ?? 1080;
|
|
1996
|
+
this._transformer = new SpatialLandmarkTransformer({
|
|
1997
|
+
width: this._width,
|
|
1998
|
+
height: this._height,
|
|
1999
|
+
fov: options.cameraFov ?? 60
|
|
2000
|
+
});
|
|
2001
|
+
const self = this;
|
|
2002
|
+
const programmaticSignal$1 = (evaluator, sigOptions) => {
|
|
2003
|
+
const sig = programmaticSignal(evaluator, sigOptions);
|
|
2004
|
+
this._registeredSignals.add(sig);
|
|
2005
|
+
return sig;
|
|
2006
|
+
};
|
|
2007
|
+
const getPrimaryLandmarks = (frame) => {
|
|
2008
|
+
return this.getPoseResult(frame).people[0]?.keypoints ?? createNeutralLandmarks();
|
|
2009
|
+
};
|
|
2010
|
+
const createPoseCoord = (index, label) => {
|
|
2011
|
+
const getLm = (frame) => getPrimaryLandmarks(frame)[index] ?? {
|
|
2012
|
+
x: .5,
|
|
2013
|
+
y: .5,
|
|
2014
|
+
z: 0
|
|
2015
|
+
};
|
|
2016
|
+
return {
|
|
2017
|
+
x: programmaticSignal$1((ctx) => getLm(ctx.frame).x, {
|
|
2018
|
+
fps: this._fps,
|
|
2019
|
+
label: `${label}_x`
|
|
2020
|
+
}),
|
|
2021
|
+
y: programmaticSignal$1((ctx) => getLm(ctx.frame).y, {
|
|
2022
|
+
fps: this._fps,
|
|
2023
|
+
label: `${label}_y`
|
|
2024
|
+
}),
|
|
2025
|
+
z: programmaticSignal$1((ctx) => getLm(ctx.frame).z ?? 0, {
|
|
2026
|
+
fps: this._fps,
|
|
2027
|
+
label: `${label}_z`
|
|
2028
|
+
}),
|
|
2029
|
+
screenX: programmaticSignal$1((ctx) => this._transformer.projectNormalizedLandmark(getLm(ctx.frame)).x, {
|
|
2030
|
+
fps: this._fps,
|
|
2031
|
+
label: `${label}_screenX`
|
|
2032
|
+
}),
|
|
2033
|
+
screenY: programmaticSignal$1((ctx) => this._transformer.projectNormalizedLandmark(getLm(ctx.frame)).y, {
|
|
2034
|
+
fps: this._fps,
|
|
2035
|
+
label: `${label}_screenY`
|
|
2036
|
+
})
|
|
2037
|
+
};
|
|
2038
|
+
};
|
|
2039
|
+
const getOrCreatePoseLm = (index) => {
|
|
2040
|
+
let sig = this._poseLmCache.get(index);
|
|
2041
|
+
if (!sig) {
|
|
2042
|
+
sig = createPoseCoord(index, COCO17_KEYPOINT_NAMES[index] ?? `kpt_${index}`);
|
|
2043
|
+
this._poseLmCache.set(index, sig);
|
|
2044
|
+
}
|
|
2045
|
+
return sig;
|
|
2046
|
+
};
|
|
2047
|
+
const poseNamed = {
|
|
2048
|
+
nose: getOrCreatePoseLm(COCO17_KEYPOINTS.NOSE),
|
|
2049
|
+
leftEye: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_EYE),
|
|
2050
|
+
rightEye: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_EYE),
|
|
2051
|
+
leftEar: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_EAR),
|
|
2052
|
+
rightEar: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_EAR),
|
|
2053
|
+
shoulder: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_SHOULDER),
|
|
2054
|
+
leftShoulder: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_SHOULDER),
|
|
2055
|
+
rightShoulder: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_SHOULDER),
|
|
2056
|
+
leftElbow: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_ELBOW),
|
|
2057
|
+
rightElbow: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_ELBOW),
|
|
2058
|
+
leftWrist: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_WRIST),
|
|
2059
|
+
rightWrist: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_WRIST),
|
|
2060
|
+
leftHip: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_HIP),
|
|
2061
|
+
rightHip: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_HIP),
|
|
2062
|
+
leftKnee: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_KNEE),
|
|
2063
|
+
rightKnee: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_KNEE),
|
|
2064
|
+
leftAnkle: getOrCreatePoseLm(COCO17_KEYPOINTS.LEFT_ANKLE),
|
|
2065
|
+
rightAnkle: getOrCreatePoseLm(COCO17_KEYPOINTS.RIGHT_ANKLE),
|
|
2066
|
+
get: getOrCreatePoseLm
|
|
2067
|
+
};
|
|
2068
|
+
this.poseLandmarks = new Proxy(poseNamed, { get(target, prop, receiver) {
|
|
2069
|
+
if (typeof prop === "string" && /^\d+$/.test(prop)) return target.get(Number(prop));
|
|
2070
|
+
if (typeof prop === "number") return target.get(prop);
|
|
2071
|
+
return Reflect.get(target, prop, receiver);
|
|
2072
|
+
} });
|
|
2073
|
+
const neutralTrack = (trackId, category = "unknown") => ({
|
|
2074
|
+
trackId,
|
|
2075
|
+
category,
|
|
2076
|
+
score: 0,
|
|
2077
|
+
boundingBox: {
|
|
2078
|
+
originX: this._width / 2,
|
|
2079
|
+
originY: this._height / 2,
|
|
2080
|
+
width: 0,
|
|
2081
|
+
height: 0,
|
|
2082
|
+
normalizedX: .5,
|
|
2083
|
+
normalizedY: .5,
|
|
2084
|
+
normalizedWidth: 0,
|
|
2085
|
+
normalizedHeight: 0
|
|
2086
|
+
},
|
|
2087
|
+
centerX: this._width / 2,
|
|
2088
|
+
centerY: this._height / 2,
|
|
2089
|
+
normalizedCenterX: .5,
|
|
2090
|
+
normalizedCenterY: .5,
|
|
2091
|
+
velocity: {
|
|
2092
|
+
vx: 0,
|
|
2093
|
+
vy: 0
|
|
2094
|
+
},
|
|
2095
|
+
speed: 0,
|
|
2096
|
+
age: 0,
|
|
2097
|
+
hits: 0,
|
|
2098
|
+
active: false,
|
|
2099
|
+
isCoasting: false
|
|
2100
|
+
});
|
|
2101
|
+
const createAnchor = (getPos, label) => {
|
|
2102
|
+
const screenX = programmaticSignal$1((ctx) => getPos(ctx.frame).sx, {
|
|
2103
|
+
fps: this._fps,
|
|
2104
|
+
label: `${label}_screenX`
|
|
2105
|
+
});
|
|
2106
|
+
const screenY = programmaticSignal$1((ctx) => getPos(ctx.frame).sy, {
|
|
2107
|
+
fps: this._fps,
|
|
2108
|
+
label: `${label}_screenY`
|
|
2109
|
+
});
|
|
2110
|
+
return {
|
|
2111
|
+
x: programmaticSignal$1((ctx) => this._width > 0 ? getPos(ctx.frame).sx / this._width : 0, {
|
|
2112
|
+
fps: this._fps,
|
|
2113
|
+
label: `${label}_x`
|
|
2114
|
+
}),
|
|
2115
|
+
y: programmaticSignal$1((ctx) => this._height > 0 ? getPos(ctx.frame).sy / this._height : 0, {
|
|
2116
|
+
fps: this._fps,
|
|
2117
|
+
label: `${label}_y`
|
|
2118
|
+
}),
|
|
2119
|
+
z: programmaticSignal$1(() => 0, {
|
|
2120
|
+
fps: this._fps,
|
|
2121
|
+
label: `${label}_z`
|
|
2122
|
+
}),
|
|
2123
|
+
screenX,
|
|
2124
|
+
screenY
|
|
2125
|
+
};
|
|
2126
|
+
};
|
|
2127
|
+
const buildTrackPoseSignals = (resolveTrack, label, fixedId) => {
|
|
2128
|
+
const resolvePerson = (frame) => {
|
|
2129
|
+
const poseRes = self.getPoseResult(frame);
|
|
2130
|
+
if (poseRes.people.length === 0) return void 0;
|
|
2131
|
+
const trk = resolveTrack(frame);
|
|
2132
|
+
const targetId = fixedId ?? trk.trackId;
|
|
2133
|
+
const byId = poseRes.people.find((p) => p.trackId === targetId);
|
|
2134
|
+
if (byId) return byId;
|
|
2135
|
+
if (!trk.active && poseRes.people.length > 1) return void 0;
|
|
2136
|
+
if (poseRes.people.length === 1) return poseRes.people[0];
|
|
2137
|
+
let bestPerson;
|
|
2138
|
+
let bestIoU = .1;
|
|
2139
|
+
for (const p of poseRes.people) {
|
|
2140
|
+
const pBox = getPersonBoundingBox(p);
|
|
2141
|
+
const iou = computeIoU(pBox, pBox.width <= 1.05 && pBox.height <= 1.05 && trk.boundingBox.normalizedWidth > 0 ? {
|
|
2142
|
+
originX: trk.boundingBox.normalizedX,
|
|
2143
|
+
originY: trk.boundingBox.normalizedY,
|
|
2144
|
+
width: trk.boundingBox.normalizedWidth,
|
|
2145
|
+
height: trk.boundingBox.normalizedHeight
|
|
2146
|
+
} : trk.boundingBox);
|
|
2147
|
+
if (iou > bestIoU) {
|
|
2148
|
+
bestIoU = iou;
|
|
2149
|
+
bestPerson = p;
|
|
2150
|
+
}
|
|
2151
|
+
}
|
|
2152
|
+
return bestPerson;
|
|
2153
|
+
};
|
|
2154
|
+
const hasPose = programmaticSignal$1((ctx) => resolvePerson(ctx.frame) ? 1 : 0, {
|
|
2155
|
+
fps: this._fps,
|
|
2156
|
+
label: `${label}_hasPose`
|
|
2157
|
+
});
|
|
2158
|
+
const createCoord = (index, kptName) => {
|
|
2159
|
+
const getLm = (frame) => {
|
|
2160
|
+
const kp = resolvePerson(frame)?.keypoints[index];
|
|
2161
|
+
if (kp && kp.visibility > .05) return {
|
|
2162
|
+
x: kp.x,
|
|
2163
|
+
y: kp.y,
|
|
2164
|
+
z: 0
|
|
2165
|
+
};
|
|
2166
|
+
const trk = resolveTrack(frame);
|
|
2167
|
+
return {
|
|
2168
|
+
x: trk.normalizedCenterX,
|
|
2169
|
+
y: trk.normalizedCenterY,
|
|
2170
|
+
z: 0
|
|
2171
|
+
};
|
|
2172
|
+
};
|
|
2173
|
+
return {
|
|
2174
|
+
x: programmaticSignal$1((ctx) => getLm(ctx.frame).x, {
|
|
2175
|
+
fps: this._fps,
|
|
2176
|
+
label: `${label}_${kptName}_x`
|
|
2177
|
+
}),
|
|
2178
|
+
y: programmaticSignal$1((ctx) => getLm(ctx.frame).y, {
|
|
2179
|
+
fps: this._fps,
|
|
2180
|
+
label: `${label}_${kptName}_y`
|
|
2181
|
+
}),
|
|
2182
|
+
z: programmaticSignal$1((ctx) => getLm(ctx.frame).z ?? 0, {
|
|
2183
|
+
fps: this._fps,
|
|
2184
|
+
label: `${label}_${kptName}_z`
|
|
2185
|
+
}),
|
|
2186
|
+
screenX: programmaticSignal$1((ctx) => self._transformer.projectNormalizedLandmark(getLm(ctx.frame)).x, {
|
|
2187
|
+
fps: this._fps,
|
|
2188
|
+
label: `${label}_${kptName}_screenX`
|
|
2189
|
+
}),
|
|
2190
|
+
screenY: programmaticSignal$1((ctx) => self._transformer.projectNormalizedLandmark(getLm(ctx.frame)).y, {
|
|
2191
|
+
fps: this._fps,
|
|
2192
|
+
label: `${label}_${kptName}_screenY`
|
|
2193
|
+
})
|
|
2194
|
+
};
|
|
2195
|
+
};
|
|
2196
|
+
const coordsCache = /* @__PURE__ */ new Map();
|
|
2197
|
+
const getOrCreateCoord = (idx) => {
|
|
2198
|
+
let c = coordsCache.get(idx);
|
|
2199
|
+
if (!c) {
|
|
2200
|
+
c = createCoord(idx, COCO17_KEYPOINT_NAMES[idx] ?? `kpt_${idx}`);
|
|
2201
|
+
coordsCache.set(idx, c);
|
|
2202
|
+
}
|
|
2203
|
+
return c;
|
|
2204
|
+
};
|
|
2205
|
+
const namedPose = {
|
|
2206
|
+
hasPose,
|
|
2207
|
+
wristSpeed: programmaticSignal$1((ctx) => {
|
|
2208
|
+
const f = ctx.frame;
|
|
2209
|
+
const curr = resolvePerson(f);
|
|
2210
|
+
if (!curr) return 0;
|
|
2211
|
+
const prev = resolvePerson(Math.max(0, f - 1));
|
|
2212
|
+
const currKp = curr.keypoints;
|
|
2213
|
+
const prevKp = prev?.keypoints;
|
|
2214
|
+
const rwCurr = currKp[COCO17_KEYPOINTS.RIGHT_WRIST];
|
|
2215
|
+
const rwPrev = prevKp?.[COCO17_KEYPOINTS.RIGHT_WRIST];
|
|
2216
|
+
let rwSpeed = 0;
|
|
2217
|
+
if (rwCurr && rwPrev && rwCurr.visibility > .1 && rwPrev.visibility > .1) {
|
|
2218
|
+
const dx = (rwCurr.x - rwPrev.x) * this._width;
|
|
2219
|
+
const dy = (rwCurr.y - rwPrev.y) * this._height;
|
|
2220
|
+
rwSpeed = Math.sqrt(dx * dx + dy * dy) * this._fps;
|
|
2221
|
+
}
|
|
2222
|
+
const lwCurr = currKp[COCO17_KEYPOINTS.LEFT_WRIST];
|
|
2223
|
+
const lwPrev = prevKp?.[COCO17_KEYPOINTS.LEFT_WRIST];
|
|
2224
|
+
let lwSpeed = 0;
|
|
2225
|
+
if (lwCurr && lwPrev && lwCurr.visibility > .1 && lwPrev.visibility > .1) {
|
|
2226
|
+
const dx = (lwCurr.x - lwPrev.x) * this._width;
|
|
2227
|
+
const dy = (lwCurr.y - lwPrev.y) * this._height;
|
|
2228
|
+
lwSpeed = Math.sqrt(dx * dx + dy * dy) * this._fps;
|
|
2229
|
+
}
|
|
2230
|
+
return Math.max(rwSpeed, lwSpeed);
|
|
2231
|
+
}, {
|
|
2232
|
+
fps: this._fps,
|
|
2233
|
+
label: `${label}_wristSpeed`
|
|
2234
|
+
}),
|
|
2235
|
+
handRaised: programmaticSignal$1((ctx) => {
|
|
2236
|
+
const p = resolvePerson(ctx.frame);
|
|
2237
|
+
if (!p) return 0;
|
|
2238
|
+
const k = p.keypoints;
|
|
2239
|
+
const lw = k[COCO17_KEYPOINTS.LEFT_WRIST];
|
|
2240
|
+
const ls = k[COCO17_KEYPOINTS.LEFT_SHOULDER];
|
|
2241
|
+
const rw = k[COCO17_KEYPOINTS.RIGHT_WRIST];
|
|
2242
|
+
const rs = k[COCO17_KEYPOINTS.RIGHT_SHOULDER];
|
|
2243
|
+
const lRaised = lw && ls && lw.visibility > .1 && ls.visibility > .1 && lw.y < ls.y;
|
|
2244
|
+
const rRaised = rw && rs && rw.visibility > .1 && rs.visibility > .1 && rw.y < rs.y;
|
|
2245
|
+
return lRaised || rRaised ? 1 : 0;
|
|
2246
|
+
}, {
|
|
2247
|
+
fps: this._fps,
|
|
2248
|
+
label: `${label}_handRaised`
|
|
2249
|
+
}),
|
|
2250
|
+
bodyTiltAngle: programmaticSignal$1((ctx) => {
|
|
2251
|
+
const p = resolvePerson(ctx.frame);
|
|
2252
|
+
if (!p) return 0;
|
|
2253
|
+
const k = p.keypoints;
|
|
2254
|
+
const ls = k[COCO17_KEYPOINTS.LEFT_SHOULDER];
|
|
2255
|
+
const rs = k[COCO17_KEYPOINTS.RIGHT_SHOULDER];
|
|
2256
|
+
const lh = k[COCO17_KEYPOINTS.LEFT_HIP];
|
|
2257
|
+
const rh = k[COCO17_KEYPOINTS.RIGHT_HIP];
|
|
2258
|
+
if (!ls || !rs || !lh || !rh) return 0;
|
|
2259
|
+
const sx = (ls.x + rs.x) / 2 * this._width;
|
|
2260
|
+
const sy = (ls.y + rs.y) / 2 * this._height;
|
|
2261
|
+
const hx = (lh.x + rh.x) / 2 * this._width;
|
|
2262
|
+
const hy = (lh.y + rh.y) / 2 * this._height;
|
|
2263
|
+
const dx = sx - hx;
|
|
2264
|
+
const dy = sy - hy;
|
|
2265
|
+
if (dx === 0 && dy === 0) return 0;
|
|
2266
|
+
return Math.atan2(dx, -dy);
|
|
2267
|
+
}, {
|
|
2268
|
+
fps: this._fps,
|
|
2269
|
+
label: `${label}_bodyTiltAngle`
|
|
2270
|
+
}),
|
|
2271
|
+
nose: getOrCreateCoord(COCO17_KEYPOINTS.NOSE),
|
|
2272
|
+
leftEye: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_EYE),
|
|
2273
|
+
rightEye: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_EYE),
|
|
2274
|
+
leftEar: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_EAR),
|
|
2275
|
+
rightEar: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_EAR),
|
|
2276
|
+
shoulder: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_SHOULDER),
|
|
2277
|
+
leftShoulder: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_SHOULDER),
|
|
2278
|
+
rightShoulder: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_SHOULDER),
|
|
2279
|
+
leftElbow: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_ELBOW),
|
|
2280
|
+
rightElbow: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_ELBOW),
|
|
2281
|
+
leftWrist: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_WRIST),
|
|
2282
|
+
rightWrist: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_WRIST),
|
|
2283
|
+
leftHip: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_HIP),
|
|
2284
|
+
rightHip: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_HIP),
|
|
2285
|
+
leftKnee: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_KNEE),
|
|
2286
|
+
rightKnee: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_KNEE),
|
|
2287
|
+
leftAnkle: getOrCreateCoord(COCO17_KEYPOINTS.LEFT_ANKLE),
|
|
2288
|
+
rightAnkle: getOrCreateCoord(COCO17_KEYPOINTS.RIGHT_ANKLE),
|
|
2289
|
+
get: getOrCreateCoord
|
|
2290
|
+
};
|
|
2291
|
+
return new Proxy(namedPose, { get(target, prop, receiver) {
|
|
2292
|
+
if (typeof prop === "string" && /^\d+$/.test(prop)) return target.get(Number(prop));
|
|
2293
|
+
if (typeof prop === "number") return target.get(prop);
|
|
2294
|
+
return Reflect.get(target, prop, receiver);
|
|
2295
|
+
} });
|
|
2296
|
+
};
|
|
2297
|
+
const buildTrackSignals = (resolve$1, label, fixedCategory, fixedId) => {
|
|
2298
|
+
const getBox = (f) => resolve$1(f).boundingBox;
|
|
2299
|
+
const pose = buildTrackPoseSignals(resolve$1, `${label}_pose`, fixedId);
|
|
2300
|
+
const anchor = (name, sx, sy) => createAnchor((f) => ({
|
|
2301
|
+
sx: sx(getBox(f)),
|
|
2302
|
+
sy: sy(getBox(f))
|
|
2303
|
+
}), `${label}_${name}`);
|
|
2304
|
+
const center = anchor("center", (b) => b.originX + b.width / 2, (b) => b.originY + b.height / 2);
|
|
2305
|
+
const speed = programmaticSignal$1((ctx) => resolve$1(ctx.frame).speed, {
|
|
2306
|
+
fps: this._fps,
|
|
2307
|
+
label: `${label}_speed`
|
|
2308
|
+
});
|
|
2309
|
+
return {
|
|
2310
|
+
get trackId() {
|
|
2311
|
+
return fixedId ?? resolve$1(0).trackId;
|
|
2312
|
+
},
|
|
2313
|
+
get category() {
|
|
2314
|
+
return fixedCategory ?? resolve$1(0).category;
|
|
2315
|
+
},
|
|
2316
|
+
bounds: {
|
|
2317
|
+
x: programmaticSignal$1((ctx) => getBox(ctx.frame).normalizedX, {
|
|
2318
|
+
fps: this._fps,
|
|
2319
|
+
label: `${label}_normX`
|
|
2320
|
+
}),
|
|
2321
|
+
y: programmaticSignal$1((ctx) => getBox(ctx.frame).normalizedY, {
|
|
2322
|
+
fps: this._fps,
|
|
2323
|
+
label: `${label}_normY`
|
|
2324
|
+
}),
|
|
2325
|
+
width: programmaticSignal$1((ctx) => getBox(ctx.frame).normalizedWidth, {
|
|
2326
|
+
fps: this._fps,
|
|
2327
|
+
label: `${label}_normWidth`
|
|
2328
|
+
}),
|
|
2329
|
+
height: programmaticSignal$1((ctx) => getBox(ctx.frame).normalizedHeight, {
|
|
2330
|
+
fps: this._fps,
|
|
2331
|
+
label: `${label}_normHeight`
|
|
2332
|
+
}),
|
|
2333
|
+
screenX: programmaticSignal$1((ctx) => getBox(ctx.frame).originX, {
|
|
2334
|
+
fps: this._fps,
|
|
2335
|
+
label: `${label}_screenX`
|
|
2336
|
+
}),
|
|
2337
|
+
screenY: programmaticSignal$1((ctx) => getBox(ctx.frame).originY, {
|
|
2338
|
+
fps: this._fps,
|
|
2339
|
+
label: `${label}_screenY`
|
|
2340
|
+
}),
|
|
2341
|
+
screenWidth: programmaticSignal$1((ctx) => getBox(ctx.frame).width, {
|
|
2342
|
+
fps: this._fps,
|
|
2343
|
+
label: `${label}_screenWidth`
|
|
2344
|
+
}),
|
|
2345
|
+
screenHeight: programmaticSignal$1((ctx) => getBox(ctx.frame).height, {
|
|
2346
|
+
fps: this._fps,
|
|
2347
|
+
label: `${label}_screenHeight`
|
|
2348
|
+
}),
|
|
2349
|
+
aspectRatio: programmaticSignal$1((ctx) => {
|
|
2350
|
+
const b = getBox(ctx.frame);
|
|
2351
|
+
return b.height > 0 ? b.width / b.height : 1;
|
|
2352
|
+
}, {
|
|
2353
|
+
fps: this._fps,
|
|
2354
|
+
label: `${label}_aspectRatio`
|
|
2355
|
+
}),
|
|
2356
|
+
area: programmaticSignal$1((ctx) => {
|
|
2357
|
+
const b = getBox(ctx.frame);
|
|
2358
|
+
return b.width * b.height;
|
|
2359
|
+
}, {
|
|
2360
|
+
fps: this._fps,
|
|
2361
|
+
label: `${label}_area`
|
|
2362
|
+
})
|
|
2363
|
+
},
|
|
2364
|
+
anchors: {
|
|
2365
|
+
topLeft: anchor("topLeft", (b) => b.originX, (b) => b.originY),
|
|
2366
|
+
topCenter: anchor("topCenter", (b) => b.originX + b.width / 2, (b) => b.originY),
|
|
2367
|
+
topRight: anchor("topRight", (b) => b.originX + b.width, (b) => b.originY),
|
|
2368
|
+
centerLeft: anchor("centerLeft", (b) => b.originX, (b) => b.originY + b.height / 2),
|
|
2369
|
+
center,
|
|
2370
|
+
centerRight: anchor("centerRight", (b) => b.originX + b.width, (b) => b.originY + b.height / 2),
|
|
2371
|
+
bottomLeft: anchor("bottomLeft", (b) => b.originX, (b) => b.originY + b.height),
|
|
2372
|
+
bottomCenter: anchor("bottomCenter", (b) => b.originX + b.width / 2, (b) => b.originY + b.height),
|
|
2373
|
+
bottomRight: anchor("bottomRight", (b) => b.originX + b.width, (b) => b.originY + b.height)
|
|
2374
|
+
},
|
|
2375
|
+
kinematics: {
|
|
2376
|
+
vx: programmaticSignal$1((ctx) => resolve$1(ctx.frame).velocity.vx, {
|
|
2377
|
+
fps: this._fps,
|
|
2378
|
+
label: `${label}_vx`
|
|
2379
|
+
}),
|
|
2380
|
+
vy: programmaticSignal$1((ctx) => resolve$1(ctx.frame).velocity.vy, {
|
|
2381
|
+
fps: this._fps,
|
|
2382
|
+
label: `${label}_vy`
|
|
2383
|
+
}),
|
|
2384
|
+
speed,
|
|
2385
|
+
acceleration: speed.velocity(1),
|
|
2386
|
+
headingRad: programmaticSignal$1((ctx) => {
|
|
2387
|
+
const v = resolve$1(ctx.frame).velocity;
|
|
2388
|
+
return Math.atan2(v.vy, v.vx);
|
|
2389
|
+
}, {
|
|
2390
|
+
fps: this._fps,
|
|
2391
|
+
label: `${label}_headingRad`
|
|
2392
|
+
}),
|
|
2393
|
+
headingDeg: programmaticSignal$1((ctx) => {
|
|
2394
|
+
const v = resolve$1(ctx.frame).velocity;
|
|
2395
|
+
return Math.atan2(v.vy, v.vx) * 180 / Math.PI;
|
|
2396
|
+
}, {
|
|
2397
|
+
fps: this._fps,
|
|
2398
|
+
label: `${label}_headingDeg`
|
|
2399
|
+
})
|
|
2400
|
+
},
|
|
2401
|
+
confidence: programmaticSignal$1((ctx) => resolve$1(ctx.frame).score, {
|
|
2402
|
+
fps: this._fps,
|
|
2403
|
+
label: `${label}_score`
|
|
2404
|
+
}),
|
|
2405
|
+
active: programmaticSignal$1((ctx) => resolve$1(ctx.frame).active ? 1 : 0, {
|
|
2406
|
+
fps: this._fps,
|
|
2407
|
+
label: `${label}_active`
|
|
2408
|
+
}),
|
|
2409
|
+
isCoasting: programmaticSignal$1((ctx) => resolve$1(ctx.frame).isCoasting ? 1 : 0, {
|
|
2410
|
+
fps: this._fps,
|
|
2411
|
+
label: `${label}_isCoasting`
|
|
2412
|
+
}),
|
|
2413
|
+
age: programmaticSignal$1((ctx) => resolve$1(ctx.frame).age, {
|
|
2414
|
+
fps: this._fps,
|
|
2415
|
+
label: `${label}_age`
|
|
2416
|
+
}),
|
|
2417
|
+
pose,
|
|
2418
|
+
center,
|
|
2419
|
+
topCenter: this[`__pending`],
|
|
2420
|
+
bottomCenter: center,
|
|
2421
|
+
topLeft: center,
|
|
2422
|
+
bottomLeft: center
|
|
2423
|
+
};
|
|
2424
|
+
};
|
|
2425
|
+
const buildTrack = (resolve$1, label, fixedCategory, fixedId) => {
|
|
2426
|
+
const sig = buildTrackSignals(resolve$1, label, fixedCategory, fixedId);
|
|
2427
|
+
const anchors = sig.anchors;
|
|
2428
|
+
Object.assign(sig, {
|
|
2429
|
+
topCenter: anchors.topCenter,
|
|
2430
|
+
bottomCenter: anchors.bottomCenter,
|
|
2431
|
+
topLeft: anchors.topLeft,
|
|
2432
|
+
bottomLeft: anchors.bottomLeft
|
|
2433
|
+
});
|
|
2434
|
+
return sig;
|
|
2435
|
+
};
|
|
2436
|
+
const getTrackById = (trackId) => {
|
|
2437
|
+
let sig = this._trackCache.get(trackId);
|
|
2438
|
+
if (!sig) {
|
|
2439
|
+
sig = buildTrack((frame) => this.getObjectResult(frame).objects.find((o) => o.trackId === trackId) ?? neutralTrack(trackId), `track_${trackId}`, void 0, trackId);
|
|
2440
|
+
this._trackCache.set(trackId, sig);
|
|
2441
|
+
}
|
|
2442
|
+
return sig;
|
|
2443
|
+
};
|
|
2444
|
+
const getTrackByCategory = (category, rank = 0) => {
|
|
2445
|
+
const cacheKey = `${category.toLowerCase()}_${rank}`;
|
|
2446
|
+
let sig = this._categoryCache.get(cacheKey);
|
|
2447
|
+
if (!sig) {
|
|
2448
|
+
sig = buildTrack((frame) => {
|
|
2449
|
+
return this.getObjectResult(frame).objects.filter((o) => o.category.toLowerCase() === category.toLowerCase() && o.active)[rank] ?? neutralTrack(0, category);
|
|
2450
|
+
}, `cat_${category}_${rank}`, category);
|
|
2451
|
+
this._categoryCache.set(cacheKey, sig);
|
|
2452
|
+
}
|
|
2453
|
+
return sig;
|
|
2454
|
+
};
|
|
2455
|
+
const primaryTrack = buildTrack((frame) => {
|
|
2456
|
+
const res = this.getObjectResult(frame);
|
|
2457
|
+
if (res.objects.length === 0) return neutralTrack(0);
|
|
2458
|
+
let best = res.objects[0];
|
|
2459
|
+
for (let i = 1; i < res.objects.length; i++) if (res.objects[i].score > best.score) best = res.objects[i];
|
|
2460
|
+
return best;
|
|
2461
|
+
}, "primary_object");
|
|
2462
|
+
const activeCategories = (frame) => {
|
|
2463
|
+
const set = /* @__PURE__ */ new Set();
|
|
2464
|
+
for (const o of this.getObjectResult(frame).objects) if (o.active) set.add(o.category);
|
|
2465
|
+
return Array.from(set).sort();
|
|
2466
|
+
};
|
|
2467
|
+
const buildClassSignals = (category) => {
|
|
2468
|
+
const lower = category.toLowerCase();
|
|
2469
|
+
let sig = this._classCache.get(lower);
|
|
2470
|
+
if (sig) return sig;
|
|
2471
|
+
sig = {
|
|
2472
|
+
count: programmaticSignal$1((ctx) => this.getObjectResult(ctx.frame).objects.filter((o) => o.active && o.category.toLowerCase() === lower).length, {
|
|
2473
|
+
fps: this._fps,
|
|
2474
|
+
label: `class_${lower}_count`
|
|
2475
|
+
}),
|
|
2476
|
+
maxConfidence: programmaticSignal$1((ctx) => {
|
|
2477
|
+
let max = 0;
|
|
2478
|
+
for (const o of this.getObjectResult(ctx.frame).objects) if (o.active && o.category.toLowerCase() === lower && o.score > max) max = o.score;
|
|
2479
|
+
return max;
|
|
2480
|
+
}, {
|
|
2481
|
+
fps: this._fps,
|
|
2482
|
+
label: `class_${lower}_maxConf`
|
|
2483
|
+
}),
|
|
2484
|
+
present: programmaticSignal$1((ctx) => this.getObjectResult(ctx.frame).objects.some((o) => o.active && o.category.toLowerCase() === lower) ? 1 : 0, {
|
|
2485
|
+
fps: this._fps,
|
|
2486
|
+
label: `class_${lower}_present`
|
|
2487
|
+
}),
|
|
2488
|
+
primary: getTrackByCategory(category)
|
|
2489
|
+
};
|
|
2490
|
+
this._classCache.set(lower, sig);
|
|
2491
|
+
return sig;
|
|
2492
|
+
};
|
|
2493
|
+
this.classes = {
|
|
2494
|
+
get: buildClassSignals,
|
|
2495
|
+
get names() {
|
|
2496
|
+
return activeCategories(frameSignal.value);
|
|
2497
|
+
},
|
|
2498
|
+
histogram: {
|
|
2499
|
+
get: (ctx) => self.getClassHistogramTensor(ctx?.frame ?? frameSignal.value),
|
|
2500
|
+
get value() {
|
|
2501
|
+
return self.getClassHistogramTensor(frameSignal.value);
|
|
2502
|
+
}
|
|
2503
|
+
}
|
|
2504
|
+
};
|
|
2505
|
+
this.objects = {
|
|
2506
|
+
get: getTrackById,
|
|
2507
|
+
byCategory: getTrackByCategory,
|
|
2508
|
+
primary: primaryTrack,
|
|
2509
|
+
count: programmaticSignal$1((ctx) => this.getObjectResult(ctx.frame).objects.filter((o) => o.active).length, {
|
|
2510
|
+
fps: this._fps,
|
|
2511
|
+
label: "tracked_objects_count"
|
|
2512
|
+
}),
|
|
2513
|
+
hasCategory: (category) => buildClassSignals(category).present,
|
|
2514
|
+
getActiveTracks: (frame) => this.getObjectResult(frame).objects.filter((o) => o.active),
|
|
2515
|
+
get detectedCategories() {
|
|
2516
|
+
return activeCategories(frameSignal.value);
|
|
2517
|
+
}
|
|
2518
|
+
};
|
|
2519
|
+
const maskForTrack = (frame, trackId) => this.getMaskResult(frame).find((m) => m.trackId === trackId);
|
|
2520
|
+
const subjectMask = (frame) => selectSubjectMask(this.getMaskResult(frame));
|
|
2521
|
+
const buildMaskTrackSignals = (resolve$1, label) => {
|
|
2522
|
+
const boxFor = (frame) => {
|
|
2523
|
+
const mask = resolve$1(frame);
|
|
2524
|
+
if (!mask || mask.trackId === void 0) return void 0;
|
|
2525
|
+
return this.getObjectResult(frame).objects.find((o) => o.trackId === mask.trackId);
|
|
2526
|
+
};
|
|
2527
|
+
return {
|
|
2528
|
+
get trackId() {
|
|
2529
|
+
return resolve$1(frameSignal.value)?.trackId ?? 0;
|
|
2530
|
+
},
|
|
2531
|
+
get category() {
|
|
2532
|
+
return resolve$1(frameSignal.value)?.category ?? "unknown";
|
|
2533
|
+
},
|
|
2534
|
+
area: programmaticSignal$1((ctx) => resolve$1(ctx.frame)?.area ?? 0, {
|
|
2535
|
+
fps: this._fps,
|
|
2536
|
+
label: `${label}_area`
|
|
2537
|
+
}),
|
|
2538
|
+
coverage: programmaticSignal$1((ctx) => resolve$1(ctx.frame)?.coverage ?? 0, {
|
|
2539
|
+
fps: this._fps,
|
|
2540
|
+
label: `${label}_coverage`
|
|
2541
|
+
}),
|
|
2542
|
+
solidity: programmaticSignal$1((ctx) => {
|
|
2543
|
+
const mask = resolve$1(ctx.frame);
|
|
2544
|
+
const box = boxFor(ctx.frame);
|
|
2545
|
+
if (!mask || !box) return 0;
|
|
2546
|
+
const bboxArea = box.boundingBox.width * box.boundingBox.height;
|
|
2547
|
+
return bboxArea > 0 ? mask.area / bboxArea : 0;
|
|
2548
|
+
}, {
|
|
2549
|
+
fps: this._fps,
|
|
2550
|
+
label: `${label}_solidity`
|
|
2551
|
+
}),
|
|
2552
|
+
bboxFill: programmaticSignal$1((ctx) => {
|
|
2553
|
+
const mask = resolve$1(ctx.frame);
|
|
2554
|
+
if (!mask || mask.width * mask.height === 0) return 0;
|
|
2555
|
+
const box = boxFor(ctx.frame);
|
|
2556
|
+
if (!box) return 0;
|
|
2557
|
+
return box.boundingBox.width * box.boundingBox.height / (mask.width * mask.height);
|
|
2558
|
+
}, {
|
|
2559
|
+
fps: this._fps,
|
|
2560
|
+
label: `${label}_bboxFill`
|
|
2561
|
+
})
|
|
2562
|
+
};
|
|
2563
|
+
};
|
|
2564
|
+
this.masks = {
|
|
2565
|
+
get: (trackId) => {
|
|
2566
|
+
let sig = this._maskTrackCache.get(trackId);
|
|
2567
|
+
if (!sig) {
|
|
2568
|
+
sig = buildMaskTrackSignals((frame) => maskForTrack(frame, trackId), `mask_${trackId}`);
|
|
2569
|
+
this._maskTrackCache.set(trackId, sig);
|
|
2570
|
+
}
|
|
2571
|
+
return sig;
|
|
2572
|
+
},
|
|
2573
|
+
subject: buildMaskTrackSignals(subjectMask, "subject"),
|
|
2574
|
+
count: programmaticSignal$1((ctx) => new Set(this.getMaskResult(ctx.frame).map((m) => m.trackId ?? -1)).size, {
|
|
2575
|
+
fps: this._fps,
|
|
2576
|
+
label: "mask_count"
|
|
2577
|
+
})
|
|
2578
|
+
};
|
|
2579
|
+
this.segmentation = {
|
|
2580
|
+
humanSilhouette: this.masks.subject,
|
|
2581
|
+
subject: this.masks.subject,
|
|
2582
|
+
instanceMasks: this.masks,
|
|
2583
|
+
matte: {
|
|
2584
|
+
coverage: programmaticSignal$1((ctx) => self.getMatteResult(ctx.frame)?.coverage ?? 0, {
|
|
2585
|
+
fps: this._fps,
|
|
2586
|
+
label: "matte_coverage"
|
|
2587
|
+
}),
|
|
2588
|
+
at: (frame) => self.getMatteResult(frame)
|
|
2589
|
+
},
|
|
2590
|
+
get stencilTexture() {
|
|
2591
|
+
return self._stencilTexture;
|
|
2592
|
+
}
|
|
2593
|
+
};
|
|
2594
|
+
this.poseLandmarksTensor = {
|
|
2595
|
+
get: (ctx) => self.getPoseLandmarksTensor(ctx?.frame ?? frameSignal.value),
|
|
2596
|
+
get value() {
|
|
2597
|
+
return self.getPoseLandmarksTensor(frameSignal.value);
|
|
2598
|
+
}
|
|
2599
|
+
};
|
|
2600
|
+
this.objectsTensor = {
|
|
2601
|
+
get: (ctx) => self.getObjectsTensor(ctx?.frame ?? frameSignal.value),
|
|
2602
|
+
get value() {
|
|
2603
|
+
return self.getObjectsTensor(frameSignal.value);
|
|
2604
|
+
}
|
|
2605
|
+
};
|
|
2606
|
+
this.masksTensor = {
|
|
2607
|
+
get: (ctx) => self.getMasksTensor(ctx?.frame ?? frameSignal.value),
|
|
2608
|
+
get value() {
|
|
2609
|
+
return self.getMasksTensor(frameSignal.value);
|
|
2610
|
+
}
|
|
2611
|
+
};
|
|
2612
|
+
this.histogramTensor = {
|
|
2613
|
+
get: (ctx) => self.getClassHistogramTensor(ctx?.frame ?? frameSignal.value),
|
|
2614
|
+
get value() {
|
|
2615
|
+
return self.getClassHistogramTensor(frameSignal.value);
|
|
2616
|
+
}
|
|
2617
|
+
};
|
|
2618
|
+
}
|
|
2619
|
+
setStencilTexture(texture) {
|
|
2620
|
+
this._stencilTexture = texture;
|
|
2621
|
+
}
|
|
2622
|
+
get stencilTexture() {
|
|
2623
|
+
return this._stencilTexture;
|
|
2624
|
+
}
|
|
2625
|
+
/**
|
|
2626
|
+
* Reads the most recent result at or before `frame`. Vision results are written one
|
|
2627
|
+
* frame behind the plate (the node renderer reads back the previous frame to infer),
|
|
2628
|
+
* and pinned-signal layout runs before the current frame's inference, so a strict
|
|
2629
|
+
* per-frame lookup would always miss. Walking back a few frames keeps signals live at
|
|
2630
|
+
* a stable one-frame lag instead of snapping to neutral defaults.
|
|
2631
|
+
*/
|
|
2632
|
+
latestAtOrBefore(cache, frame, maxBack = 8) {
|
|
2633
|
+
const start = this.clamp(frame);
|
|
2634
|
+
const floor = Math.max(0, start - maxBack);
|
|
2635
|
+
for (let f = start; f >= floor; f--) {
|
|
2636
|
+
const value = cache.get(f);
|
|
2637
|
+
if (value !== void 0) return value;
|
|
2638
|
+
}
|
|
2639
|
+
}
|
|
2640
|
+
invalidateSignals() {
|
|
2641
|
+
for (const sig of this._registeredSignals) sig.invalidate();
|
|
2642
|
+
}
|
|
2643
|
+
setPoseResult(frame, result) {
|
|
2644
|
+
this._poseCache.set(this.clamp(frame), result);
|
|
2645
|
+
this.invalidateSignals();
|
|
2646
|
+
}
|
|
2647
|
+
getPoseResult(frame) {
|
|
2648
|
+
return this.latestAtOrBefore(this._poseCache, frame) ?? createNeutralPoseResult();
|
|
2649
|
+
}
|
|
2650
|
+
setObjectResult(frame, result) {
|
|
2651
|
+
const normalized = "objects" in result ? result : {
|
|
2652
|
+
objects: result,
|
|
2653
|
+
rawDetections: []
|
|
2654
|
+
};
|
|
2655
|
+
this._objectCache.set(this.clamp(frame), {
|
|
2656
|
+
objects: normalized.objects,
|
|
2657
|
+
rawDetections: normalized.rawDetections ?? []
|
|
2658
|
+
});
|
|
2659
|
+
this.invalidateSignals();
|
|
2660
|
+
}
|
|
2661
|
+
getObjectResult(frame) {
|
|
2662
|
+
return this.latestAtOrBefore(this._objectCache, frame) ?? createNeutralObjectResult();
|
|
2663
|
+
}
|
|
2664
|
+
setMaskResult(frame, masks) {
|
|
2665
|
+
this._maskCache.set(this.clamp(frame), masks);
|
|
2666
|
+
this.invalidateSignals();
|
|
2667
|
+
}
|
|
2668
|
+
getMaskResult(frame) {
|
|
2669
|
+
return this.latestAtOrBefore(this._maskCache, frame) ?? [];
|
|
2670
|
+
}
|
|
2671
|
+
setMatteResult(frame, matte) {
|
|
2672
|
+
this._matteCache.set(this.clamp(frame), matte);
|
|
2673
|
+
this.invalidateSignals();
|
|
2674
|
+
}
|
|
2675
|
+
getMatteResult(frame) {
|
|
2676
|
+
return this.latestAtOrBefore(this._matteCache, frame);
|
|
2677
|
+
}
|
|
2678
|
+
/**
|
|
2679
|
+
* Serializable snapshot of one frame's object / class / mask signals.
|
|
2680
|
+
* Synchronous — reads only the per-frame caches, never triggers inference.
|
|
2681
|
+
*/
|
|
2682
|
+
summary(frame = frameSignal.value) {
|
|
2683
|
+
const f = this.clamp(frame);
|
|
2684
|
+
const res = this.getObjectResult(f);
|
|
2685
|
+
const objects = res.objects.map((o) => ({
|
|
2686
|
+
trackId: o.trackId,
|
|
2687
|
+
category: o.category,
|
|
2688
|
+
score: o.score,
|
|
2689
|
+
center: [o.centerX, o.centerY],
|
|
2690
|
+
speed: o.speed,
|
|
2691
|
+
active: o.active
|
|
2692
|
+
}));
|
|
2693
|
+
const classSet = /* @__PURE__ */ new Set();
|
|
2694
|
+
for (const o of res.objects) if (o.active) classSet.add(o.category);
|
|
2695
|
+
const masks = {};
|
|
2696
|
+
for (const m of this.getMaskResult(f)) masks[m.category] = (masks[m.category] ?? 0) + 1;
|
|
2697
|
+
return {
|
|
2698
|
+
frame: f,
|
|
2699
|
+
objects,
|
|
2700
|
+
classes: Array.from(classSet).sort(),
|
|
2701
|
+
masks
|
|
2702
|
+
};
|
|
2703
|
+
}
|
|
2704
|
+
/** Primary person's 17 keypoints as a [17, 3] float32 tensor. */
|
|
2705
|
+
getPoseLandmarksTensor(frame) {
|
|
2706
|
+
const kpts = this.getPoseResult(frame).people[0]?.keypoints ?? [];
|
|
2707
|
+
const data = new Float32Array(51);
|
|
2708
|
+
for (let i = 0; i < 17; i++) {
|
|
2709
|
+
data[i * 3] = kpts[i]?.x ?? .5;
|
|
2710
|
+
data[i * 3 + 1] = kpts[i]?.y ?? .5;
|
|
2711
|
+
data[i * 3 + 2] = kpts[i]?.visibility ?? 0;
|
|
2712
|
+
}
|
|
2713
|
+
return {
|
|
2714
|
+
data,
|
|
2715
|
+
shape: [17, 3],
|
|
2716
|
+
dtype: "float32"
|
|
2717
|
+
};
|
|
2718
|
+
}
|
|
2719
|
+
/** Up to 16 active tracks as [16, 8]: [active, categoryHash, x, y, w, h, vx, vy]. */
|
|
2720
|
+
getObjectsTensor(frame, maxObjects = 16) {
|
|
2721
|
+
const res = this.getObjectResult(frame);
|
|
2722
|
+
const data = new Float32Array(maxObjects * 8);
|
|
2723
|
+
const count = Math.min(maxObjects, res.objects.length);
|
|
2724
|
+
for (let i = 0; i < count; i++) {
|
|
2725
|
+
const obj = res.objects[i];
|
|
2726
|
+
const off = i * 8;
|
|
2727
|
+
let catHash = 0;
|
|
2728
|
+
for (let c = 0; c < obj.category.length; c++) catHash = catHash * 31 + obj.category.charCodeAt(c) & 65535;
|
|
2729
|
+
data[off] = obj.active ? 1 : 0;
|
|
2730
|
+
data[off + 1] = catHash;
|
|
2731
|
+
data[off + 2] = obj.centerX;
|
|
2732
|
+
data[off + 3] = obj.centerY;
|
|
2733
|
+
data[off + 4] = obj.boundingBox.width;
|
|
2734
|
+
data[off + 5] = obj.boundingBox.height;
|
|
2735
|
+
data[off + 6] = obj.velocity.vx;
|
|
2736
|
+
data[off + 7] = obj.velocity.vy;
|
|
2737
|
+
}
|
|
2738
|
+
return {
|
|
2739
|
+
data,
|
|
2740
|
+
shape: [maxObjects, 8],
|
|
2741
|
+
dtype: "float32"
|
|
2742
|
+
};
|
|
2743
|
+
}
|
|
2744
|
+
/** Per-mask coverage + solidity as [16, 2]. */
|
|
2745
|
+
getMasksTensor(frame, maxMasks = 16) {
|
|
2746
|
+
const masks = this.getMaskResult(frame);
|
|
2747
|
+
const data = new Float32Array(maxMasks * 2);
|
|
2748
|
+
const count = Math.min(maxMasks, masks.length);
|
|
2749
|
+
for (let i = 0; i < count; i++) {
|
|
2750
|
+
const m = masks[i];
|
|
2751
|
+
data[i * 2] = m.coverage;
|
|
2752
|
+
const box = this.getObjectResult(frame).objects.find((o) => o.trackId === m.trackId);
|
|
2753
|
+
const bboxArea = box ? box.boundingBox.width * box.boundingBox.height : 0;
|
|
2754
|
+
data[i * 2 + 1] = bboxArea > 0 ? m.area / bboxArea : 0;
|
|
2755
|
+
}
|
|
2756
|
+
return {
|
|
2757
|
+
data,
|
|
2758
|
+
shape: [maxMasks, 2],
|
|
2759
|
+
dtype: "float32"
|
|
2760
|
+
};
|
|
2761
|
+
}
|
|
2762
|
+
/** Per-class detection histogram for the frame as [nc] float32 (COCO 80 by default). */
|
|
2763
|
+
getClassHistogramTensor(frame, nc = 80) {
|
|
2764
|
+
const data = new Float32Array(nc);
|
|
2765
|
+
for (const o of this.getObjectResult(frame).objects) if (o.active) {
|
|
2766
|
+
const idx = COCO_CLASS_INDEX_BY_NAME.get(o.category.toLowerCase());
|
|
2767
|
+
if (idx !== void 0) data[idx] += 1;
|
|
2768
|
+
}
|
|
2769
|
+
return {
|
|
2770
|
+
data,
|
|
2771
|
+
shape: [nc],
|
|
2772
|
+
dtype: "float32"
|
|
2773
|
+
};
|
|
2774
|
+
}
|
|
2775
|
+
clamp(frame) {
|
|
2776
|
+
return Math.max(0, Math.min(this._totalFrames - 1, Math.round(frame)));
|
|
2777
|
+
}
|
|
2778
|
+
};
|
|
2779
|
+
function createVisionBundle(options) {
|
|
2780
|
+
return new VisionBundle(options);
|
|
2781
|
+
}
|
|
2782
|
+
const COCO_CLASS_INDEX_BY_NAME = new Map(COCO_CLASSES.map((name, index) => [name, index]));
|
|
2783
|
+
|
|
2784
|
+
//#endregion
|
|
2785
|
+
//#region src/spatial/spatial-pin.ts
|
|
2786
|
+
function pinNodeToLandmark(node, target, options = {}) {
|
|
2787
|
+
const offX = options.offsetX ?? 0;
|
|
2788
|
+
const offY = options.offsetY ?? 0;
|
|
2789
|
+
const offZ = options.offsetZ ?? 0;
|
|
2790
|
+
if (!target || typeof target !== "object") return node;
|
|
2791
|
+
if ("screenX" in target && target.screenX && typeof target.screenX === "object") {
|
|
2792
|
+
const signals = target;
|
|
2793
|
+
node.x = offX !== 0 ? signals.screenX.add(offX) : signals.screenX;
|
|
2794
|
+
node.y = offY !== 0 ? signals.screenY.add(offY) : signals.screenY;
|
|
2795
|
+
node.z = offZ !== 0 ? signals.z.add(offZ) : signals.z;
|
|
2796
|
+
} else if ("x" in target && typeof target.x === "object" && target.x !== null) {
|
|
2797
|
+
const signals = target;
|
|
2798
|
+
node.x = offX !== 0 ? signals.x.add(offX) : signals.x;
|
|
2799
|
+
node.y = offY !== 0 ? signals.y.add(offY) : signals.y;
|
|
2800
|
+
node.z = offZ !== 0 ? signals.z.add(offZ) : signals.z;
|
|
2801
|
+
} else {
|
|
2802
|
+
const staticCoord = target;
|
|
2803
|
+
node.x = (staticCoord.screenX ?? staticCoord.x) + offX;
|
|
2804
|
+
node.y = (staticCoord.screenY ?? staticCoord.y) + offY;
|
|
2805
|
+
node.z = (staticCoord.z ?? 0) + offZ;
|
|
2806
|
+
}
|
|
2807
|
+
return node;
|
|
2808
|
+
}
|
|
2809
|
+
function pinNodeToObject(node, target, options = {}) {
|
|
2810
|
+
if (!target || typeof target !== "object") return node;
|
|
2811
|
+
if ("anchors" in target && target.anchors) {
|
|
2812
|
+
const track = target;
|
|
2813
|
+
const anchorName = options.anchor ?? (options.matchWidth || options.matchHeight ? "topLeft" : "center");
|
|
2814
|
+
const rawAnchor = track.anchors[anchorName] ?? track.center;
|
|
2815
|
+
let resolvedAnchor = rawAnchor;
|
|
2816
|
+
if (options.smoothFrames && options.smoothFrames > 1) resolvedAnchor = {
|
|
2817
|
+
x: rawAnchor.x.smooth(options.smoothFrames),
|
|
2818
|
+
y: rawAnchor.y.smooth(options.smoothFrames),
|
|
2819
|
+
z: rawAnchor.z,
|
|
2820
|
+
screenX: rawAnchor.screenX.smooth(options.smoothFrames),
|
|
2821
|
+
screenY: rawAnchor.screenY.smooth(options.smoothFrames)
|
|
2822
|
+
};
|
|
2823
|
+
if (options.matchWidth) node.width = options.smoothFrames && options.smoothFrames > 1 ? track.bounds.screenWidth.smooth(options.smoothFrames) : track.bounds.screenWidth;
|
|
2824
|
+
if (options.matchHeight) node.height = options.smoothFrames && options.smoothFrames > 1 ? track.bounds.screenHeight.smooth(options.smoothFrames) : track.bounds.screenHeight;
|
|
2825
|
+
if (options.hideWhenLost) node.opacity = track.active;
|
|
2826
|
+
return pinNodeToLandmark(node, resolvedAnchor, options);
|
|
2827
|
+
}
|
|
2828
|
+
return pinNodeToLandmark(node, target, options);
|
|
2829
|
+
}
|
|
2830
|
+
|
|
2831
|
+
//#endregion
|
|
2832
|
+
//#region src/vision-node.ts
|
|
2833
|
+
/** Tasks a config turns on; detection is the default when nothing is enabled explicitly. */
|
|
2834
|
+
function enabledTasks(config) {
|
|
2835
|
+
const tasks = [];
|
|
2836
|
+
if (config.enableDetection === true) tasks.push("detect");
|
|
2837
|
+
if (config.enableSegmentation === true) tasks.push("segment");
|
|
2838
|
+
if (config.enablePose === true) tasks.push("pose");
|
|
2839
|
+
if (config.enableMatte === true) tasks.push("matte");
|
|
2840
|
+
return tasks.length > 0 ? tasks : ["detect"];
|
|
2841
|
+
}
|
|
2842
|
+
/**
|
|
2843
|
+
* Vision node.
|
|
2844
|
+
*
|
|
2845
|
+
* LAZY BY CONSTRUCTION: creating a VisionNode performs ZERO I/O — no model downloads, no
|
|
2846
|
+
* sessions, no file probes. The first inference call (or an explicit `await vision.ready()`)
|
|
2847
|
+
* downloads the required models.
|
|
2848
|
+
*/
|
|
2849
|
+
var VisionNode = class VisionNode {
|
|
2850
|
+
id;
|
|
2851
|
+
kind = "vision";
|
|
2852
|
+
source;
|
|
2853
|
+
config;
|
|
2854
|
+
vision;
|
|
2855
|
+
_runner;
|
|
2856
|
+
/** Runner construction options (modelsDir, provider, store, …) — injectable for tests/agents. */
|
|
2857
|
+
_runnerOptions;
|
|
2858
|
+
constructor(source, config = {}, bundleOptions = {}, runnerOptions = {}) {
|
|
2859
|
+
this.id = `vision-${Math.random().toString(36).slice(2, 9)}`;
|
|
2860
|
+
this.source = source;
|
|
2861
|
+
this.config = config;
|
|
2862
|
+
this._runnerOptions = runnerOptions;
|
|
2863
|
+
this.vision = new VisionBundle({
|
|
2864
|
+
...bundleOptions,
|
|
2865
|
+
config
|
|
2866
|
+
});
|
|
2867
|
+
}
|
|
2868
|
+
/** Lazily-created runner; downloads happen on its first inference. */
|
|
2869
|
+
runner() {
|
|
2870
|
+
if (!this._runner) {
|
|
2871
|
+
const { variant, confidence, classes, modelsDir, baseUrl, maskThreshold, featherRadius } = this.config;
|
|
2872
|
+
this._runner = VisionRunner.create({
|
|
2873
|
+
...this._runnerOptions,
|
|
2874
|
+
...variant !== void 0 ? { variant } : {},
|
|
2875
|
+
...confidence !== void 0 ? { confidence } : {},
|
|
2876
|
+
...classes !== void 0 ? { classes } : {},
|
|
2877
|
+
...modelsDir !== void 0 ? { modelsDir } : {},
|
|
2878
|
+
...baseUrl !== void 0 ? { baseUrl } : {},
|
|
2879
|
+
...maskThreshold !== void 0 ? { maskThreshold } : {},
|
|
2880
|
+
...featherRadius !== void 0 ? { featherRadius } : {}
|
|
2881
|
+
});
|
|
2882
|
+
}
|
|
2883
|
+
return this._runner;
|
|
2884
|
+
}
|
|
2885
|
+
/** Explicit warm-up — downloads the models for the enabled tasks, ahead of first inference. */
|
|
2886
|
+
async ready() {
|
|
2887
|
+
const runner = this.runner();
|
|
2888
|
+
await runner.preload(enabledTasks(this.config));
|
|
2889
|
+
return runner;
|
|
2890
|
+
}
|
|
2891
|
+
close() {
|
|
2892
|
+
this._runner?.close();
|
|
2893
|
+
this._runner = void 0;
|
|
2894
|
+
}
|
|
2895
|
+
static attach(source, config = {}, bundleOptions = {}) {
|
|
2896
|
+
const node = new VisionNode(source, config, bundleOptions);
|
|
2897
|
+
const vision = node.vision;
|
|
2898
|
+
Object.defineProperties(vision, {
|
|
2899
|
+
node: {
|
|
2900
|
+
value: node.toNode(),
|
|
2901
|
+
enumerable: true
|
|
2902
|
+
},
|
|
2903
|
+
ready: { value: () => node.ready() },
|
|
2904
|
+
runner: { value: () => node.runner() },
|
|
2905
|
+
close: { value: () => node.close() }
|
|
2906
|
+
});
|
|
2907
|
+
return vision;
|
|
2908
|
+
}
|
|
2909
|
+
toNode() {
|
|
2910
|
+
return {
|
|
2911
|
+
id: this.id,
|
|
2912
|
+
kind: "vision",
|
|
2913
|
+
source: this.source,
|
|
2914
|
+
config: this.config
|
|
2915
|
+
};
|
|
2916
|
+
}
|
|
2917
|
+
};
|
|
2918
|
+
|
|
2919
|
+
//#endregion
|
|
2920
|
+
//#region src/index.ts
|
|
2921
|
+
var src_exports = /* @__PURE__ */ __exportAll({
|
|
2922
|
+
AutoSessionProvider: () => AutoSessionProvider,
|
|
2923
|
+
COCO17_BONES: () => COCO17_BONES,
|
|
2924
|
+
COCO17_KEYPOINTS: () => COCO17_KEYPOINTS,
|
|
2925
|
+
COCO17_KEYPOINT_COUNT: () => COCO17_KEYPOINT_COUNT,
|
|
2926
|
+
COCO17_KEYPOINT_NAMES: () => COCO17_KEYPOINT_NAMES,
|
|
2927
|
+
COCO_CLASSES: () => COCO_CLASSES,
|
|
2928
|
+
NodeSessionProvider: () => NodeSessionProvider,
|
|
2929
|
+
NodeWebGPUProvider: () => NodeWebGPUProvider,
|
|
2930
|
+
PREPROCESS_BY_FAMILY: () => PREPROCESS_BY_FAMILY,
|
|
2931
|
+
PoseSkeletonRenderer: () => PoseSkeletonRenderer,
|
|
2932
|
+
SegmentationTexturePool: () => SegmentationTexturePool,
|
|
2933
|
+
SpatialLandmarkTransformer: () => SpatialLandmarkTransformer,
|
|
2934
|
+
TemporalObjectTracker: () => TemporalObjectTracker,
|
|
2935
|
+
VISION_MODELS: () => VISION_MODELS,
|
|
2936
|
+
VISION_TASKS: () => VISION_TASKS,
|
|
2937
|
+
VISION_VARIANTS: () => VISION_VARIANTS,
|
|
2938
|
+
VisionBundle: () => VisionBundle,
|
|
2939
|
+
VisionModelStore: () => VisionModelStore,
|
|
2940
|
+
VisionNode: () => VisionNode,
|
|
2941
|
+
VisionRunner: () => VisionRunner,
|
|
2942
|
+
analyzeSequence: () => analyzeSequence,
|
|
2943
|
+
computeInputTransform: () => computeInputTransform,
|
|
2944
|
+
computeIoU: () => computeIoU,
|
|
2945
|
+
createDefaultSessionProvider: () => createDefaultSessionProvider,
|
|
2946
|
+
createNeutralLandmarks: () => createNeutralLandmarks,
|
|
2947
|
+
createNeutralObjectResult: () => createNeutralObjectResult,
|
|
2948
|
+
createNeutralPoseResult: () => createNeutralPoseResult,
|
|
2949
|
+
createNodeWebGPUProvider: () => createNodeWebGPUProvider,
|
|
2950
|
+
createVisionBundle: () => createVisionBundle,
|
|
2951
|
+
decodeRtmdetIns: () => decodeRtmdetIns,
|
|
2952
|
+
decodeRtmo: () => decodeRtmo,
|
|
2953
|
+
decodeSelfie: () => decodeSelfie,
|
|
2954
|
+
ensureNodeWebGPU: () => ensureNodeWebGPU,
|
|
2955
|
+
getDefaultModelsDir: () => getDefaultModelsDir,
|
|
2956
|
+
getPersonBoundingBox: () => getPersonBoundingBox,
|
|
2957
|
+
imageToTensor: () => imageToTensor,
|
|
2958
|
+
maskBounds: () => maskBounds,
|
|
2959
|
+
matchPoseToTracks: () => matchPoseToTracks,
|
|
2960
|
+
mergeSubjectMask: () => mergeSubjectMask,
|
|
2961
|
+
modelKeyFor: () => modelKeyFor,
|
|
2962
|
+
pinNodeToLandmark: () => pinNodeToLandmark,
|
|
2963
|
+
pinNodeToObject: () => pinNodeToObject,
|
|
2964
|
+
preprocessSpec: () => preprocessSpec,
|
|
2965
|
+
probabilityToAlpha: () => probabilityToAlpha,
|
|
2966
|
+
selectSubjectMask: () => selectSubjectMask,
|
|
2967
|
+
setDefaultSessionProvider: () => setDefaultSessionProvider,
|
|
2968
|
+
toSourcePoint: () => toSourcePoint
|
|
2969
|
+
});
|
|
2970
|
+
|
|
2971
|
+
//#endregion
|
|
2972
|
+
export { PoseSkeletonRenderer as A, computeInputTransform as B, hasWebGPU as C, VisionModelStore as D, createNeutralPoseResult as E, COCO17_KEYPOINT_COUNT as F, TemporalObjectTracker as G, preprocessSpec as H, decodeRtmo as I, computeIoU as K, decodeRtmdetIns as L, COCO17_KEYPOINTS as M, COCO17_KEYPOINT_NAMES as N, getDefaultModelsDir as O, decodeSelfie as P, probabilityToAlpha as R, createWebGPUProvider as S, createNeutralObjectResult as T, toSourcePoint as U, imageToTensor as V, analyzeSequence as W, setDefaultSessionProvider as _, VisionBundle as a, ensureNodeWebGPU as b, matchPoseToTracks as c, mergeSubjectMask as d, selectSubjectMask as f, createDefaultSessionProvider as g, NodeSessionProvider as h, pinNodeToObject as i, COCO17_BONES as j, SegmentationTexturePool as k, SpatialLandmarkTransformer as l, AutoSessionProvider as m, VisionNode as n, createVisionBundle as o, VisionRunner as p, pinNodeToLandmark as r, getPersonBoundingBox as s, src_exports as t, maskBounds as u, NodeWebGPUProvider as v, createNeutralLandmarks as w, WebGPUProvider as x, createNodeWebGPUProvider as y, PREPROCESS_BY_FAMILY as z };
|
|
2973
|
+
//# sourceMappingURL=src-C2sA_wDT.mjs.map
|