@camstack/addon-pipeline 1.2.295 → 1.2.296

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/THIRD_PARTY_MODELS.md +8 -0
  2. package/dist/audio-analyzer/index.js +2 -2
  3. package/dist/audio-analyzer/index.mjs +2 -2
  4. package/dist/{default-detection-model-CAUBVbgK.mjs → default-detection-model-Co578D8C.mjs} +75 -46
  5. package/dist/{default-detection-model-wNEISVA9.js → default-detection-model-D1daTtqT.js} +75 -46
  6. package/dist/detection-pipeline/index.js +1023 -80
  7. package/dist/detection-pipeline/index.mjs +1023 -80
  8. package/dist/{dist-BsVcf5wO.js → dist-8up-f2TX.js} +877 -365
  9. package/dist/{dist-CuxSNLKW.mjs → dist-CCd0Q3nr.mjs} +872 -366
  10. package/dist/motion-wasm/index.js +1 -1
  11. package/dist/motion-wasm/index.mjs +1 -1
  12. package/dist/{node-Cqb1QiXk.mjs → node-DgMSXSWP.mjs} +1 -1
  13. package/dist/{node-UFk6I2f6.js → node-lpQgHes9.js} +1 -1
  14. package/dist/pipeline-runner/index.js +789 -203
  15. package/dist/pipeline-runner/index.mjs +789 -203
  16. package/dist/{process-memory-CJV29sPf.js → process-memory-CX_92V_r.js} +1 -1
  17. package/dist/{process-memory-D0Nvs9rr.mjs → process-memory-DFC_O5zE.mjs} +1 -1
  18. package/dist/recorder/index.js +4 -6
  19. package/dist/recorder/index.mjs +4 -6
  20. package/dist/{segment-demux-js-DGwmf5hu.js → segment-demux-js-DzBx6NN2.js} +1 -1
  21. package/dist/{segment-demux-js-DIDWw1iE.mjs → segment-demux-js-FZbBuk3F.mjs} +1 -1
  22. package/dist/session-decode/{decode-worker-child.js → decode-worker-main.js} +481 -72
  23. package/dist/session-decode/{decode-worker-child.mjs → decode-worker-main.mjs} +482 -71
  24. package/dist/stream-broker/_stub.js +2 -2
  25. package/dist/stream-broker/{_virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-B7nBFqva.mjs → _virtual_mf-localSharedImportMap___mfe_internal__addon_stream_broker_widgets-C_i7oFBl.mjs} +2 -2
  26. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-DUGQKsKL.mjs +26 -0
  27. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-CkbplMHA.mjs +26 -0
  28. package/dist/stream-broker/demux-worker-child.js +1 -1
  29. package/dist/stream-broker/demux-worker-child.mjs +1 -1
  30. package/dist/stream-broker/{hostInit-BKlb3qac.mjs → hostInit-BPtppL3W.mjs} +2 -2
  31. package/dist/stream-broker/index.js +4 -4
  32. package/dist/stream-broker/index.mjs +4 -4
  33. package/dist/stream-broker/remoteEntry.js +1 -1
  34. package/dist/{worker-protocol-B2MfQLlu.js → worker-protocol-C-G8qmye.js} +3 -1
  35. package/dist/{worker-protocol-C_W-P_g-.mjs → worker-protocol-D_NzPcnh.mjs} +3 -1
  36. package/package.json +1 -1
  37. package/python/inference_pool.py +422 -64
  38. package/python/test_inference_pool_compile_off_loop.py +414 -0
  39. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_types__loadShare__.js-CCIyBvRa.mjs +0 -26
  40. package/dist/stream-broker/_virtual_mf___mfe_internal__addon_stream_broker_widgets__loadShare___mf_0_camstack_mf_1_ui_mf_2_library__loadShare__.js-DeD_UFdb.mjs +0 -26
@@ -61,6 +61,12 @@ under [Non-commercial models](#non-commercial-models).
61
61
  https://github.com/ShiqiYu/libfacedetection.train.
62
62
  Trained on WIDER FACE, whose images are CC BY-NC-ND.
63
63
  - **AuraFace v1** — fal.ai, Apache-2.0. https://huggingface.co/fal/AuraFace-v1
64
+ - **SigLIP 2** — © Google LLC, Apache-2.0 (`Apache-2.0.txt`).
65
+ https://huggingface.co/google/siglip2-base-patch16-224 ·
66
+ https://github.com/google-research/big_vision. CamStack's SigLIP 2 files are
67
+ modified: the vision input scaling `2x - 1` is baked into the graph, the text
68
+ graph slices its input to the model's 64 positions, the tokenizer lower-cases,
69
+ and the towers are converted to ONNX, OpenVINO IR and CoreML FP16.
64
70
  - **EasyOCR** — JaidedAI, Apache-2.0. https://github.com/JaidedAI/EasyOCR
65
71
  - **U²-Net** — Xuebin Qin et al., Apache-2.0. https://github.com/xuebinqin/U-2-Net
66
72
  - **YAMNet** — © Google LLC, Apache-2.0.
@@ -202,6 +208,8 @@ commercial use**:
202
208
  | --- | --- | --- | --- | --- | --- |
203
209
  | `mobileclip-s1`, `mobileclip-s2` (image) | Apple `ml-mobileclip` | MIT / **Apple Machine Learning Research Model License** | per Apple | Model Derivative: converted to OpenVINO IR and CoreML FP16 | Mirror `clip/mobileclip-s1`, `clip/mobileclip-s2` |
204
210
  | `mobileclip-s1-text`, `mobileclip-s2-text` | Apple `ml-mobileclip`, with the CLIP tokenizer | MIT / **Apple Machine Learning Research Model License** | per Apple | Model Derivative: quantised to INT8 ONNX and converted to OpenVINO IR | Mirror `clip/mobileclip-s1`, `clip/mobileclip-s2` |
211
+ | `siglip2-b16-224` (image) | Google SigLIP 2 base patch16-224 (`google/siglip2-base-patch16-224`) | Apache-2.0 / Apache-2.0 | WebLI (Google internal, not released) | input scaling `y = 2x - 1` baked ahead of the patch embedding; converted to OpenVINO IR and CoreML FP16 (an FP32 ONNX is hosted as the verification reference) (`scripts/build-siglip2-model.py`) | Mirror `clip/siglip2` |
212
+ | `siglip2-b16-224-text` | Google SigLIP 2 base patch16-224, with its Gemma tokenizer | Apache-2.0 / Apache-2.0 | WebLI (Google internal, not released) | input sliced to the model's 64 positions in the graph; ONNX weights FP16, OpenVINO IR and CoreML FP16; a `Lowercase` normalizer prepended to the tokenizer | Mirror `clip/siglip2` |
205
213
 
206
214
  ### Segmentation
207
215
 
@@ -3,8 +3,8 @@ Object.defineProperties(exports, {
3
3
  [Symbol.toStringTag]: { value: "Module" }
4
4
  });
5
5
  const require_chunk = require("../chunk-emK7D4bc.js");
6
- const require_dist = require("../dist-BsVcf5wO.js");
7
- const require_process_memory = require("../process-memory-CJV29sPf.js");
6
+ const require_dist = require("../dist-8up-f2TX.js");
7
+ const require_process_memory = require("../process-memory-CX_92V_r.js");
8
8
  let node_fs = require("node:fs");
9
9
  node_fs = require_chunk.__toESM(node_fs);
10
10
  let node_path = require("node:path");
@@ -1,6 +1,6 @@
1
1
  import { n as __require } from "../chunk-DnnnRqeS.mjs";
2
- import { E as HF_BASE_URL, G as audioAnalyzerCapability, Rn as EventCategory, W as audioAnalysisCapability, Wt as resolvePoolMemoryPolicy, _n as hydrateSchema, bt as mapAudioLabelToMacro, en as errMsg, f as DEFAULT_AUDIO_ANALYZER_CONFIG, fn as audioChunkBytesPerSample, gn as expandAudioChunkToF32le, j as PoolMemoryWatchdog, n as AUDIO_BACKEND_CHOICES, sn as BaseAddon, xn as nodePin } from "../dist-CuxSNLKW.mjs";
3
- import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-D0Nvs9rr.mjs";
2
+ import { E as HF_BASE_URL, G as audioAnalyzerCapability, Gt as resolvePoolMemoryPolicy, Sn as nodePin, W as audioAnalysisCapability, _n as expandAudioChunkToF32le, bt as mapAudioLabelToMacro, cn as BaseAddon, f as DEFAULT_AUDIO_ANALYZER_CONFIG, j as PoolMemoryWatchdog, n as AUDIO_BACKEND_CHOICES, pn as audioChunkBytesPerSample, tn as errMsg, vn as hydrateSchema, zn as EventCategory } from "../dist-CCd0Q3nr.mjs";
3
+ import { n as pickNodePlatformArch, t as readProcessMemory } from "../process-memory-DFC_O5zE.mjs";
4
4
  import * as fs from "node:fs";
5
5
  import * as path$1 from "node:path";
6
6
  import { downloadFile } from "@camstack/system/addon-utils";
@@ -1,4 +1,4 @@
1
- import { K as audioKindId, d as COCO_TO_MACRO, ht as hfModelUrl, r as AUDIO_MACRO_LABELS, u as COCO_80_LABELS } from "./dist-CuxSNLKW.mjs";
1
+ import { K as audioKindId, d as COCO_TO_MACRO, ht as hfModelUrl, r as AUDIO_MACRO_LABELS, u as COCO_80_LABELS } from "./dist-CCd0Q3nr.mjs";
2
2
  import { randomUUID } from "node:crypto";
3
3
  //#region src/detection-pipeline/pipeline/landmark-precision-gate.ts
4
4
  /**
@@ -1184,55 +1184,84 @@ var INSTANCE_SEGMENTATION_MODELS = [
1184
1184
  }
1185
1185
  }
1186
1186
  ];
1187
- var CLIP_EMBEDDING_MODELS = [{
1188
- id: "mobileclip-s1",
1189
- name: "MobileCLIP S1",
1190
- description: "MobileCLIP S1 — Apple balanced CLIP vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
1191
- inputSize: {
1192
- width: 256,
1193
- height: 256
1187
+ var CLIP_EMBEDDING_MODELS = [
1188
+ {
1189
+ id: "mobileclip-s1",
1190
+ name: "MobileCLIP S1",
1191
+ description: "MobileCLIP S1 — Apple balanced CLIP vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
1192
+ inputSize: {
1193
+ width: 256,
1194
+ height: 256
1195
+ },
1196
+ labels: [{
1197
+ id: "embedding",
1198
+ name: "CLIP Embedding"
1199
+ }],
1200
+ preprocessMode: "resize",
1201
+ inputNormalization: "none",
1202
+ formats: {
1203
+ openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-vision.xml"), 55),
1204
+ coreml: {
1205
+ url: hf("clip/mobileclip-s1/coreml/camstack-mobileclip-s1-vision.mlpackage"),
1206
+ sizeMB: 65,
1207
+ isDirectory: true,
1208
+ files: [...MLPACKAGE_FILES],
1209
+ runtimes: ["python"]
1210
+ }
1211
+ }
1194
1212
  },
1195
- labels: [{
1196
- id: "embedding",
1197
- name: "CLIP Embedding"
1198
- }],
1199
- preprocessMode: "resize",
1200
- inputNormalization: "none",
1201
- formats: {
1202
- openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-vision.xml"), 55),
1203
- coreml: {
1204
- url: hf("clip/mobileclip-s1/coreml/camstack-mobileclip-s1-vision.mlpackage"),
1205
- sizeMB: 65,
1206
- isDirectory: true,
1207
- files: [...MLPACKAGE_FILES],
1208
- runtimes: ["python"]
1213
+ {
1214
+ id: "mobileclip-s2",
1215
+ name: "MobileCLIP S2",
1216
+ description: "MobileCLIP S2 — Apple high-accuracy CLIP vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
1217
+ inputSize: {
1218
+ width: 256,
1219
+ height: 256
1220
+ },
1221
+ labels: [{
1222
+ id: "embedding",
1223
+ name: "CLIP Embedding"
1224
+ }],
1225
+ preprocessMode: "resize",
1226
+ inputNormalization: "none",
1227
+ formats: {
1228
+ openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-vision.xml"), 90),
1229
+ coreml: {
1230
+ url: hf("clip/mobileclip-s2/coreml/camstack-mobileclip-s2-vision.mlpackage"),
1231
+ sizeMB: 110,
1232
+ isDirectory: true,
1233
+ files: [...MLPACKAGE_FILES],
1234
+ runtimes: ["python"]
1235
+ }
1209
1236
  }
1210
- }
1211
- }, {
1212
- id: "mobileclip-s2",
1213
- name: "MobileCLIP S2",
1214
- description: "MobileCLIP S2 — Apple high-accuracy CLIP vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
1215
- inputSize: {
1216
- width: 256,
1217
- height: 256
1218
1237
  },
1219
- labels: [{
1220
- id: "embedding",
1221
- name: "CLIP Embedding"
1222
- }],
1223
- preprocessMode: "resize",
1224
- inputNormalization: "none",
1225
- formats: {
1226
- openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-vision.xml"), 90),
1227
- coreml: {
1228
- url: hf("clip/mobileclip-s2/coreml/camstack-mobileclip-s2-vision.mlpackage"),
1229
- sizeMB: 110,
1230
- isDirectory: true,
1231
- files: [...MLPACKAGE_FILES],
1232
- runtimes: ["python"]
1238
+ {
1239
+ id: "siglip2-b16-224",
1240
+ name: "SigLIP2 B/16",
1241
+ description: "Google SigLIP2 base, patch 16, 224×224 — Apache-2.0 CLIP vision encoder, 768-dim (fp16 OpenVINO/CoreML)",
1242
+ inputSize: {
1243
+ width: 224,
1244
+ height: 224
1245
+ },
1246
+ labels: [{
1247
+ id: "embedding",
1248
+ name: "CLIP Embedding"
1249
+ }],
1250
+ preprocessMode: "resize",
1251
+ inputNormalization: "none",
1252
+ license: "Apache-2.0",
1253
+ formats: {
1254
+ openvino: ovFormat(hf("clip/siglip2/openvino/camstack-siglip2-b16-224-vision.xml"), 186),
1255
+ coreml: {
1256
+ url: hf("clip/siglip2/coreml/camstack-siglip2-b16-224-vision.mlpackage"),
1257
+ sizeMB: 185,
1258
+ isDirectory: true,
1259
+ files: [...MLPACKAGE_FILES],
1260
+ runtimes: ["python"]
1261
+ }
1233
1262
  }
1234
1263
  }
1235
- }];
1264
+ ];
1236
1265
  var AUDIO_CLASSIFIER_MODELS = [{
1237
1266
  id: "yamnet-onnx",
1238
1267
  name: "YAMNet",
@@ -1671,7 +1700,7 @@ var STEP_FACE_EMBEDDING = new FaceEmbeddingStep({
1671
1700
  outputClasses: ["identity"],
1672
1701
  labelTier: 2,
1673
1702
  models: [...FACE_EMBEDDING_MODELS],
1674
- defaultModelId: "arcface-r100",
1703
+ defaultModelId: "auraface-r100",
1675
1704
  modelScope: "cluster",
1676
1705
  defaultConfidence: 0,
1677
1706
  cadence: {
@@ -1,4 +1,4 @@
1
- const require_dist = require("./dist-BsVcf5wO.js");
1
+ const require_dist = require("./dist-8up-f2TX.js");
2
2
  let node_crypto = require("node:crypto");
3
3
  //#region src/detection-pipeline/pipeline/landmark-precision-gate.ts
4
4
  /**
@@ -1184,55 +1184,84 @@ var INSTANCE_SEGMENTATION_MODELS = [
1184
1184
  }
1185
1185
  }
1186
1186
  ];
1187
- var CLIP_EMBEDDING_MODELS = [{
1188
- id: "mobileclip-s1",
1189
- name: "MobileCLIP S1",
1190
- description: "MobileCLIP S1 — Apple balanced CLIP vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
1191
- inputSize: {
1192
- width: 256,
1193
- height: 256
1187
+ var CLIP_EMBEDDING_MODELS = [
1188
+ {
1189
+ id: "mobileclip-s1",
1190
+ name: "MobileCLIP S1",
1191
+ description: "MobileCLIP S1 — Apple balanced CLIP vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
1192
+ inputSize: {
1193
+ width: 256,
1194
+ height: 256
1195
+ },
1196
+ labels: [{
1197
+ id: "embedding",
1198
+ name: "CLIP Embedding"
1199
+ }],
1200
+ preprocessMode: "resize",
1201
+ inputNormalization: "none",
1202
+ formats: {
1203
+ openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-vision.xml"), 55),
1204
+ coreml: {
1205
+ url: hf("clip/mobileclip-s1/coreml/camstack-mobileclip-s1-vision.mlpackage"),
1206
+ sizeMB: 65,
1207
+ isDirectory: true,
1208
+ files: [...MLPACKAGE_FILES],
1209
+ runtimes: ["python"]
1210
+ }
1211
+ }
1194
1212
  },
1195
- labels: [{
1196
- id: "embedding",
1197
- name: "CLIP Embedding"
1198
- }],
1199
- preprocessMode: "resize",
1200
- inputNormalization: "none",
1201
- formats: {
1202
- openvino: ovFormat(hf("clip/mobileclip-s1/openvino/camstack-mobileclip-s1-vision.xml"), 55),
1203
- coreml: {
1204
- url: hf("clip/mobileclip-s1/coreml/camstack-mobileclip-s1-vision.mlpackage"),
1205
- sizeMB: 65,
1206
- isDirectory: true,
1207
- files: [...MLPACKAGE_FILES],
1208
- runtimes: ["python"]
1213
+ {
1214
+ id: "mobileclip-s2",
1215
+ name: "MobileCLIP S2",
1216
+ description: "MobileCLIP S2 — Apple high-accuracy CLIP vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
1217
+ inputSize: {
1218
+ width: 256,
1219
+ height: 256
1220
+ },
1221
+ labels: [{
1222
+ id: "embedding",
1223
+ name: "CLIP Embedding"
1224
+ }],
1225
+ preprocessMode: "resize",
1226
+ inputNormalization: "none",
1227
+ formats: {
1228
+ openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-vision.xml"), 90),
1229
+ coreml: {
1230
+ url: hf("clip/mobileclip-s2/coreml/camstack-mobileclip-s2-vision.mlpackage"),
1231
+ sizeMB: 110,
1232
+ isDirectory: true,
1233
+ files: [...MLPACKAGE_FILES],
1234
+ runtimes: ["python"]
1235
+ }
1209
1236
  }
1210
- }
1211
- }, {
1212
- id: "mobileclip-s2",
1213
- name: "MobileCLIP S2",
1214
- description: "MobileCLIP S2 — Apple high-accuracy CLIP vision encoder, 512-dim, 256×256 (fp16 OpenVINO/CoreML)",
1215
- inputSize: {
1216
- width: 256,
1217
- height: 256
1218
1237
  },
1219
- labels: [{
1220
- id: "embedding",
1221
- name: "CLIP Embedding"
1222
- }],
1223
- preprocessMode: "resize",
1224
- inputNormalization: "none",
1225
- formats: {
1226
- openvino: ovFormat(hf("clip/mobileclip-s2/openvino/camstack-mobileclip-s2-vision.xml"), 90),
1227
- coreml: {
1228
- url: hf("clip/mobileclip-s2/coreml/camstack-mobileclip-s2-vision.mlpackage"),
1229
- sizeMB: 110,
1230
- isDirectory: true,
1231
- files: [...MLPACKAGE_FILES],
1232
- runtimes: ["python"]
1238
+ {
1239
+ id: "siglip2-b16-224",
1240
+ name: "SigLIP2 B/16",
1241
+ description: "Google SigLIP2 base, patch 16, 224×224 — Apache-2.0 CLIP vision encoder, 768-dim (fp16 OpenVINO/CoreML)",
1242
+ inputSize: {
1243
+ width: 224,
1244
+ height: 224
1245
+ },
1246
+ labels: [{
1247
+ id: "embedding",
1248
+ name: "CLIP Embedding"
1249
+ }],
1250
+ preprocessMode: "resize",
1251
+ inputNormalization: "none",
1252
+ license: "Apache-2.0",
1253
+ formats: {
1254
+ openvino: ovFormat(hf("clip/siglip2/openvino/camstack-siglip2-b16-224-vision.xml"), 186),
1255
+ coreml: {
1256
+ url: hf("clip/siglip2/coreml/camstack-siglip2-b16-224-vision.mlpackage"),
1257
+ sizeMB: 185,
1258
+ isDirectory: true,
1259
+ files: [...MLPACKAGE_FILES],
1260
+ runtimes: ["python"]
1261
+ }
1233
1262
  }
1234
1263
  }
1235
- }];
1264
+ ];
1236
1265
  var AUDIO_CLASSIFIER_MODELS = [{
1237
1266
  id: "yamnet-onnx",
1238
1267
  name: "YAMNet",
@@ -1671,7 +1700,7 @@ var STEP_FACE_EMBEDDING = new FaceEmbeddingStep({
1671
1700
  outputClasses: ["identity"],
1672
1701
  labelTier: 2,
1673
1702
  models: [...FACE_EMBEDDING_MODELS],
1674
- defaultModelId: "arcface-r100",
1703
+ defaultModelId: "auraface-r100",
1675
1704
  modelScope: "cluster",
1676
1705
  defaultConfidence: 0,
1677
1706
  cadence: {