pulsevad 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pulsevad/__init__.py ADDED
@@ -0,0 +1,24 @@
1
+ """PulseVAD: ultra-tiny 2.1k streaming VAD for microcontrollers and edge devices."""
2
+
3
+ from pulsevad.model import PulseVAD
4
+ from pulsevad.frontend import MelFrontend
5
+ from pulsevad.utils_vad import (
6
+ load_pulsevad,
7
+ read_audio,
8
+ predict_window,
9
+ get_speech_timestamps,
10
+ get_model_path,
11
+ )
12
+
13
+ __version__ = "0.1.0"
14
+
15
+ __all__ = [
16
+ "PulseVAD",
17
+ "MelFrontend",
18
+ "load_pulsevad",
19
+ "read_audio",
20
+ "predict_window",
21
+ "get_speech_timestamps",
22
+ "get_model_path",
23
+ "__version__",
24
+ ]
pulsevad/augment.py ADDED
@@ -0,0 +1,77 @@
1
+ """Acoustic augmentation (spec phase-03 §2.1, §3.3): SNR mixing math, synthetic
2
+ wind noise (Mirabilii-style proxy), and pyroomacoustics room reverb.
3
+ """
4
+
5
+ import numpy as np
6
+ from scipy.signal import butter, fftconvolve, sosfilt
7
+
8
+ SR = 16_000
9
+
10
+
11
+ def rms(x: np.ndarray) -> float:
12
+ return float(np.sqrt(np.mean(np.square(x), dtype=np.float64)))
13
+
14
+
15
+ def mix_at_snr(speech: np.ndarray, noise: np.ndarray, snr_db: float) -> np.ndarray:
16
+ """x = s + g*w with g set so RMS(s)/RMS(g*w) = 10^(snr/20); anti-clip guard."""
17
+ gain = rms(speech) / (rms(noise) + 1e-8) * 10.0 ** (-snr_db / 20.0)
18
+ mixed = speech + gain * noise
19
+ peak = float(np.max(np.abs(mixed)))
20
+ if peak > 1.0:
21
+ mixed = mixed / (peak + 1e-5)
22
+ return mixed
23
+
24
+
25
+ def pink_noise(n: int, rng: np.random.Generator) -> np.ndarray:
26
+ """1/f power noise via FFT spectrum weighting (amplitude ~ 1/sqrt(f))."""
27
+ white = rng.standard_normal(n)
28
+ spec = np.fft.rfft(white)
29
+ freqs = np.fft.rfftfreq(n)
30
+ freqs[0] = freqs[1] # avoid div-by-zero on DC
31
+ shaped = spec / np.sqrt(freqs)
32
+ p = np.fft.irfft(shaped, n)
33
+ return p / (rms(p) + 1e-8)
34
+
35
+
36
+ def wind_noise(n: int, sr: int = SR, rng: np.random.Generator | None = None) -> np.ndarray:
37
+ """Synthetic airflow wind: pink noise, 500 Hz Butterworth low-pass, slow
38
+ time-varying gust envelope. Mirabilii et al. 2022-style proxy."""
39
+ rng = rng or np.random.default_rng()
40
+ x = sosfilt(butter(4, 500, btype="low", fs=sr, output="sos"), pink_noise(n, rng))
41
+ env = sosfilt(butter(2, 0.5, btype="low", fs=sr, output="sos"), rng.standard_normal(n))
42
+ env = env / (rms(env) + 1e-8)
43
+ x = x * (1.0 + 0.5 * env)
44
+ return x / (rms(x) + 1e-8)
45
+
46
+
47
+ def simulate_rir(sr: int = SR, rng: np.random.Generator | None = None) -> np.ndarray:
48
+ """Random ShoeBox RIR: L in [3,8] m, W in [3,6], H in [2.5,4], T60 in [0.15,0.5]."""
49
+ import pyroomacoustics as pra
50
+
51
+ rng = rng or np.random.default_rng()
52
+ room_dim = [rng.uniform(3, 8), rng.uniform(3, 6), rng.uniform(2.5, 4)]
53
+ t60 = rng.uniform(0.15, 0.5)
54
+ e_abs, max_order = pra.inverse_sabine(t60, room_dim)
55
+ # max_order can be huge/invalid for small rooms + long T60 without ray tracing
56
+ max_order = min(max(int(max_order), 1), 50)
57
+ room = pra.ShoeBox(
58
+ room_dim, fs=sr, materials=pra.Material(e_abs), max_order=max_order
59
+ )
60
+ for _ in range(5): # place source/mic with >=0.5 m separation
61
+ src = [rng.uniform(0.3, d - 0.3) for d in room_dim]
62
+ mic = [rng.uniform(0.3, d - 0.3) for d in room_dim]
63
+ if np.linalg.norm(np.array(src) - np.array(mic)) >= 0.5:
64
+ break
65
+ room.add_source(src)
66
+ room.add_microphone(mic)
67
+ room.compute_rir() # RIR-only: simulate() needs source signals
68
+ # ponytail: cap the image-source order — full tail for T60=0.5s small rooms
69
+ # needs ray tracing; 50 images gives a correct-sounding tail in ~3ms less.
70
+ rir = np.asarray(room.rir[0][0], dtype=np.float32)
71
+ return rir / (np.max(np.abs(rir)) + 1e-8)
72
+
73
+
74
+ def reverb_apply(speech: np.ndarray, rir: np.ndarray) -> np.ndarray:
75
+ """Convolve with an RIR, keep the input length, restore input RMS."""
76
+ y = fftconvolve(speech, rir)[: len(speech)]
77
+ return y * (rms(speech) / (rms(y) + 1e-8))
@@ -0,0 +1,417 @@
1
+ """Feature cache builder (spec phase-03 §3.4 + build plan §4.2).
2
+
3
+ Windows: 200 ms (3200 samples), hop 100 ms (50% overlap).
4
+ Per-window mix (paper-faithful, config libri_dns_full_no_pure_noise_v2):
5
+ 25% clean / 25% wind @ -5 dB / 50% DNS+MUSAN noise @ SNR in {-10,-5,0,+5,+10} dB,
6
+ reverb on 50% of noisy windows (pre-generated RIR pool), 0% pure-noise files.
7
+ Labels: majority vote over the 21 frame centers (>=11 speech frames -> 1).
8
+ """
9
+
10
+ import json
11
+ from collections import OrderedDict
12
+ from pathlib import Path
13
+
14
+ import numpy as np
15
+ import soundfile as sf
16
+ import torch
17
+
18
+ from pulsevad.augment import (
19
+ mix_at_snr,
20
+ reverb_apply,
21
+ simulate_rir,
22
+ wind_noise,
23
+ )
24
+ from pulsevad.frontend import MelFrontend
25
+ from pulsevad.label_corpus import AUDIO_EXTS, read_mono, segments_to_frame_flags
26
+
27
+ WINDOW_SAMPLES = 3_200
28
+ HOP_SAMPLES = 1_600
29
+ FRAME_SEC = 0.01
30
+ FRAMES_PER_WINDOW = 21
31
+ STFT_FRAME_SEC = 0.01 # STFT center=True frame k sits at sample k*160
32
+
33
+ WIND_SNR_DB = -5.0
34
+ NOISE_SNRS_DB = [-10.0, -5.0, 0.0, 5.0, 10.0]
35
+
36
+
37
+ def majority_label(frame_flags: np.ndarray) -> int:
38
+ """1 iff strictly more than half of the window's frames are speech."""
39
+ return int(frame_flags.sum() * 2 > len(frame_flags))
40
+
41
+
42
+ def file_frame_flags(segments: list, n_samples: int, sr: int = 16_000) -> np.ndarray:
43
+ n_frames = int(n_samples / (sr * FRAME_SEC)) + 1
44
+ return segments_to_frame_flags(segments, n_frames, FRAME_SEC)
45
+
46
+
47
+ def window_label(flags: np.ndarray, start: int) -> int:
48
+ """Window starting at sample `start` (multiple of hop): its 21 STFT frame
49
+ centers sit at samples start + j*160 -> flag indices start//160 + j."""
50
+ first = start // 160
51
+ return majority_label(flags[first : first + FRAMES_PER_WINDOW])
52
+
53
+
54
+ def _count_windows(n_samples: int) -> int:
55
+ return max(0, (n_samples - WINDOW_SAMPLES) // HOP_SAMPLES + 1)
56
+
57
+
58
+ def build_noise_pool(noise_dirs: list) -> list:
59
+ """[(path, n_frames)] for 16 kHz wav files long enough to cover a window."""
60
+ pool = []
61
+ for d in noise_dirs:
62
+ d = Path(d)
63
+ if not d.exists():
64
+ continue
65
+ for p in sorted(d.rglob("*.wav")):
66
+ info = sf.info(p)
67
+ if info.samplerate == 16_000 and info.frames >= WINDOW_SAMPLES:
68
+ pool.append([str(p), info.frames])
69
+ if not pool:
70
+ raise FileNotFoundError(f"no usable 16 kHz noise wavs under {noise_dirs}")
71
+ return pool
72
+
73
+
74
+ def load_noise_window(pool: list, rng: np.random.Generator) -> np.ndarray:
75
+ """One-shot loader (tests / low-volume use). The cache builders use
76
+ NoiseReader instead — a fresh handle per window is ruinous over NFS."""
77
+ return NoiseReader(pool).load_window(rng)
78
+
79
+
80
+ class NoiseReader:
81
+ """Random noise-window reader with an LRU cache of open file handles.
82
+ ponytail: without handle reuse, cache building pays ~1.6M open/seek/close
83
+ round-trips on the volume; with 64 cached handles the hit rate is near 100%.
84
+ Raise cache size if the noise pool grows past a few hundred files.
85
+ """
86
+
87
+ def __init__(self, pool: list, cache_size: int = 64) -> None:
88
+ self.pool = pool
89
+ self.cache_size = cache_size
90
+ self._handles: OrderedDict[str, sf.SoundFile] = OrderedDict()
91
+
92
+ def _handle(self, path: str) -> sf.SoundFile:
93
+ h = self._handles.get(path)
94
+ if h is None:
95
+ if len(self._handles) >= self.cache_size:
96
+ _, old = self._handles.popitem(last=False) # evict LRU
97
+ old.close()
98
+ h = sf.SoundFile(path)
99
+ self._handles[path] = h
100
+ else:
101
+ self._handles.move_to_end(path)
102
+ return h
103
+
104
+ def load_window(self, rng: np.random.Generator) -> np.ndarray:
105
+ path, n_frames = self.pool[rng.integers(len(self.pool))]
106
+ start = int(rng.integers(0, n_frames - WINDOW_SAMPLES + 1))
107
+ f = self._handle(path)
108
+ f.seek(start)
109
+ noise = f.read(WINDOW_SAMPLES, dtype="float32", always_2d=True)[:, 0]
110
+ if len(noise) < WINDOW_SAMPLES: # defensive: file truncated after scan
111
+ reps = WINDOW_SAMPLES // len(noise) + 1
112
+ noise = np.tile(noise, reps)[:WINDOW_SAMPLES]
113
+ return noise
114
+
115
+
116
+ def draw_category(rng: np.random.Generator) -> str:
117
+ r = rng.random()
118
+ return "clean" if r < 0.25 else "wind" if r < 0.5 else "noise"
119
+
120
+
121
+ def augment_window(
122
+ seg: np.ndarray,
123
+ category: str,
124
+ rng: np.random.Generator,
125
+ noise_reader: NoiseReader,
126
+ rir_pool: list,
127
+ snr_db: float | None = None,
128
+ ) -> np.ndarray:
129
+ """Reverb 50% of noisy windows, then mix wind or corpus noise. `snr_db`
130
+ overrides the uniform noise SNR draw (used by the eval-set builders)."""
131
+ if category == "clean":
132
+ return seg
133
+ if category != "pure_noise" and rng.random() < 0.5:
134
+ seg = reverb_apply(seg, rir_pool[rng.integers(len(rir_pool))])
135
+ if category == "wind":
136
+ return mix_at_snr(seg, wind_noise(WINDOW_SAMPLES, rng=rng), WIND_SNR_DB)
137
+ noise = noise_reader.load_window(rng)
138
+ if snr_db is None:
139
+ snr_db = NOISE_SNRS_DB[rng.integers(len(NOISE_SNRS_DB))]
140
+ return mix_at_snr(seg, noise, snr_db)
141
+
142
+
143
+ def _speech_files(speech_dir: Path, labels_dir: Path) -> list:
144
+ files = []
145
+ for p in sorted(Path(speech_dir).rglob("*")):
146
+ if p.suffix.lower() in AUDIO_EXTS:
147
+ lj = Path(labels_dir) / p.relative_to(speech_dir).with_suffix(".json")
148
+ if lj.exists():
149
+ files.append((str(p), str(lj)))
150
+ if not files:
151
+ n_audio = sum(
152
+ 1
153
+ for p in Path(speech_dir).rglob("*")
154
+ if p.suffix.lower() in AUDIO_EXTS
155
+ )
156
+ n_json = sum(1 for _ in Path(labels_dir).rglob("*.json"))
157
+ raise FileNotFoundError(
158
+ f"no labeled speech files: found {n_audio} audio files under "
159
+ f"{speech_dir} but only {n_json} label manifests under {labels_dir}. "
160
+ f"If {n_json} == 0, run ::label; if both counts are >0, check the "
161
+ f"manifest sub-path layout mirrors the speech tree."
162
+ )
163
+ return files
164
+
165
+
166
+ def _frontend_batch(frontend: MelFrontend, windows: list[np.ndarray]) -> np.ndarray:
167
+ x = torch.from_numpy(
168
+ np.stack(windows).astype(np.float32, copy=False)
169
+ ) # scipy reverb/mixing returns float64; the frontend is float32
170
+ with torch.no_grad():
171
+ return frontend(x).numpy().astype(np.float32) # (W, 64, 21)
172
+
173
+
174
+ _W = {} # per-process worker state (noise reader, RIR pool, frontend)
175
+
176
+
177
+ def _worker_init(noise_dirs, rir_pool_size, base_seed, noise_only_frac,
178
+ max_windows_per_file):
179
+ """ProcessPoolExecutor initializer: build the heavy per-process state once."""
180
+ _W["reader"] = NoiseReader(build_noise_pool(noise_dirs))
181
+ rng = np.random.default_rng(base_seed)
182
+ _W["rir_pool"] = [simulate_rir(rng=rng) for _ in range(rir_pool_size)]
183
+ _W["frontend"] = MelFrontend()
184
+ _W["noise_only_frac"] = noise_only_frac
185
+ _W["max_windows_per_file"] = max_windows_per_file
186
+
187
+
188
+ def _process_file(item):
189
+ """Featurize one file -> (offset, n, features, labels, cat_counts).
190
+
191
+ Per-file RNG (base_seed + 7919*file_idx) keeps results deterministic
192
+ regardless of process scheduling.
193
+ """
194
+ file_idx, path, lj, n_win, offset, base_seed = item
195
+ rng = np.random.default_rng(base_seed + 7919 * file_idx)
196
+ reader, rir_pool = _W["reader"], _W["rir_pool"]
197
+ wav = read_mono(Path(path))
198
+ seg_manifest = json.loads(Path(lj).read_text())
199
+ flags = file_frame_flags(seg_manifest["segments"], len(wav))
200
+
201
+ starts = list(range(0, len(wav) - WINDOW_SAMPLES + 1, HOP_SAMPLES))
202
+ if not starts:
203
+ return offset, 0, None, None, None
204
+ mwpf = _W["max_windows_per_file"]
205
+ if mwpf is not None and len(starts) > mwpf:
206
+ step = len(starts) / mwpf
207
+ starts = [starts[int(k * step)] for k in range(mwpf)]
208
+
209
+ windows, win_labels = [], []
210
+ cat_counts = {"clean": 0, "wind": 0, "noise": 0}
211
+ for s0 in starts:
212
+ if rng.random() < _W["noise_only_frac"]:
213
+ # Pure-noise window labeled 0: without these, the model never sees
214
+ # noise without speech (silence-window SNR mixing zeroes the noise
215
+ # out) and scores pure noise at ~0.5 — failing the phase-07
216
+ # pure-noise FPR < 5% gate.
217
+ windows.append(reader.load_window(rng))
218
+ win_labels.append(0)
219
+ cat_counts["noise"] += 1
220
+ continue
221
+ category = draw_category(rng)
222
+ seg = augment_window(
223
+ wav[s0 : s0 + WINDOW_SAMPLES].astype(np.float32).copy(),
224
+ category, rng, reader, rir_pool,
225
+ )
226
+ cat_counts[category] += 1
227
+ windows.append(seg)
228
+ win_labels.append(window_label(flags, s0))
229
+
230
+ feats = _frontend_batch(_W["frontend"], windows)
231
+ return offset, len(windows), feats, np.array(win_labels, dtype=np.uint8), cat_counts
232
+
233
+
234
+ def _max_starts(starts, max_windows_per_file=None):
235
+ if max_windows_per_file is not None and len(starts) > max_windows_per_file:
236
+ step = len(starts) / max_windows_per_file
237
+ return [starts[int(k * step)] for k in range(max_windows_per_file)]
238
+ return starts
239
+
240
+
241
+ def build_cache(
242
+ speech_dir,
243
+ labels_dir,
244
+ noise_dirs: list,
245
+ out_dir,
246
+ val_frac: float = 0.05,
247
+ seed: int = 0,
248
+ rir_pool_size: int = 200,
249
+ max_windows_per_file: int | None = None,
250
+ noise_only_frac: float = 0.10,
251
+ workers: int = 8,
252
+ ) -> dict:
253
+ """Two-pass build: count windows from label durations, pre-allocate memmaps,
254
+ fill via a process pool (one task per file). Returns the manifest."""
255
+ from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_completed
256
+ from multiprocessing import cpu_count
257
+
258
+ rng = np.random.default_rng(seed)
259
+ out_dir = Path(out_dir)
260
+ out_dir.mkdir(parents=True, exist_ok=True)
261
+
262
+ files = _speech_files(speech_dir, labels_dir)
263
+ rng.shuffle(files)
264
+ n_val = max(1, int(len(files) * val_frac)) if len(files) > 20 else 0
265
+ splits = {"train": files[n_val:], "val": files[:n_val]}
266
+
267
+ manifest = {
268
+ "window_samples": WINDOW_SAMPLES,
269
+ "hop_samples": HOP_SAMPLES,
270
+ "seed": seed,
271
+ "workers": workers,
272
+ "splits": {},
273
+ }
274
+
275
+ for tag, split_files in splits.items():
276
+ base_seed = seed + (1 if tag == "val" else 0)
277
+ print(f"[build:{tag}] reading {len(split_files)} label manifests…", flush=True)
278
+ with ThreadPoolExecutor(max_workers=32) as ex:
279
+ durations = list(
280
+ ex.map(
281
+ lambda lj: json.loads(Path(lj).read_text())["duration_sec"],
282
+ [lj for _, lj in split_files],
283
+ )
284
+ )
285
+ counts = [_count_windows(int(d * 16_000)) for d in durations]
286
+ if max_windows_per_file is not None:
287
+ counts = [min(c, max_windows_per_file) for c in counts]
288
+ n_total = sum(counts)
289
+
290
+ feats = np.lib.format.open_memmap(
291
+ out_dir / f"{tag}_features.npy", mode="w+",
292
+ dtype=np.float32, shape=(n_total, 64, 21),
293
+ )
294
+ labels = np.lib.format.open_memmap(
295
+ out_dir / f"{tag}_labels.npy", mode="w+", dtype=np.uint8, shape=(n_total,)
296
+ )
297
+
298
+ # work items with precomputed memmap offsets (prefix sums)
299
+ items, offsets, done = [], [], 0
300
+ for file_idx, ((path, lj), n_win) in enumerate(zip(split_files, counts)):
301
+ offsets.append(done)
302
+ done += n_win
303
+ if n_win:
304
+ items.append((file_idx, path, lj, n_win, offsets[-1], base_seed))
305
+
306
+ n_workers = max(1, min(workers, cpu_count() or workers))
307
+ print(f"[build:{tag}] {n_total} windows over {len(items)} files "
308
+ f"on {n_workers} workers", flush=True)
309
+ cat_counts = {"clean": 0, "wind": 0, "noise": 0}
310
+ written = 0
311
+ with ProcessPoolExecutor(
312
+ max_workers=n_workers,
313
+ initializer=_worker_init,
314
+ initargs=(noise_dirs, rir_pool_size, base_seed, noise_only_frac,
315
+ max_windows_per_file),
316
+ ) as pool:
317
+ futs = {pool.submit(_process_file, it): it[4] for it in items}
318
+ done = 0
319
+ for fut in as_completed(futs):
320
+ offset, n, f, lab, cats = fut.result()
321
+ if n:
322
+ feats[offset:offset + n] = f
323
+ labels[offset:offset + n] = lab
324
+ for k in cat_counts:
325
+ cat_counts[k] += cats[k]
326
+ written += n
327
+ done += 1
328
+ if done % 500 == 0:
329
+ print(f"[build:{tag}] {done}/{len(items)} files, "
330
+ f"{written}/{n_total} windows "
331
+ f"({100 * written / max(n_total, 1):.1f}%)", flush=True)
332
+
333
+ manifest["splits"][tag] = {
334
+ "features": f"{tag}_features.npy",
335
+ "labels": f"{tag}_labels.npy",
336
+ "n": n_total,
337
+ "hours": round(n_total * HOP_SAMPLES / 16_000 / 3600, 2),
338
+ "categories": cat_counts,
339
+ }
340
+ print(f"[{tag}] {n_total} windows ({cat_counts})", flush=True)
341
+
342
+ (out_dir / "manifest.json").write_text(json.dumps(manifest, indent=2))
343
+ return manifest
344
+
345
+
346
+ EVAL_CATEGORIES = ["clean", "windy", "dns_synthetic", "speech_noise", "pure_noise"]
347
+
348
+
349
+ def build_eval_sets(
350
+ speech_dir,
351
+ labels_dir,
352
+ noise_dirs: list,
353
+ out_dir,
354
+ n_windows: int = 2_000,
355
+ seed: int = 123,
356
+ rir_pool_size: int = 50,
357
+ save_audio: bool = False,
358
+ ) -> dict:
359
+ """Five held-out categories (build plan §8.2), drawn from the full corpus.
360
+
361
+ ponytail: 'speech_noise' is a proxy for the DNS Speech+Noise category —
362
+ real multi-talker DNS recordings aren't downloaded, so we mix speech with
363
+ DNS/MUSAN noise at close-range SNRs {-5,0,+5}. 'pure_noise' has all-zero
364
+ labels and gates the false-positive check (<5%) in Phase 7.
365
+ """
366
+ rng = np.random.default_rng(seed)
367
+ out_dir = Path(out_dir)
368
+ out_dir.mkdir(parents=True, exist_ok=True)
369
+
370
+ files = _speech_files(speech_dir, labels_dir)
371
+ rng.shuffle(files)
372
+ noise_reader = NoiseReader(build_noise_pool(noise_dirs))
373
+ print(f"[eval] noise pool ready; simulating {rir_pool_size} RIRs…", flush=True)
374
+ rir_pool = [simulate_rir(rng=rng) for _ in range(rir_pool_size)]
375
+ frontend = MelFrontend()
376
+
377
+ summary = {}
378
+ for cat in EVAL_CATEGORIES:
379
+ feats, labels = [], []
380
+ while len(feats) < n_windows:
381
+ path, lj = files[rng.integers(len(files))]
382
+ wav = read_mono(Path(path))
383
+ seg_manifest = json.loads(Path(lj).read_text())
384
+ flags = file_frame_flags(seg_manifest["segments"], len(wav))
385
+ starts = list(range(0, len(wav) - WINDOW_SAMPLES + 1, HOP_SAMPLES))
386
+ if not starts:
387
+ continue
388
+ s0 = starts[rng.integers(len(starts))]
389
+
390
+ if cat == "pure_noise":
391
+ seg = noise_reader.load_window(rng)
392
+ label = 0
393
+ else:
394
+ seg = augment_window(
395
+ wav[s0 : s0 + WINDOW_SAMPLES].astype(np.float32).copy(),
396
+ {"windy": "wind", "dns_synthetic": "noise",
397
+ "speech_noise": "noise"}.get(cat, "clean"),
398
+ rng, noise_reader, rir_pool,
399
+ snr_db=NOISE_SNRS_DB[1:4][rng.integers(3)] if cat == "speech_noise" else None,
400
+ )
401
+ label = window_label(flags, s0)
402
+ feats.append(seg)
403
+ labels.append(label)
404
+
405
+ f = _frontend_batch(frontend, feats)
406
+ np.save(out_dir / f"eval_{cat}_features.npy", f)
407
+ np.save(out_dir / f"eval_{cat}_labels.npy", np.array(labels, dtype=np.uint8))
408
+ if save_audio:
409
+ # raw mixed audio so competitor VADs (Silero, MarbleNet) can score
410
+ # the SAME windows we score — the apples-to-apples phase-7 eval
411
+ np.save(out_dir / f"eval_{cat}_audio.npy",
412
+ np.stack(feats).astype(np.float32))
413
+ summary[cat] = {"n": n_windows, "speech_fraction": float(np.mean(labels))}
414
+ print(f"[eval:{cat}] {n_windows} windows, speech fraction {summary[cat]['speech_fraction']:.3f}")
415
+
416
+ (out_dir / "eval_manifest.json").write_text(json.dumps(summary, indent=2))
417
+ return summary
@@ -0,0 +1 @@
1
+ """PulseVAD pre-trained model weights and embedded headers."""
Binary file
Binary file
Binary file
Binary file
Binary file
Binary file