omnius 1.0.638 → 1.0.640

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,9 +6,11 @@ WAV paths. It never records audio, installs dependencies, downloads models,
6
6
  or enables CUDA. The parent runtime checksums the pinned ONNX artifact before
7
7
  this worker can start.
8
8
 
9
- The fbank path is deliberately the official WeSpeaker ONNX preprocessing:
10
- torchaudio.compliance.kaldi.fbank with 80 bins, 25 ms / 10 ms Hamming frames,
11
- dither=0 and utterance CMN (without CVN).
9
+ The fbank path is a pinned, CPU-only NumPy implementation of the Kaldi
10
+ configuration used by WeSpeaker's ONNX example: 80 bins, 25 ms / 10 ms Hamming
11
+ frames, dither=0 and utterance CMN (without CVN). It intentionally imports
12
+ neither Torch nor Torchaudio, so an upstream Torchaudio binary can never bind
13
+ against or replace JetPack's vendor Torch ABI.
12
14
  """
13
15
  from __future__ import annotations
14
16
 
@@ -22,8 +24,8 @@ import traceback
22
24
  import wave
23
25
 
24
26
 
25
- # Set before importing Torch or ONNX Runtime so a globally configured Jetson
26
- # CUDA device cannot become an accidental provider for speaker embeddings.
27
+ # Set before importing ONNX Runtime so a globally configured Jetson CUDA device
28
+ # cannot become an accidental provider for speaker embeddings.
27
29
  os.environ["CUDA_VISIBLE_DEVICES"] = ""
28
30
 
29
31
  MODEL_NAME = "wespeaker-voxceleb-campplus"
@@ -34,6 +36,11 @@ MAX_DURATION_SECONDS = 30.0
34
36
  MIN_RMS = 0.003
35
37
  MIN_PEAK = 0.01
36
38
  MAX_CLIPPED_FRACTION = 0.01
39
+ FBANK_IMPLEMENTATION = "kaldi-fbank-numpy-v1"
40
+ # SHA-256 over the rounded deterministic 2-second calibration fbank after
41
+ # utterance CMN. This protects the source implementation from silent feature
42
+ # drift while avoiding irrelevant last-bit FFT differences across CPU builds.
43
+ FBANK_VALIDATION_SIGNATURE = "sha256:19bb5552b8068d5cd2ccdbb07ac7a7ed1137400d3bdc3c54fb9c21d4f06e16fa"
37
44
 
38
45
 
39
46
  def emit(payload: dict) -> None:
@@ -98,18 +105,93 @@ def quality_gate(samples, metrics):
98
105
  return "usable"
99
106
 
100
107
 
108
+ def kaldi_fbank80_numpy(waveform, np):
109
+ """Kaldi-compatible fbank for the fixed WeSpeaker CAM++ ONNX contract.
110
+
111
+ This mirrors WeSpeaker's published Kaldi fbank configuration: mono 16k
112
+ input, snip_edges=True,
113
+ remove_dc_offset=True, preemphasis=0.97, round_to_power_of_two=True,
114
+ 80 triangular 20Hz..Nyquist mel bins, Hamming window and log power. The
115
+ caller passes PCM-scaled float32 samples (not normalized [-1, 1] values).
116
+ """
117
+ dtype = np.float32
118
+ waveform = np.ascontiguousarray(waveform, dtype=dtype)
119
+ frame_size = 400
120
+ frame_shift = 160
121
+ fft_size = 512
122
+ if waveform.ndim != 1 or waveform.size < frame_size:
123
+ raise RuntimeError("Kaldi fbank requires at least one 25 ms mono frame")
124
+ frame_count = 1 + (waveform.size - frame_size) // frame_shift
125
+ frames = np.lib.stride_tricks.as_strided(
126
+ waveform,
127
+ shape=(frame_count, frame_size),
128
+ strides=(frame_shift * waveform.strides[0], waveform.strides[0]),
129
+ writeable=False,
130
+ ).copy()
131
+ # Published Kaldi fbank frame/window defaults.
132
+ frames -= np.mean(frames, axis=1, dtype=dtype, keepdims=True)
133
+ frames[:, 1:] -= dtype(0.97) * frames[:, :-1]
134
+ frames[:, 0] -= dtype(0.97) * frames[:, 0]
135
+ sample_index = np.arange(frame_size, dtype=dtype)
136
+ hamming = dtype(0.54) - dtype(0.46) * np.cos(
137
+ dtype(2.0 * np.pi) * sample_index / dtype(frame_size - 1)
138
+ )
139
+ frames *= hamming
140
+ padded = np.pad(frames, ((0, 0), (0, fft_size - frame_size)), mode="constant")
141
+ # NumPy's rfft may calculate internally at float64 precision. Cast its
142
+ # power spectrum back to float32 before the Kaldi-style matrix multiply.
143
+ spectrum = np.asarray(np.abs(np.fft.rfft(padded, axis=1)) ** 2, dtype=dtype)
144
+
145
+ nyquist = dtype(8000.0)
146
+ low_hz = dtype(20.0)
147
+ mel = lambda hz: dtype(1127.0) * np.log(dtype(1.0) + hz / dtype(700.0))
148
+ mel_low = mel(low_hz)
149
+ mel_high = mel(nyquist)
150
+ mel_step = (mel_high - mel_low) / dtype(81.0)
151
+ bin_index = np.arange(80, dtype=dtype)[:, None]
152
+ left = mel_low + bin_index * mel_step
153
+ center = mel_low + (bin_index + dtype(1.0)) * mel_step
154
+ right = mel_low + (bin_index + dtype(2.0)) * mel_step
155
+ fft_bin_hz = dtype(16000.0 / fft_size) * np.arange(fft_size // 2, dtype=dtype)[None, :]
156
+ fft_bin_mel = mel(fft_bin_hz)
157
+ up = (fft_bin_mel - left) / (center - left)
158
+ down = (right - fft_bin_mel) / (right - center)
159
+ banks = np.maximum(dtype(0.0), np.minimum(up, down))
160
+ banks = np.pad(banks, ((0, 0), (0, 1)), mode="constant")
161
+ energies = np.matmul(spectrum, np.asarray(banks.T, dtype=dtype))
162
+ return np.log(np.maximum(energies, np.finfo(dtype).eps)).astype(dtype, copy=False)
163
+
164
+
165
+ def validate_kaldi_fbank_numpy(np):
166
+ """Validate the no-Torch preprocessing path before readiness is true."""
167
+ t = np.arange(SAMPLE_RATE * 2, dtype=np.float32) / np.float32(SAMPLE_RATE)
168
+ samples = (
169
+ np.float32(0.075) * np.sin(np.float32(2.0 * np.pi * 220.0) * t)
170
+ + np.float32(0.025) * np.sin(np.float32(2.0 * np.pi * 440.0) * t)
171
+ ).astype(np.float32)
172
+ features = kaldi_fbank80_numpy(samples * np.float32(1 << 15), np)
173
+ features = (features - np.mean(features, axis=0, dtype=np.float32, keepdims=True)).astype(np.float32)
174
+ if features.shape != (198, 80) or not np.all(np.isfinite(features)):
175
+ raise RuntimeError("NumPy Kaldi fbank calibration produced an invalid feature matrix")
176
+ if not np.all(np.abs(np.mean(features, axis=0)) < np.float32(5e-5)):
177
+ raise RuntimeError("NumPy Kaldi fbank calibration failed utterance CMN")
178
+ rounded = np.rint(features * np.float32(10_000)).astype("<i4", copy=False)
179
+ signature = "sha256:" + hashlib.sha256(rounded.tobytes()).hexdigest()
180
+ if signature != FBANK_VALIDATION_SIGNATURE:
181
+ raise RuntimeError(
182
+ "NumPy Kaldi fbank calibration signature mismatch: " + signature
183
+ )
184
+ return signature
185
+
186
+
101
187
  class WeSpeakerCamPlus:
102
188
  def __init__(self, model_path: str):
103
189
  started = time.perf_counter()
104
190
  import numpy as np
105
191
  import onnxruntime as ort
106
- import torch
107
- from torchaudio.compliance import kaldi
108
192
 
109
193
  self.np = np
110
- self.torch = torch
111
- self.kaldi = kaldi
112
- torch.set_num_threads(1)
194
+ self.preprocessing_signature = validate_kaldi_fbank_numpy(np)
113
195
  options = ort.SessionOptions()
114
196
  options.inter_op_num_threads = 1
115
197
  options.intra_op_num_threads = 1
@@ -134,27 +216,14 @@ class WeSpeakerCamPlus:
134
216
  self.model_load_ms = (time.perf_counter() - started) * 1000
135
217
 
136
218
  def features(self, samples):
137
- # Mirrors wespeaker/bin/infer_onnx.py exactly. Its torchaudio.load
138
- # values are normalized PCM16 floats and multiplied by 2**15 before
139
- # Kaldi fbank; samples are decoded PCM16 values in that same range.
140
- waveform = self.torch.from_numpy(
141
- self.np.ascontiguousarray(samples * (1 << 15), dtype=self.np.float32)
142
- ).unsqueeze(0)
143
- mat = self.kaldi.fbank(
144
- waveform,
145
- num_mel_bins=80,
146
- frame_length=25,
147
- frame_shift=10,
148
- dither=0.0,
149
- sample_frequency=SAMPLE_RATE,
150
- window_type="hamming",
151
- use_energy=False,
152
- )
219
+ # WeSpeaker's ONNX inference multiplies normalized PCM values by 2**15
220
+ # before Kaldi fbank. Decode follows the same convention above.
221
+ mat = kaldi_fbank80_numpy(samples * self.np.float32(1 << 15), self.np)
153
222
  if mat.ndim != 2 or mat.shape[0] < 1 or mat.shape[1] != 80:
154
- raise RuntimeError(f"Official WeSpeaker fbank produced an invalid shape: {tuple(mat.shape)}")
223
+ raise RuntimeError(f"NumPy Kaldi fbank produced an invalid shape: {tuple(mat.shape)}")
155
224
  # CMN without CVN, over the full input utterance, per official script.
156
- mat = mat - self.torch.mean(mat, dim=0)
157
- return mat.unsqueeze(0).detach().cpu().numpy().astype(self.np.float32, copy=False)
225
+ mat = mat - self.np.mean(mat, axis=0, dtype=self.np.float32, keepdims=True)
226
+ return mat[None, :, :].astype(self.np.float32, copy=False)
158
227
 
159
228
  def embed(self, samples):
160
229
  started = time.perf_counter()
@@ -186,9 +255,22 @@ class WeSpeakerCamPlus:
186
255
 
187
256
  def main() -> int:
188
257
  parser = argparse.ArgumentParser()
189
- parser.add_argument("--model", required=True)
190
- parser.add_argument("--model-digest", required=True)
258
+ parser.add_argument("--model")
259
+ parser.add_argument("--model-digest")
260
+ parser.add_argument("--preprocessing-probe", action="store_true")
191
261
  args = parser.parse_args()
262
+ if args.preprocessing_probe:
263
+ import numpy as np
264
+ emit(
265
+ {
266
+ "type": "preprocessing_probe",
267
+ "implementation": FBANK_IMPLEMENTATION,
268
+ "validation_signature": validate_kaldi_fbank_numpy(np),
269
+ }
270
+ )
271
+ return 0
272
+ if not args.model or not args.model_digest:
273
+ parser.error("--model and --model-digest are required unless --preprocessing-probe is used")
192
274
  worker = WeSpeakerCamPlus(args.model)
193
275
  worker.warm()
194
276
  emit(
@@ -202,6 +284,8 @@ def main() -> int:
202
284
  "dimension": EMBEDDING_DIMENSION,
203
285
  "model_load_ms": round(worker.model_load_ms, 3),
204
286
  "warmed": True,
287
+ "preprocessing": FBANK_IMPLEMENTATION,
288
+ "preprocessing_signature": worker.preprocessing_signature,
205
289
  }
206
290
  )
207
291
  for line in sys.stdin:
@@ -463,6 +463,38 @@ def amplify(samples):
463
463
  return samples
464
464
  return np.clip(samples * gain, -1.0, 1.0)
465
465
 
466
+
467
+ def pcm16_signal_metrics(samples):
468
+ """Authoritative stats for the raw PCM16 intake before any ASR call."""
469
+ count = int(len(samples))
470
+ if count <= 0:
471
+ return {
472
+ "sampleRateHz": SAMPLE_RATE, "channels": CHANNELS, "bitsPerSample": 16,
473
+ "sampleCount": 0, "durationMs": 0.0, "peakPcm16": 0,
474
+ "rmsPcm16": 0.0, "activeSampleCount": 0, "activeSampleRatio": 0.0,
475
+ }
476
+ # Input originates in int16 stdin. Rounding only restores that exact integer
477
+ # scale for reporting; it never modifies the samples supplied to Whisper.
478
+ pcm = np.rint(samples * 32768.0).astype(np.int32, copy=False)
479
+ magnitude = np.abs(pcm)
480
+ peak = int(np.max(magnitude))
481
+ active = int(np.count_nonzero(magnitude >= 8))
482
+ return {
483
+ "sampleRateHz": SAMPLE_RATE, "channels": CHANNELS, "bitsPerSample": 16,
484
+ "sampleCount": count, "durationMs": count * 1000.0 / SAMPLE_RATE,
485
+ "peakPcm16": peak, "rmsPcm16": float(np.sqrt(np.mean(pcm.astype(np.float64) ** 2))),
486
+ "activeSampleCount": active, "activeSampleRatio": active / count,
487
+ }
488
+
489
+
490
+ def digitally_silent_pcm16(samples):
491
+ metrics = pcm16_signal_metrics(samples)
492
+ return (
493
+ metrics["peakPcm16"] <= 4
494
+ and metrics["rmsPcm16"] <= 1.0
495
+ and metrics["activeSampleCount"] == 0
496
+ ), metrics
497
+
466
498
  # ---------------------------------------------------------------------------
467
499
  # Main transcription loop — utterance FSM
468
500
  # ---------------------------------------------------------------------------
@@ -546,6 +578,11 @@ def main():
546
578
  })
547
579
 
548
580
  def run_transcribe(m, samples):
581
+ # Every live PCM ASR call goes through this exact gate. It is stricter
582
+ # than VAD only for digital silence, so it cannot reject quiet speech.
583
+ silent, _ = digitally_silent_pcm16(samples)
584
+ if silent:
585
+ return {"text": "", "segments": []}
549
586
  return m.transcribe(
550
587
  samples,
551
588
  fp16=fp16,
@@ -644,6 +681,10 @@ def main():
644
681
 
645
682
  def transcribe_final(samples, speech_block_count=0, rms_values=None, peak=0.0):
646
683
  nonlocal last_final_text
684
+ silent, signal = digitally_silent_pcm16(samples)
685
+ if silent:
686
+ emit({"type": "no_speech", "kind": "no_speech", "reason": "digital_silence", "text": "", "signal": signal})
687
+ return
647
688
  if not utterance_quality_ok(samples, speech_block_count, rms_values or [], peak):
648
689
  return
649
690
  samples = amplify(samples)