omnius 1.0.638 → 1.0.640
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +1314 -857
- package/dist/scripts/audio-speaker-embedding-worker.py +115 -31
- package/dist/scripts/live-whisper.py +41 -0
- package/dist/scripts/ocr-advanced.py +273 -140
- package/dist/scripts/transcribe-file.py +61 -6
- package/dist/update-worker.js +0 -1
- package/docs/DISCOVERY.json +26 -6
- package/docs/rest/endpoints/chat.md +5 -1
- package/docs/rest/endpoints/voice-vision.md +26 -23
- package/npm-shrinkwrap.json +2 -2
- package/package.json +1 -1
|
@@ -6,9 +6,11 @@ WAV paths. It never records audio, installs dependencies, downloads models,
|
|
|
6
6
|
or enables CUDA. The parent runtime checksums the pinned ONNX artifact before
|
|
7
7
|
this worker can start.
|
|
8
8
|
|
|
9
|
-
The fbank path is
|
|
10
|
-
|
|
11
|
-
dither=0 and utterance CMN (without CVN).
|
|
9
|
+
The fbank path is a pinned, CPU-only NumPy implementation of the Kaldi
|
|
10
|
+
configuration used by WeSpeaker's ONNX example: 80 bins, 25 ms / 10 ms Hamming
|
|
11
|
+
frames, dither=0 and utterance CMN (without CVN). It intentionally imports
|
|
12
|
+
neither Torch nor Torchaudio, so an upstream Torchaudio binary can never bind
|
|
13
|
+
against or replace JetPack's vendor Torch ABI.
|
|
12
14
|
"""
|
|
13
15
|
from __future__ import annotations
|
|
14
16
|
|
|
@@ -22,8 +24,8 @@ import traceback
|
|
|
22
24
|
import wave
|
|
23
25
|
|
|
24
26
|
|
|
25
|
-
# Set before importing
|
|
26
|
-
#
|
|
27
|
+
# Set before importing ONNX Runtime so a globally configured Jetson CUDA device
|
|
28
|
+
# cannot become an accidental provider for speaker embeddings.
|
|
27
29
|
os.environ["CUDA_VISIBLE_DEVICES"] = ""
|
|
28
30
|
|
|
29
31
|
MODEL_NAME = "wespeaker-voxceleb-campplus"
|
|
@@ -34,6 +36,11 @@ MAX_DURATION_SECONDS = 30.0
|
|
|
34
36
|
MIN_RMS = 0.003
|
|
35
37
|
MIN_PEAK = 0.01
|
|
36
38
|
MAX_CLIPPED_FRACTION = 0.01
|
|
39
|
+
FBANK_IMPLEMENTATION = "kaldi-fbank-numpy-v1"
|
|
40
|
+
# SHA-256 over the rounded deterministic 2-second calibration fbank after
|
|
41
|
+
# utterance CMN. This protects the source implementation from silent feature
|
|
42
|
+
# drift while avoiding irrelevant last-bit FFT differences across CPU builds.
|
|
43
|
+
FBANK_VALIDATION_SIGNATURE = "sha256:19bb5552b8068d5cd2ccdbb07ac7a7ed1137400d3bdc3c54fb9c21d4f06e16fa"
|
|
37
44
|
|
|
38
45
|
|
|
39
46
|
def emit(payload: dict) -> None:
|
|
@@ -98,18 +105,93 @@ def quality_gate(samples, metrics):
|
|
|
98
105
|
return "usable"
|
|
99
106
|
|
|
100
107
|
|
|
108
|
+
def kaldi_fbank80_numpy(waveform, np):
|
|
109
|
+
"""Kaldi-compatible fbank for the fixed WeSpeaker CAM++ ONNX contract.
|
|
110
|
+
|
|
111
|
+
This mirrors WeSpeaker's published Kaldi fbank configuration: mono 16k
|
|
112
|
+
input, snip_edges=True,
|
|
113
|
+
remove_dc_offset=True, preemphasis=0.97, round_to_power_of_two=True,
|
|
114
|
+
80 triangular 20Hz..Nyquist mel bins, Hamming window and log power. The
|
|
115
|
+
caller passes PCM-scaled float32 samples (not normalized [-1, 1] values).
|
|
116
|
+
"""
|
|
117
|
+
dtype = np.float32
|
|
118
|
+
waveform = np.ascontiguousarray(waveform, dtype=dtype)
|
|
119
|
+
frame_size = 400
|
|
120
|
+
frame_shift = 160
|
|
121
|
+
fft_size = 512
|
|
122
|
+
if waveform.ndim != 1 or waveform.size < frame_size:
|
|
123
|
+
raise RuntimeError("Kaldi fbank requires at least one 25 ms mono frame")
|
|
124
|
+
frame_count = 1 + (waveform.size - frame_size) // frame_shift
|
|
125
|
+
frames = np.lib.stride_tricks.as_strided(
|
|
126
|
+
waveform,
|
|
127
|
+
shape=(frame_count, frame_size),
|
|
128
|
+
strides=(frame_shift * waveform.strides[0], waveform.strides[0]),
|
|
129
|
+
writeable=False,
|
|
130
|
+
).copy()
|
|
131
|
+
# Published Kaldi fbank frame/window defaults.
|
|
132
|
+
frames -= np.mean(frames, axis=1, dtype=dtype, keepdims=True)
|
|
133
|
+
frames[:, 1:] -= dtype(0.97) * frames[:, :-1]
|
|
134
|
+
frames[:, 0] -= dtype(0.97) * frames[:, 0]
|
|
135
|
+
sample_index = np.arange(frame_size, dtype=dtype)
|
|
136
|
+
hamming = dtype(0.54) - dtype(0.46) * np.cos(
|
|
137
|
+
dtype(2.0 * np.pi) * sample_index / dtype(frame_size - 1)
|
|
138
|
+
)
|
|
139
|
+
frames *= hamming
|
|
140
|
+
padded = np.pad(frames, ((0, 0), (0, fft_size - frame_size)), mode="constant")
|
|
141
|
+
# NumPy's rfft may calculate internally at float64 precision. Cast its
|
|
142
|
+
# power spectrum back to float32 before the Kaldi-style matrix multiply.
|
|
143
|
+
spectrum = np.asarray(np.abs(np.fft.rfft(padded, axis=1)) ** 2, dtype=dtype)
|
|
144
|
+
|
|
145
|
+
nyquist = dtype(8000.0)
|
|
146
|
+
low_hz = dtype(20.0)
|
|
147
|
+
mel = lambda hz: dtype(1127.0) * np.log(dtype(1.0) + hz / dtype(700.0))
|
|
148
|
+
mel_low = mel(low_hz)
|
|
149
|
+
mel_high = mel(nyquist)
|
|
150
|
+
mel_step = (mel_high - mel_low) / dtype(81.0)
|
|
151
|
+
bin_index = np.arange(80, dtype=dtype)[:, None]
|
|
152
|
+
left = mel_low + bin_index * mel_step
|
|
153
|
+
center = mel_low + (bin_index + dtype(1.0)) * mel_step
|
|
154
|
+
right = mel_low + (bin_index + dtype(2.0)) * mel_step
|
|
155
|
+
fft_bin_hz = dtype(16000.0 / fft_size) * np.arange(fft_size // 2, dtype=dtype)[None, :]
|
|
156
|
+
fft_bin_mel = mel(fft_bin_hz)
|
|
157
|
+
up = (fft_bin_mel - left) / (center - left)
|
|
158
|
+
down = (right - fft_bin_mel) / (right - center)
|
|
159
|
+
banks = np.maximum(dtype(0.0), np.minimum(up, down))
|
|
160
|
+
banks = np.pad(banks, ((0, 0), (0, 1)), mode="constant")
|
|
161
|
+
energies = np.matmul(spectrum, np.asarray(banks.T, dtype=dtype))
|
|
162
|
+
return np.log(np.maximum(energies, np.finfo(dtype).eps)).astype(dtype, copy=False)
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def validate_kaldi_fbank_numpy(np):
|
|
166
|
+
"""Validate the no-Torch preprocessing path before readiness is true."""
|
|
167
|
+
t = np.arange(SAMPLE_RATE * 2, dtype=np.float32) / np.float32(SAMPLE_RATE)
|
|
168
|
+
samples = (
|
|
169
|
+
np.float32(0.075) * np.sin(np.float32(2.0 * np.pi * 220.0) * t)
|
|
170
|
+
+ np.float32(0.025) * np.sin(np.float32(2.0 * np.pi * 440.0) * t)
|
|
171
|
+
).astype(np.float32)
|
|
172
|
+
features = kaldi_fbank80_numpy(samples * np.float32(1 << 15), np)
|
|
173
|
+
features = (features - np.mean(features, axis=0, dtype=np.float32, keepdims=True)).astype(np.float32)
|
|
174
|
+
if features.shape != (198, 80) or not np.all(np.isfinite(features)):
|
|
175
|
+
raise RuntimeError("NumPy Kaldi fbank calibration produced an invalid feature matrix")
|
|
176
|
+
if not np.all(np.abs(np.mean(features, axis=0)) < np.float32(5e-5)):
|
|
177
|
+
raise RuntimeError("NumPy Kaldi fbank calibration failed utterance CMN")
|
|
178
|
+
rounded = np.rint(features * np.float32(10_000)).astype("<i4", copy=False)
|
|
179
|
+
signature = "sha256:" + hashlib.sha256(rounded.tobytes()).hexdigest()
|
|
180
|
+
if signature != FBANK_VALIDATION_SIGNATURE:
|
|
181
|
+
raise RuntimeError(
|
|
182
|
+
"NumPy Kaldi fbank calibration signature mismatch: " + signature
|
|
183
|
+
)
|
|
184
|
+
return signature
|
|
185
|
+
|
|
186
|
+
|
|
101
187
|
class WeSpeakerCamPlus:
|
|
102
188
|
def __init__(self, model_path: str):
|
|
103
189
|
started = time.perf_counter()
|
|
104
190
|
import numpy as np
|
|
105
191
|
import onnxruntime as ort
|
|
106
|
-
import torch
|
|
107
|
-
from torchaudio.compliance import kaldi
|
|
108
192
|
|
|
109
193
|
self.np = np
|
|
110
|
-
self.
|
|
111
|
-
self.kaldi = kaldi
|
|
112
|
-
torch.set_num_threads(1)
|
|
194
|
+
self.preprocessing_signature = validate_kaldi_fbank_numpy(np)
|
|
113
195
|
options = ort.SessionOptions()
|
|
114
196
|
options.inter_op_num_threads = 1
|
|
115
197
|
options.intra_op_num_threads = 1
|
|
@@ -134,27 +216,14 @@ class WeSpeakerCamPlus:
|
|
|
134
216
|
self.model_load_ms = (time.perf_counter() - started) * 1000
|
|
135
217
|
|
|
136
218
|
def features(self, samples):
|
|
137
|
-
#
|
|
138
|
-
#
|
|
139
|
-
|
|
140
|
-
waveform = self.torch.from_numpy(
|
|
141
|
-
self.np.ascontiguousarray(samples * (1 << 15), dtype=self.np.float32)
|
|
142
|
-
).unsqueeze(0)
|
|
143
|
-
mat = self.kaldi.fbank(
|
|
144
|
-
waveform,
|
|
145
|
-
num_mel_bins=80,
|
|
146
|
-
frame_length=25,
|
|
147
|
-
frame_shift=10,
|
|
148
|
-
dither=0.0,
|
|
149
|
-
sample_frequency=SAMPLE_RATE,
|
|
150
|
-
window_type="hamming",
|
|
151
|
-
use_energy=False,
|
|
152
|
-
)
|
|
219
|
+
# WeSpeaker's ONNX inference multiplies normalized PCM values by 2**15
|
|
220
|
+
# before Kaldi fbank. Decode follows the same convention above.
|
|
221
|
+
mat = kaldi_fbank80_numpy(samples * self.np.float32(1 << 15), self.np)
|
|
153
222
|
if mat.ndim != 2 or mat.shape[0] < 1 or mat.shape[1] != 80:
|
|
154
|
-
raise RuntimeError(f"
|
|
223
|
+
raise RuntimeError(f"NumPy Kaldi fbank produced an invalid shape: {tuple(mat.shape)}")
|
|
155
224
|
# CMN without CVN, over the full input utterance, per official script.
|
|
156
|
-
mat = mat - self.
|
|
157
|
-
return mat.
|
|
225
|
+
mat = mat - self.np.mean(mat, axis=0, dtype=self.np.float32, keepdims=True)
|
|
226
|
+
return mat[None, :, :].astype(self.np.float32, copy=False)
|
|
158
227
|
|
|
159
228
|
def embed(self, samples):
|
|
160
229
|
started = time.perf_counter()
|
|
@@ -186,9 +255,22 @@ class WeSpeakerCamPlus:
|
|
|
186
255
|
|
|
187
256
|
def main() -> int:
|
|
188
257
|
parser = argparse.ArgumentParser()
|
|
189
|
-
parser.add_argument("--model"
|
|
190
|
-
parser.add_argument("--model-digest"
|
|
258
|
+
parser.add_argument("--model")
|
|
259
|
+
parser.add_argument("--model-digest")
|
|
260
|
+
parser.add_argument("--preprocessing-probe", action="store_true")
|
|
191
261
|
args = parser.parse_args()
|
|
262
|
+
if args.preprocessing_probe:
|
|
263
|
+
import numpy as np
|
|
264
|
+
emit(
|
|
265
|
+
{
|
|
266
|
+
"type": "preprocessing_probe",
|
|
267
|
+
"implementation": FBANK_IMPLEMENTATION,
|
|
268
|
+
"validation_signature": validate_kaldi_fbank_numpy(np),
|
|
269
|
+
}
|
|
270
|
+
)
|
|
271
|
+
return 0
|
|
272
|
+
if not args.model or not args.model_digest:
|
|
273
|
+
parser.error("--model and --model-digest are required unless --preprocessing-probe is used")
|
|
192
274
|
worker = WeSpeakerCamPlus(args.model)
|
|
193
275
|
worker.warm()
|
|
194
276
|
emit(
|
|
@@ -202,6 +284,8 @@ def main() -> int:
|
|
|
202
284
|
"dimension": EMBEDDING_DIMENSION,
|
|
203
285
|
"model_load_ms": round(worker.model_load_ms, 3),
|
|
204
286
|
"warmed": True,
|
|
287
|
+
"preprocessing": FBANK_IMPLEMENTATION,
|
|
288
|
+
"preprocessing_signature": worker.preprocessing_signature,
|
|
205
289
|
}
|
|
206
290
|
)
|
|
207
291
|
for line in sys.stdin:
|
|
@@ -463,6 +463,38 @@ def amplify(samples):
|
|
|
463
463
|
return samples
|
|
464
464
|
return np.clip(samples * gain, -1.0, 1.0)
|
|
465
465
|
|
|
466
|
+
|
|
467
|
+
def pcm16_signal_metrics(samples):
|
|
468
|
+
"""Authoritative stats for the raw PCM16 intake before any ASR call."""
|
|
469
|
+
count = int(len(samples))
|
|
470
|
+
if count <= 0:
|
|
471
|
+
return {
|
|
472
|
+
"sampleRateHz": SAMPLE_RATE, "channels": CHANNELS, "bitsPerSample": 16,
|
|
473
|
+
"sampleCount": 0, "durationMs": 0.0, "peakPcm16": 0,
|
|
474
|
+
"rmsPcm16": 0.0, "activeSampleCount": 0, "activeSampleRatio": 0.0,
|
|
475
|
+
}
|
|
476
|
+
# Input originates in int16 stdin. Rounding only restores that exact integer
|
|
477
|
+
# scale for reporting; it never modifies the samples supplied to Whisper.
|
|
478
|
+
pcm = np.rint(samples * 32768.0).astype(np.int32, copy=False)
|
|
479
|
+
magnitude = np.abs(pcm)
|
|
480
|
+
peak = int(np.max(magnitude))
|
|
481
|
+
active = int(np.count_nonzero(magnitude >= 8))
|
|
482
|
+
return {
|
|
483
|
+
"sampleRateHz": SAMPLE_RATE, "channels": CHANNELS, "bitsPerSample": 16,
|
|
484
|
+
"sampleCount": count, "durationMs": count * 1000.0 / SAMPLE_RATE,
|
|
485
|
+
"peakPcm16": peak, "rmsPcm16": float(np.sqrt(np.mean(pcm.astype(np.float64) ** 2))),
|
|
486
|
+
"activeSampleCount": active, "activeSampleRatio": active / count,
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def digitally_silent_pcm16(samples):
|
|
491
|
+
metrics = pcm16_signal_metrics(samples)
|
|
492
|
+
return (
|
|
493
|
+
metrics["peakPcm16"] <= 4
|
|
494
|
+
and metrics["rmsPcm16"] <= 1.0
|
|
495
|
+
and metrics["activeSampleCount"] == 0
|
|
496
|
+
), metrics
|
|
497
|
+
|
|
466
498
|
# ---------------------------------------------------------------------------
|
|
467
499
|
# Main transcription loop — utterance FSM
|
|
468
500
|
# ---------------------------------------------------------------------------
|
|
@@ -546,6 +578,11 @@ def main():
|
|
|
546
578
|
})
|
|
547
579
|
|
|
548
580
|
def run_transcribe(m, samples):
|
|
581
|
+
# Every live PCM ASR call goes through this exact gate. It is stricter
|
|
582
|
+
# than VAD only for digital silence, so it cannot reject quiet speech.
|
|
583
|
+
silent, _ = digitally_silent_pcm16(samples)
|
|
584
|
+
if silent:
|
|
585
|
+
return {"text": "", "segments": []}
|
|
549
586
|
return m.transcribe(
|
|
550
587
|
samples,
|
|
551
588
|
fp16=fp16,
|
|
@@ -644,6 +681,10 @@ def main():
|
|
|
644
681
|
|
|
645
682
|
def transcribe_final(samples, speech_block_count=0, rms_values=None, peak=0.0):
|
|
646
683
|
nonlocal last_final_text
|
|
684
|
+
silent, signal = digitally_silent_pcm16(samples)
|
|
685
|
+
if silent:
|
|
686
|
+
emit({"type": "no_speech", "kind": "no_speech", "reason": "digital_silence", "text": "", "signal": signal})
|
|
687
|
+
return
|
|
647
688
|
if not utterance_quality_ok(samples, speech_block_count, rms_values or [], peak):
|
|
648
689
|
return
|
|
649
690
|
samples = amplify(samples)
|