wakebox 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- wakebox-0.1.1/PKG-INFO +31 -0
- wakebox-0.1.1/README.md +17 -0
- wakebox-0.1.1/pyproject.toml +36 -0
- wakebox-0.1.1/setup.cfg +4 -0
- wakebox-0.1.1/src/wakebox/__init__.py +6 -0
- wakebox-0.1.1/src/wakebox/engine.py +146 -0
- wakebox-0.1.1/src/wakebox/microphone.py +55 -0
- wakebox-0.1.1/src/wakebox/models/embedding_model.onnx +0 -0
- wakebox-0.1.1/src/wakebox/models/melspectrogram.onnx +0 -0
- wakebox-0.1.1/src/wakebox/py.typed +0 -0
- wakebox-0.1.1/src/wakebox.egg-info/PKG-INFO +31 -0
- wakebox-0.1.1/src/wakebox.egg-info/SOURCES.txt +14 -0
- wakebox-0.1.1/src/wakebox.egg-info/dependency_links.txt +1 -0
- wakebox-0.1.1/src/wakebox.egg-info/requires.txt +7 -0
- wakebox-0.1.1/src/wakebox.egg-info/top_level.txt +1 -0
- wakebox-0.1.1/tests/test_engine.py +41 -0
wakebox-0.1.1/PKG-INFO
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: wakebox
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: On-device wake word. No key.
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Requires-Python: >=3.11
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
Requires-Dist: numpy
|
|
9
|
+
Requires-Dist: onnxruntime
|
|
10
|
+
Requires-Dist: sounddevice
|
|
11
|
+
Provides-Extra: dev
|
|
12
|
+
Requires-Dist: pytest; extra == "dev"
|
|
13
|
+
Requires-Dist: soundfile; extra == "dev"
|
|
14
|
+
|
|
15
|
+
On-device wake word, no key.
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
pip install wakebox
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
Create a phrase file at [https://wakebox.ai](https://wakebox.ai), then:
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
from wakebox import Engine
|
|
25
|
+
import numpy as np
|
|
26
|
+
|
|
27
|
+
engine = Engine(["phrase.onnx"], sensitivities=[0.5])
|
|
28
|
+
index = engine.process(np.zeros(1280, dtype=np.int16)) # 0, or None
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
`process` takes 16 kHz mono int16 audio and returns the phrase index, or `None`. Sensitivity is a number from 0 to 1 on each file. It is not stored in the file.
|
wakebox-0.1.1/README.md
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
On-device wake word, no key.
|
|
2
|
+
|
|
3
|
+
```bash
|
|
4
|
+
pip install wakebox
|
|
5
|
+
```
|
|
6
|
+
|
|
7
|
+
Create a phrase file at [https://wakebox.ai](https://wakebox.ai), then:
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
from wakebox import Engine
|
|
11
|
+
import numpy as np
|
|
12
|
+
|
|
13
|
+
engine = Engine(["phrase.onnx"], sensitivities=[0.5])
|
|
14
|
+
index = engine.process(np.zeros(1280, dtype=np.int16)) # 0, or None
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
`process` takes 16 kHz mono int16 audio and returns the phrase index, or `None`. Sensitivity is a number from 0 to 1 on each file. It is not stored in the file.
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "wakebox"
|
|
3
|
+
version = "0.1.1"
|
|
4
|
+
description = "On-device wake word. No key."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.11"
|
|
7
|
+
license = "MIT"
|
|
8
|
+
dependencies = ["numpy", "onnxruntime", "sounddevice"]
|
|
9
|
+
|
|
10
|
+
[project.optional-dependencies]
|
|
11
|
+
dev = ["pytest", "soundfile"]
|
|
12
|
+
|
|
13
|
+
[tool.setuptools.packages.find]
|
|
14
|
+
where = ["src"]
|
|
15
|
+
|
|
16
|
+
[tool.setuptools.package-data]
|
|
17
|
+
wakebox = ["models/*.onnx"]
|
|
18
|
+
|
|
19
|
+
[tool.pytest.ini_options]
|
|
20
|
+
testpaths = ["tests"]
|
|
21
|
+
|
|
22
|
+
[tool.ruff]
|
|
23
|
+
line-length = 120
|
|
24
|
+
src = ["src", "tests"]
|
|
25
|
+
|
|
26
|
+
[tool.ruff.lint]
|
|
27
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
28
|
+
|
|
29
|
+
[tool.mypy]
|
|
30
|
+
python_version = "3.12"
|
|
31
|
+
strict = true
|
|
32
|
+
packages = ["wakebox"]
|
|
33
|
+
|
|
34
|
+
[[tool.mypy.overrides]]
|
|
35
|
+
module = ["onnxruntime", "onnxruntime.*", "sounddevice", "sounddevice.*", "numpy", "numpy.*"]
|
|
36
|
+
ignore_missing_imports = true
|
wakebox-0.1.1/setup.cfg
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""On-device wake word engine.
|
|
2
|
+
|
|
3
|
+
The shared runtime is the frozen mel frontend and speech embedding. Each
|
|
4
|
+
phrase file is a small head. Sensitivity is chosen here, not stored in the file.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import cast
|
|
11
|
+
|
|
12
|
+
import numpy as np
|
|
13
|
+
import onnxruntime as ort
|
|
14
|
+
|
|
15
|
+
SAMPLE_RATE = 16000
|
|
16
|
+
MEL_WINDOW = 76
|
|
17
|
+
MEL_STRIDE = 8
|
|
18
|
+
EMBEDDING_DIM = 96
|
|
19
|
+
HEAD_FRAMES = 16
|
|
20
|
+
REFRACTORY_SEC = 1.0
|
|
21
|
+
# Scores for a real phrase sit in a narrow band near the top. Map the
|
|
22
|
+
# caller's 0–1 control onto that band. 0 is strict, 1 is loose.
|
|
23
|
+
THRESHOLD_STRICT = 0.995
|
|
24
|
+
THRESHOLD_LOOSE = 0.85
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def threshold_for_sensitivity(sensitivity: float) -> float:
|
|
28
|
+
value = float(np.clip(sensitivity, 0.0, 1.0))
|
|
29
|
+
return THRESHOLD_STRICT + (THRESHOLD_LOOSE - THRESHOLD_STRICT) * value
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _session(model: str | Path | bytes) -> ort.InferenceSession:
|
|
33
|
+
options = ort.SessionOptions()
|
|
34
|
+
options.inter_op_num_threads = 1
|
|
35
|
+
options.intra_op_num_threads = 1
|
|
36
|
+
if isinstance(model, bytes):
|
|
37
|
+
return ort.InferenceSession(model, sess_options=options, providers=["CPUExecutionProvider"])
|
|
38
|
+
return ort.InferenceSession(str(model), sess_options=options, providers=["CPUExecutionProvider"])
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class Engine:
|
|
42
|
+
"""Load phrase files and score 16 kHz mono int16 frames.
|
|
43
|
+
|
|
44
|
+
``process`` returns the index of the phrase that fired, or ``None``.
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
def __init__(self, phrases: list[str | Path | bytes], sensitivities: list[float] | None = None):
|
|
48
|
+
if not phrases:
|
|
49
|
+
raise ValueError("load at least one phrase file")
|
|
50
|
+
if sensitivities is None:
|
|
51
|
+
sensitivities = [0.5] * len(phrases)
|
|
52
|
+
if len(sensitivities) != len(phrases):
|
|
53
|
+
raise ValueError("sensitivities must match the phrase files")
|
|
54
|
+
here = Path(__file__).resolve().parent / "models"
|
|
55
|
+
self._mel = _session(here / "melspectrogram.onnx")
|
|
56
|
+
self._embedding = _session(here / "embedding_model.onnx")
|
|
57
|
+
self._mel_input = self._mel.get_inputs()[0].name
|
|
58
|
+
self._embedding_input = self._embedding.get_inputs()[0].name
|
|
59
|
+
self._heads = [_session(phrase) for phrase in phrases]
|
|
60
|
+
self._head_input = self._heads[0].get_inputs()[0].name
|
|
61
|
+
self.sensitivities = [float(value) for value in sensitivities]
|
|
62
|
+
self._audio = np.zeros(0, dtype=np.int16)
|
|
63
|
+
self._scored = 0
|
|
64
|
+
self._previous = [0.0] * len(self._heads)
|
|
65
|
+
self._last_fire = [-1e9] * len(self._heads)
|
|
66
|
+
self._samples = 0
|
|
67
|
+
|
|
68
|
+
def set_sensitivity(self, index: int, sensitivity: float) -> None:
|
|
69
|
+
self.sensitivities[index] = float(sensitivity)
|
|
70
|
+
|
|
71
|
+
def process(self, frame: np.ndarray) -> int | None:
|
|
72
|
+
samples = np.asarray(frame, dtype=np.int16).reshape(-1)
|
|
73
|
+
if samples.size == 0:
|
|
74
|
+
return None
|
|
75
|
+
self._samples += int(samples.size)
|
|
76
|
+
self._audio = np.concatenate([self._audio, samples])
|
|
77
|
+
max_samples = SAMPLE_RATE * 4
|
|
78
|
+
if self._audio.size > max_samples:
|
|
79
|
+
self._audio = self._audio[-max_samples:]
|
|
80
|
+
self._scored = 0
|
|
81
|
+
embeddings = self._embed(self._audio)
|
|
82
|
+
fired: int | None = None
|
|
83
|
+
now = self._samples / SAMPLE_RATE
|
|
84
|
+
while self._scored < len(embeddings):
|
|
85
|
+
index = self._scored
|
|
86
|
+
self._scored += 1
|
|
87
|
+
if index + 1 < HEAD_FRAMES:
|
|
88
|
+
continue
|
|
89
|
+
window = embeddings[index + 1 - HEAD_FRAMES : index + 1]
|
|
90
|
+
choice = self._score(window, now)
|
|
91
|
+
if choice is not None:
|
|
92
|
+
fired = choice
|
|
93
|
+
return fired
|
|
94
|
+
|
|
95
|
+
def process_buffer(self, audio: np.ndarray, frame: int = 1280) -> list[int]:
|
|
96
|
+
"""Feed a whole clip and return the phrase index of each detection."""
|
|
97
|
+
samples = np.asarray(audio, dtype=np.int16).reshape(-1)
|
|
98
|
+
hits = []
|
|
99
|
+
for start in range(0, len(samples), frame):
|
|
100
|
+
index = self.process(samples[start : start + frame])
|
|
101
|
+
if index is not None:
|
|
102
|
+
hits.append(index)
|
|
103
|
+
return hits
|
|
104
|
+
|
|
105
|
+
def _embed(self, audio: np.ndarray) -> np.ndarray:
|
|
106
|
+
spec = self._melspectrogram(audio)
|
|
107
|
+
if spec.shape[0] < MEL_WINDOW:
|
|
108
|
+
return np.zeros((0, EMBEDDING_DIM), dtype=np.float32)
|
|
109
|
+
windows = [
|
|
110
|
+
spec[i : i + MEL_WINDOW]
|
|
111
|
+
for i in range(0, spec.shape[0] - MEL_WINDOW + 1, MEL_STRIDE)
|
|
112
|
+
]
|
|
113
|
+
batch = np.asarray(windows, dtype=np.float32)[..., None]
|
|
114
|
+
raw = self._embedding.run(None, {self._embedding_input: batch})[0]
|
|
115
|
+
return cast(np.ndarray, raw.reshape(batch.shape[0], EMBEDDING_DIM))
|
|
116
|
+
|
|
117
|
+
def _melspectrogram(self, audio: np.ndarray) -> np.ndarray:
|
|
118
|
+
data = np.ascontiguousarray(audio.astype(np.float32))[None, :]
|
|
119
|
+
spec = np.squeeze(self._mel.run(None, {self._mel_input: data})[0])
|
|
120
|
+
if spec.ndim == 1:
|
|
121
|
+
spec = spec[None, :]
|
|
122
|
+
if spec.shape[-1] != 32 and spec.shape[0] == 32:
|
|
123
|
+
spec = spec.T
|
|
124
|
+
return cast(np.ndarray, (spec / 10.0 + 2.0).astype(np.float32))
|
|
125
|
+
|
|
126
|
+
def _score(self, window: np.ndarray, now: float) -> int | None:
|
|
127
|
+
features = np.ascontiguousarray(window.astype(np.float32)[None, :])
|
|
128
|
+
best_index = None
|
|
129
|
+
best_score = -1.0
|
|
130
|
+
for index, head in enumerate(self._heads):
|
|
131
|
+
logit = np.asarray(head.run(None, {self._head_input: features})[0]).reshape(-1)
|
|
132
|
+
score = float(1.0 / (1.0 + np.exp(-float(logit[0]))))
|
|
133
|
+
threshold = threshold_for_sensitivity(self.sensitivities[index])
|
|
134
|
+
crossed = self._previous[index] < threshold <= score
|
|
135
|
+
self._previous[index] = score
|
|
136
|
+
if not crossed:
|
|
137
|
+
continue
|
|
138
|
+
if now - self._last_fire[index] < REFRACTORY_SEC:
|
|
139
|
+
continue
|
|
140
|
+
if score > best_score:
|
|
141
|
+
best_score = score
|
|
142
|
+
best_index = index
|
|
143
|
+
if best_index is None:
|
|
144
|
+
return None
|
|
145
|
+
self._last_fire[best_index] = now
|
|
146
|
+
return best_index
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Microphone helper. The frame API is what tests drive."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
from typing import Protocol
|
|
7
|
+
|
|
8
|
+
import numpy as np
|
|
9
|
+
|
|
10
|
+
from wakebox.engine import Engine
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class _Capture(Protocol):
|
|
14
|
+
def start(self) -> None: ...
|
|
15
|
+
|
|
16
|
+
def stop(self) -> None: ...
|
|
17
|
+
|
|
18
|
+
def close(self) -> None: ...
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class Microphone:
|
|
22
|
+
"""Read the default input and call back with the phrase index."""
|
|
23
|
+
|
|
24
|
+
def __init__(self, engine: Engine, on_detection: Callable[[int], None]):
|
|
25
|
+
self.engine = engine
|
|
26
|
+
self.on_detection = on_detection
|
|
27
|
+
self._stream: _Capture | None = None
|
|
28
|
+
|
|
29
|
+
def start(self) -> None:
|
|
30
|
+
import sounddevice as sd
|
|
31
|
+
|
|
32
|
+
def callback(indata: np.ndarray, frames: int, time: object, status: object) -> None:
|
|
33
|
+
del frames, time, status
|
|
34
|
+
samples = np.asarray(indata, dtype=np.int16).reshape(-1)
|
|
35
|
+
index = self.engine.process(samples)
|
|
36
|
+
if index is not None:
|
|
37
|
+
self.on_detection(index)
|
|
38
|
+
|
|
39
|
+
stream = sd.InputStream(
|
|
40
|
+
samplerate=16000,
|
|
41
|
+
channels=1,
|
|
42
|
+
dtype="int16",
|
|
43
|
+
blocksize=1280,
|
|
44
|
+
callback=callback,
|
|
45
|
+
)
|
|
46
|
+
self._stream = stream
|
|
47
|
+
stream.start()
|
|
48
|
+
|
|
49
|
+
def stop(self) -> None:
|
|
50
|
+
stream = self._stream
|
|
51
|
+
self._stream = None
|
|
52
|
+
if stream is None:
|
|
53
|
+
return
|
|
54
|
+
stream.stop()
|
|
55
|
+
stream.close()
|
|
Binary file
|
|
Binary file
|
|
File without changes
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: wakebox
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: On-device wake word. No key.
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Requires-Python: >=3.11
|
|
7
|
+
Description-Content-Type: text/markdown
|
|
8
|
+
Requires-Dist: numpy
|
|
9
|
+
Requires-Dist: onnxruntime
|
|
10
|
+
Requires-Dist: sounddevice
|
|
11
|
+
Provides-Extra: dev
|
|
12
|
+
Requires-Dist: pytest; extra == "dev"
|
|
13
|
+
Requires-Dist: soundfile; extra == "dev"
|
|
14
|
+
|
|
15
|
+
On-device wake word, no key.
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
pip install wakebox
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
Create a phrase file at [https://wakebox.ai](https://wakebox.ai), then:
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
from wakebox import Engine
|
|
25
|
+
import numpy as np
|
|
26
|
+
|
|
27
|
+
engine = Engine(["phrase.onnx"], sensitivities=[0.5])
|
|
28
|
+
index = engine.process(np.zeros(1280, dtype=np.int16)) # 0, or None
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
`process` takes 16 kHz mono int16 audio and returns the phrase index, or `None`. Sensitivity is a number from 0 to 1 on each file. It is not stored in the file.
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
README.md
|
|
2
|
+
pyproject.toml
|
|
3
|
+
src/wakebox/__init__.py
|
|
4
|
+
src/wakebox/engine.py
|
|
5
|
+
src/wakebox/microphone.py
|
|
6
|
+
src/wakebox/py.typed
|
|
7
|
+
src/wakebox.egg-info/PKG-INFO
|
|
8
|
+
src/wakebox.egg-info/SOURCES.txt
|
|
9
|
+
src/wakebox.egg-info/dependency_links.txt
|
|
10
|
+
src/wakebox.egg-info/requires.txt
|
|
11
|
+
src/wakebox.egg-info/top_level.txt
|
|
12
|
+
src/wakebox/models/embedding_model.onnx
|
|
13
|
+
src/wakebox/models/melspectrogram.onnx
|
|
14
|
+
tests/test_engine.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
wakebox
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
from pathlib import Path
|
|
2
|
+
|
|
3
|
+
import numpy as np
|
|
4
|
+
import soundfile
|
|
5
|
+
|
|
6
|
+
from wakebox import Engine, threshold_for_sensitivity
|
|
7
|
+
|
|
8
|
+
ROOT = Path(__file__).resolve().parents[3]
|
|
9
|
+
PHRASE = ROOT / "fixtures" / "alexa.onnx"
|
|
10
|
+
AUDIO = ROOT / "fixtures" / "audio"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def test_sensitivity_is_strict_at_zero_and_loose_at_one():
|
|
14
|
+
assert threshold_for_sensitivity(0) > threshold_for_sensitivity(1)
|
|
15
|
+
assert threshold_for_sensitivity(0) == 0.995
|
|
16
|
+
assert threshold_for_sensitivity(1) == 0.85
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def test_real_phrase_fires_and_quiet_audio_does_not():
|
|
20
|
+
engine = Engine([PHRASE], sensitivities=[0.5])
|
|
21
|
+
audio, rate = soundfile.read(AUDIO / "alexa-a.flac", dtype="int16")
|
|
22
|
+
assert rate == 16000
|
|
23
|
+
hits = engine.process_buffer(audio)
|
|
24
|
+
assert hits == [0]
|
|
25
|
+
|
|
26
|
+
quiet, rate = soundfile.read(AUDIO / "quiet.wav", dtype="int16")
|
|
27
|
+
assert rate == 16000
|
|
28
|
+
assert Engine([PHRASE], sensitivities=[0.5]).process_buffer(quiet) == []
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def test_two_files_return_the_stronger_index():
|
|
32
|
+
engine = Engine([PHRASE, PHRASE], sensitivities=[0.2, 0.9])
|
|
33
|
+
audio, _ = soundfile.read(AUDIO / "alexa-b.flac", dtype="int16")
|
|
34
|
+
hits = engine.process_buffer(audio)
|
|
35
|
+
assert hits
|
|
36
|
+
assert set(hits) <= {0, 1}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def test_empty_frame_is_ignored():
|
|
40
|
+
engine = Engine([PHRASE])
|
|
41
|
+
assert engine.process(np.zeros(0, dtype=np.int16)) is None
|