python-voiceio 0.9.3__tar.gz → 0.9.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. {python_voiceio-0.9.3/python_voiceio.egg-info → python_voiceio-0.9.5}/PKG-INFO +1 -1
  2. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/pyproject.toml +1 -1
  3. {python_voiceio-0.9.3 → python_voiceio-0.9.5/python_voiceio.egg-info}/PKG-INFO +1 -1
  4. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_robustness.py +25 -7
  5. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_vad.py +33 -10
  6. python_voiceio-0.9.5/voiceio/__init__.py +1 -0
  7. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/recorder.py +32 -10
  8. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/vad.py +13 -3
  9. python_voiceio-0.9.3/voiceio/__init__.py +0 -1
  10. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/LICENSE +0 -0
  11. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/README.md +0 -0
  12. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/python_voiceio.egg-info/SOURCES.txt +0 -0
  13. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/python_voiceio.egg-info/dependency_links.txt +0 -0
  14. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/python_voiceio.egg-info/entry_points.txt +0 -0
  15. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/python_voiceio.egg-info/requires.txt +0 -0
  16. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/python_voiceio.egg-info/top_level.txt +0 -0
  17. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/setup.cfg +0 -0
  18. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_adjudicate.py +0 -0
  19. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_app_wiring.py +0 -0
  20. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_audio_quality.py +0 -0
  21. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_audit.py +0 -0
  22. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_autocorrect.py +0 -0
  23. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_autocorrect_state.py +0 -0
  24. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_backend_probes.py +0 -0
  25. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_cli.py +0 -0
  26. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_clipboard_read.py +0 -0
  27. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_commands.py +0 -0
  28. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_concurrency_lockdown.py +0 -0
  29. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_config.py +0 -0
  30. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_correct_batch.py +0 -0
  31. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_corrections.py +0 -0
  32. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_evaluate.py +0 -0
  33. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_fallback.py +0 -0
  34. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_health.py +0 -0
  35. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_hints.py +0 -0
  36. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_history.py +0 -0
  37. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_ibus_pending.py +0 -0
  38. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_ibus_ping.py +0 -0
  39. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_ibus_typer.py +0 -0
  40. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_llm.py +0 -0
  41. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_llm_api.py +0 -0
  42. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_numbers.py +0 -0
  43. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_platform.py +0 -0
  44. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_postcorrect.py +0 -0
  45. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_postprocess.py +0 -0
  46. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_prebuffer.py +0 -0
  47. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_prompt.py +0 -0
  48. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_recorder_integration.py +0 -0
  49. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_retention.py +0 -0
  50. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_security_hardening.py +0 -0
  51. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_snapshots.py +0 -0
  52. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_streaming.py +0 -0
  53. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_tokens.py +0 -0
  54. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_transcriber.py +0 -0
  55. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_tts.py +0 -0
  56. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_vocabulary.py +0 -0
  57. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_wizard.py +0 -0
  58. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/tests/test_wordfreq.py +0 -0
  59. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/__main__.py +0 -0
  60. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/app.py +0 -0
  61. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/audit.py +0 -0
  62. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/autocorrect.py +0 -0
  63. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/autocorrect_state.py +0 -0
  64. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/backends.py +0 -0
  65. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/cli.py +0 -0
  66. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/clipboard_read.py +0 -0
  67. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/commands.py +0 -0
  68. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/config.py +0 -0
  69. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/consent.py +0 -0
  70. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/corrections.py +0 -0
  71. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/demo.py +0 -0
  72. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/evaluate.py +0 -0
  73. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/feedback.py +0 -0
  74. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/health.py +0 -0
  75. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/hints.py +0 -0
  76. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/history.py +0 -0
  77. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/hotkeys/__init__.py +0 -0
  78. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/hotkeys/base.py +0 -0
  79. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/hotkeys/chain.py +0 -0
  80. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/hotkeys/evdev.py +0 -0
  81. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/hotkeys/pynput_backend.py +0 -0
  82. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/hotkeys/socket_backend.py +0 -0
  83. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/ibus/__init__.py +0 -0
  84. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/ibus/engine.py +0 -0
  85. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/ibus/pending.py +0 -0
  86. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/llm.py +0 -0
  87. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/llm_api.py +0 -0
  88. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/models/__init__.py +0 -0
  89. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/models/silero_vad.onnx +0 -0
  90. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/numbers.py +0 -0
  91. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/pidlock.py +0 -0
  92. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/platform.py +0 -0
  93. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/postcorrect.py +0 -0
  94. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/postprocess.py +0 -0
  95. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/prompt.py +0 -0
  96. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/retention.py +0 -0
  97. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/service.py +0 -0
  98. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/snapshots.py +0 -0
  99. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/sounds/__init__.py +0 -0
  100. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/sounds/commit.wav +0 -0
  101. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/sounds/start.wav +0 -0
  102. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/sounds/stop.wav +0 -0
  103. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/streaming.py +0 -0
  104. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tokens.py +0 -0
  105. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/transcriber.py +0 -0
  106. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tray/__init__.py +0 -0
  107. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tray/_icons.py +0 -0
  108. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tray/_indicator.py +0 -0
  109. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tray/_pystray.py +0 -0
  110. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tts/__init__.py +0 -0
  111. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tts/base.py +0 -0
  112. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tts/chain.py +0 -0
  113. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tts/edge_engine.py +0 -0
  114. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tts/espeak.py +0 -0
  115. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tts/piper_engine.py +0 -0
  116. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/tts/player.py +0 -0
  117. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/typers/__init__.py +0 -0
  118. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/typers/base.py +0 -0
  119. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/typers/chain.py +0 -0
  120. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/typers/clipboard.py +0 -0
  121. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/typers/ibus.py +0 -0
  122. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/typers/pynput_type.py +0 -0
  123. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/typers/wtype.py +0 -0
  124. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/typers/xdotool.py +0 -0
  125. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/typers/ydotool.py +0 -0
  126. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/vocab_stats.py +0 -0
  127. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/vocabulary.py +0 -0
  128. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/wizard.py +0 -0
  129. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/wordfreq.py +0 -0
  130. {python_voiceio-0.9.3 → python_voiceio-0.9.5}/voiceio/worker.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: python-voiceio
3
- Version: 0.9.3
3
+ Version: 0.9.5
4
4
  Summary: Voice dictation for Linux. Speak → text, locally, instantly.
5
5
  Author: Hugo Montenegro
6
6
  License-Expression: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "python-voiceio"
7
- version = "0.9.3"
7
+ version = "0.9.5"
8
8
  description = "Voice dictation for Linux. Speak → text, locally, instantly."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: python-voiceio
3
- Version: 0.9.3
3
+ Version: 0.9.5
4
4
  Summary: Voice dictation for Linux. Speak → text, locally, instantly.
5
5
  Author: Hugo Montenegro
6
6
  License-Expression: MIT
@@ -76,27 +76,45 @@ def _make_recorder():
76
76
  class TestAutoStop:
77
77
  """Auto-stop distinguishes 'spoke then paused' from 'nothing ever heard'."""
78
78
 
79
- def _feed(self, rec, secs, *, speech):
80
- """Drive _callback with `secs` of audio; speech=False → silence."""
79
+ def _feed(self, rec, secs, *, speech, amplitude=0.0):
80
+ """Drive _callback with `secs` of audio.
81
+
82
+ speech → what the (mocked) VAD reports (is_speech).
83
+ amplitude → raw sample level; 0.0 = digital silence (muted mic),
84
+ >1e-4 = a real audio signal reaching the mic.
85
+ """
81
86
  rec._vad.is_speech.return_value = speech
82
87
  frames = int(secs * rec.sample_rate)
83
- rec._callback(np.zeros((frames, 1), dtype=np.float32), frames, None, None)
88
+ data = np.full((frames, 1), amplitude, dtype=np.float32)
89
+ rec._callback(data, frames, None, None)
84
90
 
85
- def test_no_speech_fires_no_speech_reason(self):
91
+ def test_muted_mic_fires_no_speech_reason(self):
86
92
  rec = _make_recorder()
87
93
  rec._recording = True
88
94
  fired = []
89
95
  rec.set_on_auto_stop(lambda reason: fired.append(reason))
90
- # 21s of pure silence, speech never heard.
91
- self._feed(rec, 21, speech=False)
96
+ # 21s of digital zeros a muted / dead mic.
97
+ self._feed(rec, 21, speech=False, amplitude=0.0)
92
98
  assert fired == ["no_speech"]
93
99
 
100
+ def test_real_audio_vad_misses_does_not_autostop(self):
101
+ """Regression: real audio the VAD fails to flag as speech must NOT be
102
+ cut off. This is the 0.9.3 bug — a 20s dictation the Silero VAD scored
103
+ as silence auto-stopped mid-sentence."""
104
+ rec = _make_recorder()
105
+ rec._recording = True
106
+ fired = []
107
+ rec.set_on_auto_stop(lambda reason: fired.append(reason))
108
+ # 25s of clearly-present audio, but the VAD never calls it speech.
109
+ self._feed(rec, 25, speech=False, amplitude=0.05)
110
+ assert fired == [] # signal is present → neither auto-stop fires
111
+
94
112
  def test_no_speech_does_not_fire_early(self):
95
113
  rec = _make_recorder()
96
114
  rec._recording = True
97
115
  fired = []
98
116
  rec.set_on_auto_stop(lambda reason: fired.append(reason))
99
- self._feed(rec, 10, speech=False) # below the 20s threshold
117
+ self._feed(rec, 10, speech=False, amplitude=0.0) # below the 20s threshold
100
118
  assert fired == []
101
119
 
102
120
  def test_speech_then_silence_fires_silence_reason(self):
@@ -51,16 +51,39 @@ class TestSileroVad:
51
51
  chunk = np.zeros(1024, dtype=np.float32)
52
52
  assert vad.is_speech(chunk) is False
53
53
 
54
- def test_speech_detected(self, vad):
55
- # Generate a sine wave at speech-like frequency
56
- t = np.linspace(0, 0.1, 1600, dtype=np.float32)
57
- chunk = 0.5 * np.sin(2 * np.pi * 300 * t)
58
- # Run several chunks to build state
59
- for _ in range(5):
60
- result = vad.is_speech(chunk)
61
- # At least one should detect speech (sine wave isn't silence)
62
- # Note: Silero may not classify pure sine as speech, so we just test it runs
63
- assert isinstance(result, bool)
54
+ def test_v5_feeds_context_window(self, vad):
55
+ """Regression: Silero v5 consumes 512+64=576 samples the window plus
56
+ the previous window's 64-sample tail as context. Feeding a bare 512
57
+ (the pre-fix bug) makes the model score all real speech as silence
58
+ (measured: max prob 0.016 vs 1.0 with context)."""
59
+ if not vad._use_state:
60
+ pytest.skip("v4 model has no context requirement")
61
+ seen = []
62
+ real_run = vad._session.run
63
+
64
+ def spy(outs, inputs):
65
+ seen.append(inputs["input"])
66
+ return real_run(outs, inputs)
67
+
68
+ vad._session.run = spy
69
+ # Two distinct windows so we can check the context carries across.
70
+ w1 = np.arange(512, dtype=np.float32) / 512.0
71
+ w2 = np.arange(512, 1024, dtype=np.float32) / 512.0
72
+ vad.is_speech(w1)
73
+ vad.is_speech(w2)
74
+ assert seen[0].shape[-1] == 512 + 64 # 576, not a bare 512
75
+ # First call's context is zeros; second call's leading 64 samples must
76
+ # be the tail of the first window (state carried between steps).
77
+ assert np.allclose(seen[0][0, :64], 0.0)
78
+ assert np.allclose(seen[1][0, :64], w1[-64:])
79
+
80
+ def test_reset_clears_context(self, vad):
81
+ if not vad._use_state:
82
+ pytest.skip("v4 model has no context")
83
+ vad.is_speech(np.arange(512, dtype=np.float32) / 512.0)
84
+ assert not np.allclose(vad._context, 0.0) # context now populated
85
+ vad.reset()
86
+ assert np.allclose(vad._context, 0.0)
64
87
 
65
88
  def test_reset_clears_state(self, vad):
66
89
  chunk = np.zeros(1024, dtype=np.float32)
@@ -0,0 +1 @@
1
+ __version__ = "0.9.5"
@@ -15,6 +15,11 @@ if TYPE_CHECKING:
15
15
 
16
16
  log = logging.getLogger(__name__)
17
17
 
18
+ # Peak amplitude below this is treated as no signal at all (a muted mic emits
19
+ # exact zeros; a working mic's "silence" still noises above ~1e-4). Matches the
20
+ # floor in has_signal(). Used only to detect a dead mic, never speech vs pause.
21
+ _SIGNAL_FLOOR = 1e-4
22
+
18
23
 
19
24
  class RingBuffer:
20
25
  """Fixed-size ring buffer for float32 audio samples."""
@@ -111,6 +116,10 @@ class AudioRecorder:
111
116
  self._no_speech_secs = cfg.auto_stop_no_speech_secs
112
117
  self._sustained_silence = 0.0
113
118
  self._heard_speech = False
119
+ # Raw-signal (amplitude) tracking for the muted-mic auto-stop, kept
120
+ # separate from the VAD-based _heard_speech / _sustained_silence above.
121
+ self._heard_signal = False
122
+ self._silent_signal_secs = 0.0
114
123
  # Called with a reason: "silence" (spoke then paused) or "no_speech"
115
124
  # (nothing ever heard — a muted mic or a forgotten hotkey).
116
125
  self._on_auto_stop: Callable[[str], None] | None = None
@@ -221,6 +230,8 @@ class AudioRecorder:
221
230
  self._silent_chunks = 0.0
222
231
  self._sustained_silence = 0.0
223
232
  self._heard_speech = False
233
+ self._heard_signal = False
234
+ self._silent_signal_secs = 0.0
224
235
  self._pause_fired_at = 0
225
236
  self._meter_peak = 0.0
226
237
  self._meter_clipped = 0
@@ -301,8 +312,19 @@ class AudioRecorder:
301
312
 
302
313
  # Level metering
303
314
  abs_chunk = np.abs(chunk.ravel())
304
- self._meter_peak = max(self._meter_peak, float(abs_chunk.max(initial=0.0)))
315
+ chunk_peak = float(abs_chunk.max(initial=0.0))
316
+ self._meter_peak = max(self._meter_peak, chunk_peak)
305
317
  self._meter_samples += len(abs_chunk)
318
+
319
+ # Raw-signal presence, independent of the VAD. A muted mic delivers
320
+ # digital zeros; real audio (even speech the VAD fails to classify)
321
+ # sits well above the floor. This — NOT the VAD — gates the muted-mic
322
+ # auto-stop, so a recording is never cut while real audio is coming in.
323
+ if chunk_peak > _SIGNAL_FLOOR:
324
+ self._heard_signal = True
325
+ self._silent_signal_secs = 0.0
326
+ else:
327
+ self._silent_signal_secs += frames / self.sample_rate
306
328
  pinned = abs_chunk >= 0.99
307
329
  if np.count_nonzero(pinned) >= 4:
308
330
  runs = np.convolve(pinned.astype(np.int8), np.ones(4, dtype=np.int8), "valid")
@@ -345,18 +367,18 @@ class AudioRecorder:
345
367
  log.info("Auto-stopping after %.0fs of silence", self._auto_stop_secs)
346
368
  cb("silence")
347
369
 
348
- # Safety net: nothing heard at all for a long time. Because speech
349
- # resets _sustained_silence AND is gated out by not-_heard_speech, this
350
- # measures continuous silence from the very start of the recording a
351
- # muted mic (all zeros) or a hotkey pressed by accident. Distinct reason
352
- # so the app can surface it instead of failing silently.
370
+ # Safety net for a dead mic: NO raw audio signal at all for a long
371
+ # time the mic is muted / unplugged / delivering zeros, never merely
372
+ # that the VAD didn't flag speech (that would cut real dictation off
373
+ # mid-sentence). _heard_signal latches on the first non-zero chunk, so
374
+ # this only fires for a mic that was silent from the very start.
353
375
  elif (self._on_auto_stop is not None
354
376
  and self._no_speech_secs > 0
355
- and not self._heard_speech
356
- and self._sustained_silence >= self._no_speech_secs):
377
+ and not self._heard_signal
378
+ and self._silent_signal_secs >= self._no_speech_secs):
357
379
  cb = self._on_auto_stop
358
380
  self._on_auto_stop = None
359
- self._sustained_silence = 0.0
360
- log.info("Auto-stopping: no speech heard in %.0fs (mic muted?)",
381
+ self._silent_signal_secs = 0.0
382
+ log.info("Auto-stopping: no mic signal in %.0fs (muted?)",
361
383
  self._no_speech_secs)
362
384
  cb("no_speech")
@@ -16,6 +16,11 @@ log = logging.getLogger(__name__)
16
16
  _MODEL_PATH = Path(__file__).parent / "models" / "silero_vad.onnx"
17
17
  _WINDOW_SIZE = 512 # Silero expects 512 samples at 16kHz (~32ms)
18
18
  _SAMPLE_RATE = 16000
19
+ # Silero v5 consumes 576 samples per step: the 512-sample window PLUS the 64
20
+ # trailing samples of the previous window as context. Feeding a bare 512 makes
21
+ # the model score everything as non-speech (measured: max prob 0.016 vs 1.0
22
+ # with context). The official ONNX wrapper prepends this outside the graph.
23
+ _CONTEXT_SIZE = 64
19
24
 
20
25
 
21
26
  @runtime_checkable
@@ -46,6 +51,8 @@ class SileroVad:
46
51
  state_meta = [inp for inp in self._session.get_inputs() if inp.name == "state"][0]
47
52
  state_dim = state_meta.shape[2] # 128 for v5
48
53
  self._state = np.zeros((2, 1, state_dim), dtype=np.float32)
54
+ # v5 context: the previous window's trailing 64 samples.
55
+ self._context = np.zeros(_CONTEXT_SIZE, dtype=np.float32)
49
56
  else:
50
57
  self._h = np.zeros((2, 1, 64), dtype=np.float32)
51
58
  self._c = np.zeros((2, 1, 64), dtype=np.float32)
@@ -56,6 +63,7 @@ class SileroVad:
56
63
  def reset(self) -> None:
57
64
  if self._use_state:
58
65
  self._state = np.zeros_like(self._state)
66
+ self._context = np.zeros(_CONTEXT_SIZE, dtype=np.float32)
59
67
  else:
60
68
  self._h = np.zeros_like(self._h)
61
69
  self._c = np.zeros_like(self._c)
@@ -76,13 +84,15 @@ class SileroVad:
76
84
  return speech
77
85
 
78
86
  def _infer(self, window: np.ndarray) -> float:
79
- input_data = window.reshape(1, -1)
80
87
  sr = np.array(_SAMPLE_RATE, dtype=np.int64)
81
88
  if self._use_state:
82
- ort_inputs = {"input": input_data, "state": self._state, "sr": sr}
89
+ # Prepend the previous window's tail; the model needs 512+64=576.
90
+ x = np.concatenate([self._context, window])
91
+ self._context = window[-_CONTEXT_SIZE:].copy()
92
+ ort_inputs = {"input": x.reshape(1, -1), "state": self._state, "sr": sr}
83
93
  out, self._state = self._session.run(None, ort_inputs)
84
94
  else:
85
- ort_inputs = {"input": input_data, "h": self._h, "c": self._c, "sr": sr}
95
+ ort_inputs = {"input": window.reshape(1, -1), "h": self._h, "c": self._c, "sr": sr}
86
96
  out, self._h, self._c = self._session.run(None, ort_inputs)
87
97
  return float(out.squeeze())
88
98
 
@@ -1 +0,0 @@
1
- __version__ = "0.9.3"
File without changes
File without changes
File without changes