python-voiceio 0.6.0__py3-none-any.whl → 0.7.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: python-voiceio
3
- Version: 0.6.0
3
+ Version: 0.7.0
4
4
  Summary: Voice dictation for Linux. Speak → text, locally, instantly.
5
5
  Author: Hugo Montenegro
6
6
  License-Expression: MIT
@@ -1,5 +1,5 @@
1
- python_voiceio-0.6.0.dist-info/licenses/LICENSE,sha256=Gz61o8jFTAvZUZyB3nWDB3DQQVuipjfPkVu9W8hBHM0,1072
2
- voiceio/__init__.py,sha256=cID1jLnC_vj48GgMN6Yb1FA3JsQ95zNmCHmRYE8TFhY,22
1
+ python_voiceio-0.7.0.dist-info/licenses/LICENSE,sha256=Gz61o8jFTAvZUZyB3nWDB3DQQVuipjfPkVu9W8hBHM0,1072
2
+ voiceio/__init__.py,sha256=RaANGbRu5e-vehwXI1-Qe2ggPPfs1TQaZj072JdbLk4,22
3
3
  voiceio/__main__.py,sha256=xT5QCGGreYMisHO7Lh_Y-xAQ2TOzG2D7npKknGNcWSY,53
4
4
  voiceio/app.py,sha256=4o3nOv_Z_NCBr6KTHr1eSGT159ZE98dVBAy5wUOpzAo,64398
5
5
  voiceio/audit.py,sha256=vX3x-YAhmt6EZYRZd0_6wvTquZu5sxGMvH45BhSBnCk,14915
@@ -9,7 +9,7 @@ voiceio/backends.py,sha256=-rs1YhPblub4NjFb6qCD7Ih9FqBRXtPuvjVv9bQoQ1w,368
9
9
  voiceio/cli.py,sha256=_HfcWQt2Jf34YfI4W1RavsyqegQy9jrg0nwLe6DvD6E,71639
10
10
  voiceio/clipboard_read.py,sha256=CqbHSvOztcQ3u2I2EzFm0jdcDVKv56gl0T1YfWxNpmI,3390
11
11
  voiceio/commands.py,sha256=Vhtn8s5G5OcCbRN44YZ-_14Fa85ISXl_3Ix-KqSHs2A,4694
12
- voiceio/config.py,sha256=AOFaJras9rUPRDz7xeaZd4-WbWLOPUe0f0hYcQNwdJM,14042
12
+ voiceio/config.py,sha256=CHDtEigtv-aKPsDFFGkc1AiqmubiFHeU1HaEaPgoIMc,14343
13
13
  voiceio/consent.py,sha256=jtdp1TkTbicUSTCDmV4qo4SdbS4kQpYzy1Kjc_7pJVw,1883
14
14
  voiceio/corrections.py,sha256=7moKhkVvo5CZU1mZobV2-MajkcHiwliKBY1dSBFh0w0,7445
15
15
  voiceio/demo.py,sha256=QRNJuNObXCsjwGD-Cz3dh81cfb21xiuSeVyyU720U9Q,8673
@@ -18,7 +18,7 @@ voiceio/health.py,sha256=5YNu1mMzwTo6I_Z_Mu3Bs46ZHDpVmxP0AXyMNAbaczc,15707
18
18
  voiceio/hints.py,sha256=dmoiMAJ0YGTWwFc9kIPvEs_BQJN7UEUiwYuLUD_GoQQ,1415
19
19
  voiceio/history.py,sha256=v7ciyZui8MndaKtTF5TdTKmql2ALOZNYKguE0Mo_lTY,5581
20
20
  voiceio/llm.py,sha256=_m13jodMVB1iEtze3VGkBCafkg85OssTPBEVyaRabgA,9250
21
- voiceio/llm_api.py,sha256=ex1brY10a8xNtw4cHCYC0yCth3Z02NkCQaHdB4XQPag,7497
21
+ voiceio/llm_api.py,sha256=1wLLi6pliNmg-gS-bUKgwsGQSsLWAhX2XkEWXk292V8,9441
22
22
  voiceio/numbers.py,sha256=MP8jag4_F2OqUU5H14KzOmR9L8gWg48yQhLoUbGg_YI,8376
23
23
  voiceio/pidlock.py,sha256=RInlF_8H7wR0nqkdlhUfxYs7OGzHIvb5p-XDjjQSTNg,619
24
24
  voiceio/platform.py,sha256=f_8MmC5q1sCEuEPyKjRqlVvV9FExuGmwn_aJXhkpiEk,11809
@@ -29,7 +29,7 @@ voiceio/recorder.py,sha256=p1r0mH_plBD3BHqBbIGV_m2N4Mj5RAF7LfMu8u6mk5I,12813
29
29
  voiceio/retention.py,sha256=i_VWPCyykT4kQ1uKilTplpiIrACTYljn09BU1vL71A0,6019
30
30
  voiceio/service.py,sha256=jqX1opZyQh_dOJMK4bVf8Aoei0zOuuk1Ogta9BXTWGY,12106
31
31
  voiceio/snapshots.py,sha256=bm-lzIeYucDUDlPBZiFiWd-jyB2I6Ft1jRANZCO6go4,2351
32
- voiceio/streaming.py,sha256=XpF4fALTAgYqRB8zoDL81TuKlKLmUOc5_Cfd3auZFBI,21802
32
+ voiceio/streaming.py,sha256=3JMXBU1DAIDvItGRKlh7RGlD1zcGwgNcrFCKLkWdmTI,23215
33
33
  voiceio/transcriber.py,sha256=2ev_hwdusve7ZZqht9A9wtjjT72rcr7SuhmSIW8aClk,10721
34
34
  voiceio/vad.py,sha256=72_ICk4jSuSPYvTHdHYtHzY8W5uaGEOz5FTdqkolJi8,5327
35
35
  voiceio/vocabulary.py,sha256=eGV7QZs3tN6Cw5YpyoZsVEk2M5fJvie1n2XFi9CD4hw,6321
@@ -71,8 +71,8 @@ voiceio/typers/pynput_type.py,sha256=DTtkT59M-EKOAlTNkvcZrRvf_Fq52OpI8S5t_8S-QAY
71
71
  voiceio/typers/wtype.py,sha256=d1wG-HZdYDQxXIysV_Xvk9HixDpaJgl3VEoKbP5vhIs,1786
72
72
  voiceio/typers/xdotool.py,sha256=dh1zhPqUT8ihJahJ7ZKm6PtfYY087UAzADx664DvQOM,1356
73
73
  voiceio/typers/ydotool.py,sha256=dt0W9ot__W8LQG2so8vZLGIw9leK2Q6g3i-Voo0LXE4,5327
74
- python_voiceio-0.6.0.dist-info/METADATA,sha256=HdRa-d2IgSiL7VZxA7lcXavG0dGkqcKGf6Lmzdf9xxk,15877
75
- python_voiceio-0.6.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
76
- python_voiceio-0.6.0.dist-info/entry_points.txt,sha256=U64fA65zxzyLoC8bgbn2ztQVWHLsc0o0H0qCk1J9DMc,218
77
- python_voiceio-0.6.0.dist-info/top_level.txt,sha256=piwtn309lD6uexQyXdZ-efAVBJF9y6Wfr48Z-8zkNhg,8
78
- python_voiceio-0.6.0.dist-info/RECORD,,
74
+ python_voiceio-0.7.0.dist-info/METADATA,sha256=hV7gmFjuAjAJpCb9O01qqobjZfk7qxk3Kjow7l5-oaI,15877
75
+ python_voiceio-0.7.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
76
+ python_voiceio-0.7.0.dist-info/entry_points.txt,sha256=U64fA65zxzyLoC8bgbn2ztQVWHLsc0o0H0qCk1J9DMc,218
77
+ python_voiceio-0.7.0.dist-info/top_level.txt,sha256=piwtn309lD6uexQyXdZ-efAVBJF9y6Wfr48Z-8zkNhg,8
78
+ python_voiceio-0.7.0.dist-info/RECORD,,
voiceio/__init__.py CHANGED
@@ -1 +1 @@
1
- __version__ = "0.6.0"
1
+ __version__ = "0.7.0"
voiceio/config.py CHANGED
@@ -147,7 +147,12 @@ class PostCorrectConfig:
147
147
  model can be overridden here; empty falls back to the autocorrect model.
148
148
  """
149
149
  enabled: bool = False
150
- model: str = "" # empty = use [autocorrect].model
150
+ # Latency-critical (blocks the commit): default to a fast cheap model.
151
+ # Bake-off on the real correction prompt: gemini-2.5-flash-lite 0.44s
152
+ # with fixes identical to kimi-k2's 2.0s. OpenRouter model id — on a
153
+ # custom base_url set a model your endpoint serves (empty = use
154
+ # [autocorrect].model).
155
+ model: str = "google/gemini-2.5-flash-lite"
151
156
  timeout_secs: float = 8.0
152
157
  min_words: int = 4 # skip utterances shorter than this
153
158
 
voiceio/llm_api.py CHANGED
@@ -5,16 +5,73 @@ local Ollama (via /v1/chat/completions), etc. Zero dependencies beyond stdlib.
5
5
  """
6
6
  from __future__ import annotations
7
7
 
8
+ import http.client
9
+ import io
8
10
  import json
9
11
  import logging
10
12
  import os
13
+ import threading
11
14
  import urllib.error
12
- import urllib.request
15
+ from urllib.parse import urlsplit
13
16
 
14
17
  from voiceio.config import AutocorrectConfig
15
18
 
16
19
  log = logging.getLogger(__name__)
17
20
 
21
+ # Keep-alive connection pool: a fresh TLS handshake costs ~0.2-0.5s per call,
22
+ # which lands directly on postcorrect's stop-to-commit latency. Connections
23
+ # are checked OUT for the duration of a request and returned only on clean
24
+ # completion, so a deadline-abandoned postcorrect thread can never share a
25
+ # socket with a later call.
26
+ _POOL_LOCK = threading.Lock()
27
+ _CONN_POOL: dict[str, http.client.HTTPConnection] = {}
28
+
29
+
30
+ def _post_json(url: str, headers: dict, body: dict, timeout: float) -> dict:
31
+ """POST JSON over a pooled keep-alive connection; return the parsed reply.
32
+
33
+ A stale pooled connection (server closed it while idle) gets one retry on
34
+ a fresh one. Non-2xx raises urllib.error.HTTPError so callers keep their
35
+ existing status-code handling.
36
+ """
37
+ parts = urlsplit(url)
38
+ pool_key = f"{parts.scheme}://{parts.netloc}"
39
+ path = parts.path or "/"
40
+ payload = json.dumps(body).encode()
41
+
42
+ for attempt in (1, 2):
43
+ with _POOL_LOCK:
44
+ conn = _CONN_POOL.pop(pool_key, None)
45
+ pooled = conn is not None
46
+ if conn is None:
47
+ cls = (http.client.HTTPSConnection if parts.scheme == "https"
48
+ else http.client.HTTPConnection)
49
+ conn = cls(parts.hostname, parts.port, timeout=timeout)
50
+ elif conn.sock is not None:
51
+ conn.sock.settimeout(timeout)
52
+ try:
53
+ conn.request("POST", path, body=payload, headers=headers)
54
+ resp = conn.getresponse()
55
+ data = resp.read()
56
+ except (http.client.HTTPException, OSError):
57
+ conn.close()
58
+ if not pooled or attempt == 2:
59
+ raise
60
+ continue # stale keep-alive — retry once on a fresh connection
61
+ if resp.status // 100 != 2:
62
+ conn.close()
63
+ raise urllib.error.HTTPError(
64
+ url, resp.status, resp.reason, dict(resp.getheaders()),
65
+ io.BytesIO(data),
66
+ )
67
+ with _POOL_LOCK:
68
+ if pool_key in _CONN_POOL:
69
+ conn.close() # keep at most one idle connection per host
70
+ else:
71
+ _CONN_POOL[pool_key] = conn
72
+ return json.loads(data)
73
+ raise AssertionError("unreachable")
74
+
18
75
 
19
76
  _LOCAL_HOSTS = ("localhost", "127.0.0.1", "0.0.0.0", "[::1]", "::1")
20
77
  _consent_warned = False
@@ -94,11 +151,7 @@ def _anthropic_request(
94
151
  "anthropic-version": "2023-06-01",
95
152
  }
96
153
 
97
- req = urllib.request.Request(
98
- url, data=json.dumps(body).encode(), headers=headers, method="POST",
99
- )
100
- with urllib.request.urlopen(req, timeout=timeout) as resp:
101
- data = json.loads(resp.read())
154
+ data = _post_json(url, headers, body, timeout)
102
155
  # Anthropic returns content as a list of blocks; some thinking models
103
156
  # may also include `thinking` blocks which we ignore.
104
157
  blocks = data.get("content") or []
@@ -137,11 +190,7 @@ def _openai_request(
137
190
  "Authorization": f"Bearer {api_key}",
138
191
  }
139
192
 
140
- req = urllib.request.Request(
141
- url, data=json.dumps(body).encode(), headers=headers, method="POST",
142
- )
143
- with urllib.request.urlopen(req, timeout=timeout) as resp:
144
- data = json.loads(resp.read())
193
+ data = _post_json(url, headers, body, timeout)
145
194
  # Be defensive: thinking models (Kimi K2.6, GPT reasoning, etc.) can
146
195
  # return content=None when the answer is in `reasoning` / `reasoning_content`
147
196
  # instead. Also some malformed responses lack `choices` entirely.
voiceio/streaming.py CHANGED
@@ -32,7 +32,10 @@ _FREEZE_SILENCE_WINDOW_SECS = 0.3
32
32
  # and the decode path normalizes audio anyway, so an absolute threshold
33
33
  # would read a quiet mic's speech as silence (mid-word cuts) and a hot
34
34
  # noise floor as speech (freeze never fires).
35
- _FREEZE_SILENCE_RATIO = 0.15
35
+ # Calibrated on real recordings: between-words noise floor sits at
36
+ # ~0.27-0.38 of overall tail RMS, speech windows at >=0.9. 0.6 separates
37
+ # both with margin; 0.15 never fired on a mic with a normal noise floor.
38
+ _FREEZE_SILENCE_RATIO = 0.6
36
39
  _FREEZE_SILENCE_FLOOR = 1e-4 # digital silence is always quiet
37
40
  # Whisper conditions on ~224 prompt tokens; more frozen context is wasted.
38
41
  _FREEZE_CONTEXT_CHARS = 400
@@ -158,6 +161,12 @@ class StreamingSession:
158
161
  self._frozen_samples = 0
159
162
  self._frozen_raw = ""
160
163
  self._frozen_segments: list[dict] = []
164
+ # Decode backpressure: when the CPU is contended, decodes slow down
165
+ # and 1s-tick interim passes would queue back-to-back, each covering
166
+ # a longer tail — a starvation spiral. Require at least as much NEW
167
+ # audio as the last decode took before starting another interim.
168
+ self._last_decode_end = 0 # samples covered by the last decode
169
+ self._last_decode_secs = 0.0 # how long that decode took
161
170
  # Raw (pre-pipeline) text + confidence of the final pass, for history
162
171
  self.raw_final_text: str | None = None
163
172
  self.final_latency: dict = {}
@@ -294,6 +303,9 @@ class StreamingSession:
294
303
  tail = audio[self._frozen_samples:]
295
304
  if not final and len(tail) < self._sample_rate * min_seconds:
296
305
  return # not enough new audio since the last freeze
306
+ grown_secs = (len(audio) - self._last_decode_end) / self._sample_rate
307
+ if not final and grown_secs < max(min_seconds, self._last_decode_secs):
308
+ return # backpressure: don't decode faster than we can keep up
297
309
 
298
310
  # Freeze when the tail has grown long AND ends at a speech pause
299
311
  # (never cut mid-word). The freeze pass gets beam search because its
@@ -333,6 +345,8 @@ class StreamingSession:
333
345
  return
334
346
  tail_segments = list(getattr(self._transcriber, "last_segments", []))
335
347
  t_transcribe = time.monotonic() - t0
348
+ self._last_decode_end = len(audio)
349
+ self._last_decode_secs = t_transcribe
336
350
 
337
351
  self.trace.append({
338
352
  "t": round(time.monotonic() - self._t0, 2),
@@ -353,6 +367,13 @@ class StreamingSession:
353
367
  )
354
368
  raw = self._frozen_raw if freeze else _join_text(self._frozen_raw, tail_text)
355
369
 
370
+ # A slow interim decode can complete AFTER the user pressed stop
371
+ # (observed: 29s under CPU contention). Applying it would overwrite
372
+ # the preedit with stale partial text moments before the final
373
+ # commit rewrites it again — discard it. Freeze bookkeeping above
374
+ # is kept: it is beam-quality and shrinks the final tail.
375
+ if not final and self._stop_event.is_set():
376
+ return
356
377
  # An interim pass that decoded nothing must not touch the display:
357
378
  # rewriting to frozen-only text would visibly delete the un-frozen
358
379
  # words the previous pass just showed (transient decoder flake).