python-voiceio 0.6.0__py3-none-any.whl → 0.7.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {python_voiceio-0.6.0.dist-info → python_voiceio-0.7.0.dist-info}/METADATA +1 -1
- {python_voiceio-0.6.0.dist-info → python_voiceio-0.7.0.dist-info}/RECORD +10 -10
- voiceio/__init__.py +1 -1
- voiceio/config.py +6 -1
- voiceio/llm_api.py +60 -11
- voiceio/streaming.py +22 -1
- {python_voiceio-0.6.0.dist-info → python_voiceio-0.7.0.dist-info}/WHEEL +0 -0
- {python_voiceio-0.6.0.dist-info → python_voiceio-0.7.0.dist-info}/entry_points.txt +0 -0
- {python_voiceio-0.6.0.dist-info → python_voiceio-0.7.0.dist-info}/licenses/LICENSE +0 -0
- {python_voiceio-0.6.0.dist-info → python_voiceio-0.7.0.dist-info}/top_level.txt +0 -0
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
python_voiceio-0.
|
|
2
|
-
voiceio/__init__.py,sha256=
|
|
1
|
+
python_voiceio-0.7.0.dist-info/licenses/LICENSE,sha256=Gz61o8jFTAvZUZyB3nWDB3DQQVuipjfPkVu9W8hBHM0,1072
|
|
2
|
+
voiceio/__init__.py,sha256=RaANGbRu5e-vehwXI1-Qe2ggPPfs1TQaZj072JdbLk4,22
|
|
3
3
|
voiceio/__main__.py,sha256=xT5QCGGreYMisHO7Lh_Y-xAQ2TOzG2D7npKknGNcWSY,53
|
|
4
4
|
voiceio/app.py,sha256=4o3nOv_Z_NCBr6KTHr1eSGT159ZE98dVBAy5wUOpzAo,64398
|
|
5
5
|
voiceio/audit.py,sha256=vX3x-YAhmt6EZYRZd0_6wvTquZu5sxGMvH45BhSBnCk,14915
|
|
@@ -9,7 +9,7 @@ voiceio/backends.py,sha256=-rs1YhPblub4NjFb6qCD7Ih9FqBRXtPuvjVv9bQoQ1w,368
|
|
|
9
9
|
voiceio/cli.py,sha256=_HfcWQt2Jf34YfI4W1RavsyqegQy9jrg0nwLe6DvD6E,71639
|
|
10
10
|
voiceio/clipboard_read.py,sha256=CqbHSvOztcQ3u2I2EzFm0jdcDVKv56gl0T1YfWxNpmI,3390
|
|
11
11
|
voiceio/commands.py,sha256=Vhtn8s5G5OcCbRN44YZ-_14Fa85ISXl_3Ix-KqSHs2A,4694
|
|
12
|
-
voiceio/config.py,sha256=
|
|
12
|
+
voiceio/config.py,sha256=CHDtEigtv-aKPsDFFGkc1AiqmubiFHeU1HaEaPgoIMc,14343
|
|
13
13
|
voiceio/consent.py,sha256=jtdp1TkTbicUSTCDmV4qo4SdbS4kQpYzy1Kjc_7pJVw,1883
|
|
14
14
|
voiceio/corrections.py,sha256=7moKhkVvo5CZU1mZobV2-MajkcHiwliKBY1dSBFh0w0,7445
|
|
15
15
|
voiceio/demo.py,sha256=QRNJuNObXCsjwGD-Cz3dh81cfb21xiuSeVyyU720U9Q,8673
|
|
@@ -18,7 +18,7 @@ voiceio/health.py,sha256=5YNu1mMzwTo6I_Z_Mu3Bs46ZHDpVmxP0AXyMNAbaczc,15707
|
|
|
18
18
|
voiceio/hints.py,sha256=dmoiMAJ0YGTWwFc9kIPvEs_BQJN7UEUiwYuLUD_GoQQ,1415
|
|
19
19
|
voiceio/history.py,sha256=v7ciyZui8MndaKtTF5TdTKmql2ALOZNYKguE0Mo_lTY,5581
|
|
20
20
|
voiceio/llm.py,sha256=_m13jodMVB1iEtze3VGkBCafkg85OssTPBEVyaRabgA,9250
|
|
21
|
-
voiceio/llm_api.py,sha256=
|
|
21
|
+
voiceio/llm_api.py,sha256=1wLLi6pliNmg-gS-bUKgwsGQSsLWAhX2XkEWXk292V8,9441
|
|
22
22
|
voiceio/numbers.py,sha256=MP8jag4_F2OqUU5H14KzOmR9L8gWg48yQhLoUbGg_YI,8376
|
|
23
23
|
voiceio/pidlock.py,sha256=RInlF_8H7wR0nqkdlhUfxYs7OGzHIvb5p-XDjjQSTNg,619
|
|
24
24
|
voiceio/platform.py,sha256=f_8MmC5q1sCEuEPyKjRqlVvV9FExuGmwn_aJXhkpiEk,11809
|
|
@@ -29,7 +29,7 @@ voiceio/recorder.py,sha256=p1r0mH_plBD3BHqBbIGV_m2N4Mj5RAF7LfMu8u6mk5I,12813
|
|
|
29
29
|
voiceio/retention.py,sha256=i_VWPCyykT4kQ1uKilTplpiIrACTYljn09BU1vL71A0,6019
|
|
30
30
|
voiceio/service.py,sha256=jqX1opZyQh_dOJMK4bVf8Aoei0zOuuk1Ogta9BXTWGY,12106
|
|
31
31
|
voiceio/snapshots.py,sha256=bm-lzIeYucDUDlPBZiFiWd-jyB2I6Ft1jRANZCO6go4,2351
|
|
32
|
-
voiceio/streaming.py,sha256=
|
|
32
|
+
voiceio/streaming.py,sha256=3JMXBU1DAIDvItGRKlh7RGlD1zcGwgNcrFCKLkWdmTI,23215
|
|
33
33
|
voiceio/transcriber.py,sha256=2ev_hwdusve7ZZqht9A9wtjjT72rcr7SuhmSIW8aClk,10721
|
|
34
34
|
voiceio/vad.py,sha256=72_ICk4jSuSPYvTHdHYtHzY8W5uaGEOz5FTdqkolJi8,5327
|
|
35
35
|
voiceio/vocabulary.py,sha256=eGV7QZs3tN6Cw5YpyoZsVEk2M5fJvie1n2XFi9CD4hw,6321
|
|
@@ -71,8 +71,8 @@ voiceio/typers/pynput_type.py,sha256=DTtkT59M-EKOAlTNkvcZrRvf_Fq52OpI8S5t_8S-QAY
|
|
|
71
71
|
voiceio/typers/wtype.py,sha256=d1wG-HZdYDQxXIysV_Xvk9HixDpaJgl3VEoKbP5vhIs,1786
|
|
72
72
|
voiceio/typers/xdotool.py,sha256=dh1zhPqUT8ihJahJ7ZKm6PtfYY087UAzADx664DvQOM,1356
|
|
73
73
|
voiceio/typers/ydotool.py,sha256=dt0W9ot__W8LQG2so8vZLGIw9leK2Q6g3i-Voo0LXE4,5327
|
|
74
|
-
python_voiceio-0.
|
|
75
|
-
python_voiceio-0.
|
|
76
|
-
python_voiceio-0.
|
|
77
|
-
python_voiceio-0.
|
|
78
|
-
python_voiceio-0.
|
|
74
|
+
python_voiceio-0.7.0.dist-info/METADATA,sha256=hV7gmFjuAjAJpCb9O01qqobjZfk7qxk3Kjow7l5-oaI,15877
|
|
75
|
+
python_voiceio-0.7.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
|
|
76
|
+
python_voiceio-0.7.0.dist-info/entry_points.txt,sha256=U64fA65zxzyLoC8bgbn2ztQVWHLsc0o0H0qCk1J9DMc,218
|
|
77
|
+
python_voiceio-0.7.0.dist-info/top_level.txt,sha256=piwtn309lD6uexQyXdZ-efAVBJF9y6Wfr48Z-8zkNhg,8
|
|
78
|
+
python_voiceio-0.7.0.dist-info/RECORD,,
|
voiceio/__init__.py
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
__version__ = "0.
|
|
1
|
+
__version__ = "0.7.0"
|
voiceio/config.py
CHANGED
|
@@ -147,7 +147,12 @@ class PostCorrectConfig:
|
|
|
147
147
|
model can be overridden here; empty falls back to the autocorrect model.
|
|
148
148
|
"""
|
|
149
149
|
enabled: bool = False
|
|
150
|
-
|
|
150
|
+
# Latency-critical (blocks the commit): default to a fast cheap model.
|
|
151
|
+
# Bake-off on the real correction prompt: gemini-2.5-flash-lite 0.44s
|
|
152
|
+
# with fixes identical to kimi-k2's 2.0s. OpenRouter model id — on a
|
|
153
|
+
# custom base_url set a model your endpoint serves (empty = use
|
|
154
|
+
# [autocorrect].model).
|
|
155
|
+
model: str = "google/gemini-2.5-flash-lite"
|
|
151
156
|
timeout_secs: float = 8.0
|
|
152
157
|
min_words: int = 4 # skip utterances shorter than this
|
|
153
158
|
|
voiceio/llm_api.py
CHANGED
|
@@ -5,16 +5,73 @@ local Ollama (via /v1/chat/completions), etc. Zero dependencies beyond stdlib.
|
|
|
5
5
|
"""
|
|
6
6
|
from __future__ import annotations
|
|
7
7
|
|
|
8
|
+
import http.client
|
|
9
|
+
import io
|
|
8
10
|
import json
|
|
9
11
|
import logging
|
|
10
12
|
import os
|
|
13
|
+
import threading
|
|
11
14
|
import urllib.error
|
|
12
|
-
|
|
15
|
+
from urllib.parse import urlsplit
|
|
13
16
|
|
|
14
17
|
from voiceio.config import AutocorrectConfig
|
|
15
18
|
|
|
16
19
|
log = logging.getLogger(__name__)
|
|
17
20
|
|
|
21
|
+
# Keep-alive connection pool: a fresh TLS handshake costs ~0.2-0.5s per call,
|
|
22
|
+
# which lands directly on postcorrect's stop-to-commit latency. Connections
|
|
23
|
+
# are checked OUT for the duration of a request and returned only on clean
|
|
24
|
+
# completion, so a deadline-abandoned postcorrect thread can never share a
|
|
25
|
+
# socket with a later call.
|
|
26
|
+
_POOL_LOCK = threading.Lock()
|
|
27
|
+
_CONN_POOL: dict[str, http.client.HTTPConnection] = {}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _post_json(url: str, headers: dict, body: dict, timeout: float) -> dict:
|
|
31
|
+
"""POST JSON over a pooled keep-alive connection; return the parsed reply.
|
|
32
|
+
|
|
33
|
+
A stale pooled connection (server closed it while idle) gets one retry on
|
|
34
|
+
a fresh one. Non-2xx raises urllib.error.HTTPError so callers keep their
|
|
35
|
+
existing status-code handling.
|
|
36
|
+
"""
|
|
37
|
+
parts = urlsplit(url)
|
|
38
|
+
pool_key = f"{parts.scheme}://{parts.netloc}"
|
|
39
|
+
path = parts.path or "/"
|
|
40
|
+
payload = json.dumps(body).encode()
|
|
41
|
+
|
|
42
|
+
for attempt in (1, 2):
|
|
43
|
+
with _POOL_LOCK:
|
|
44
|
+
conn = _CONN_POOL.pop(pool_key, None)
|
|
45
|
+
pooled = conn is not None
|
|
46
|
+
if conn is None:
|
|
47
|
+
cls = (http.client.HTTPSConnection if parts.scheme == "https"
|
|
48
|
+
else http.client.HTTPConnection)
|
|
49
|
+
conn = cls(parts.hostname, parts.port, timeout=timeout)
|
|
50
|
+
elif conn.sock is not None:
|
|
51
|
+
conn.sock.settimeout(timeout)
|
|
52
|
+
try:
|
|
53
|
+
conn.request("POST", path, body=payload, headers=headers)
|
|
54
|
+
resp = conn.getresponse()
|
|
55
|
+
data = resp.read()
|
|
56
|
+
except (http.client.HTTPException, OSError):
|
|
57
|
+
conn.close()
|
|
58
|
+
if not pooled or attempt == 2:
|
|
59
|
+
raise
|
|
60
|
+
continue # stale keep-alive — retry once on a fresh connection
|
|
61
|
+
if resp.status // 100 != 2:
|
|
62
|
+
conn.close()
|
|
63
|
+
raise urllib.error.HTTPError(
|
|
64
|
+
url, resp.status, resp.reason, dict(resp.getheaders()),
|
|
65
|
+
io.BytesIO(data),
|
|
66
|
+
)
|
|
67
|
+
with _POOL_LOCK:
|
|
68
|
+
if pool_key in _CONN_POOL:
|
|
69
|
+
conn.close() # keep at most one idle connection per host
|
|
70
|
+
else:
|
|
71
|
+
_CONN_POOL[pool_key] = conn
|
|
72
|
+
return json.loads(data)
|
|
73
|
+
raise AssertionError("unreachable")
|
|
74
|
+
|
|
18
75
|
|
|
19
76
|
_LOCAL_HOSTS = ("localhost", "127.0.0.1", "0.0.0.0", "[::1]", "::1")
|
|
20
77
|
_consent_warned = False
|
|
@@ -94,11 +151,7 @@ def _anthropic_request(
|
|
|
94
151
|
"anthropic-version": "2023-06-01",
|
|
95
152
|
}
|
|
96
153
|
|
|
97
|
-
|
|
98
|
-
url, data=json.dumps(body).encode(), headers=headers, method="POST",
|
|
99
|
-
)
|
|
100
|
-
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
101
|
-
data = json.loads(resp.read())
|
|
154
|
+
data = _post_json(url, headers, body, timeout)
|
|
102
155
|
# Anthropic returns content as a list of blocks; some thinking models
|
|
103
156
|
# may also include `thinking` blocks which we ignore.
|
|
104
157
|
blocks = data.get("content") or []
|
|
@@ -137,11 +190,7 @@ def _openai_request(
|
|
|
137
190
|
"Authorization": f"Bearer {api_key}",
|
|
138
191
|
}
|
|
139
192
|
|
|
140
|
-
|
|
141
|
-
url, data=json.dumps(body).encode(), headers=headers, method="POST",
|
|
142
|
-
)
|
|
143
|
-
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
144
|
-
data = json.loads(resp.read())
|
|
193
|
+
data = _post_json(url, headers, body, timeout)
|
|
145
194
|
# Be defensive: thinking models (Kimi K2.6, GPT reasoning, etc.) can
|
|
146
195
|
# return content=None when the answer is in `reasoning` / `reasoning_content`
|
|
147
196
|
# instead. Also some malformed responses lack `choices` entirely.
|
voiceio/streaming.py
CHANGED
|
@@ -32,7 +32,10 @@ _FREEZE_SILENCE_WINDOW_SECS = 0.3
|
|
|
32
32
|
# and the decode path normalizes audio anyway, so an absolute threshold
|
|
33
33
|
# would read a quiet mic's speech as silence (mid-word cuts) and a hot
|
|
34
34
|
# noise floor as speech (freeze never fires).
|
|
35
|
-
|
|
35
|
+
# Calibrated on real recordings: between-words noise floor sits at
|
|
36
|
+
# ~0.27-0.38 of overall tail RMS, speech windows at >=0.9. 0.6 separates
|
|
37
|
+
# both with margin; 0.15 never fired on a mic with a normal noise floor.
|
|
38
|
+
_FREEZE_SILENCE_RATIO = 0.6
|
|
36
39
|
_FREEZE_SILENCE_FLOOR = 1e-4 # digital silence is always quiet
|
|
37
40
|
# Whisper conditions on ~224 prompt tokens; more frozen context is wasted.
|
|
38
41
|
_FREEZE_CONTEXT_CHARS = 400
|
|
@@ -158,6 +161,12 @@ class StreamingSession:
|
|
|
158
161
|
self._frozen_samples = 0
|
|
159
162
|
self._frozen_raw = ""
|
|
160
163
|
self._frozen_segments: list[dict] = []
|
|
164
|
+
# Decode backpressure: when the CPU is contended, decodes slow down
|
|
165
|
+
# and 1s-tick interim passes would queue back-to-back, each covering
|
|
166
|
+
# a longer tail — a starvation spiral. Require at least as much NEW
|
|
167
|
+
# audio as the last decode took before starting another interim.
|
|
168
|
+
self._last_decode_end = 0 # samples covered by the last decode
|
|
169
|
+
self._last_decode_secs = 0.0 # how long that decode took
|
|
161
170
|
# Raw (pre-pipeline) text + confidence of the final pass, for history
|
|
162
171
|
self.raw_final_text: str | None = None
|
|
163
172
|
self.final_latency: dict = {}
|
|
@@ -294,6 +303,9 @@ class StreamingSession:
|
|
|
294
303
|
tail = audio[self._frozen_samples:]
|
|
295
304
|
if not final and len(tail) < self._sample_rate * min_seconds:
|
|
296
305
|
return # not enough new audio since the last freeze
|
|
306
|
+
grown_secs = (len(audio) - self._last_decode_end) / self._sample_rate
|
|
307
|
+
if not final and grown_secs < max(min_seconds, self._last_decode_secs):
|
|
308
|
+
return # backpressure: don't decode faster than we can keep up
|
|
297
309
|
|
|
298
310
|
# Freeze when the tail has grown long AND ends at a speech pause
|
|
299
311
|
# (never cut mid-word). The freeze pass gets beam search because its
|
|
@@ -333,6 +345,8 @@ class StreamingSession:
|
|
|
333
345
|
return
|
|
334
346
|
tail_segments = list(getattr(self._transcriber, "last_segments", []))
|
|
335
347
|
t_transcribe = time.monotonic() - t0
|
|
348
|
+
self._last_decode_end = len(audio)
|
|
349
|
+
self._last_decode_secs = t_transcribe
|
|
336
350
|
|
|
337
351
|
self.trace.append({
|
|
338
352
|
"t": round(time.monotonic() - self._t0, 2),
|
|
@@ -353,6 +367,13 @@ class StreamingSession:
|
|
|
353
367
|
)
|
|
354
368
|
raw = self._frozen_raw if freeze else _join_text(self._frozen_raw, tail_text)
|
|
355
369
|
|
|
370
|
+
# A slow interim decode can complete AFTER the user pressed stop
|
|
371
|
+
# (observed: 29s under CPU contention). Applying it would overwrite
|
|
372
|
+
# the preedit with stale partial text moments before the final
|
|
373
|
+
# commit rewrites it again — discard it. Freeze bookkeeping above
|
|
374
|
+
# is kept: it is beam-quality and shrinks the final tail.
|
|
375
|
+
if not final and self._stop_event.is_set():
|
|
376
|
+
return
|
|
356
377
|
# An interim pass that decoded nothing must not touch the display:
|
|
357
378
|
# rewriting to frozen-only text would visibly delete the un-frozen
|
|
358
379
|
# words the previous pass just showed (transient decoder flake).
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|