python-voiceio 0.4.1__py3-none-any.whl → 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {python_voiceio-0.4.1.dist-info → python_voiceio-0.6.0.dist-info}/METADATA +1 -1
- {python_voiceio-0.4.1.dist-info → python_voiceio-0.6.0.dist-info}/RECORD +16 -16
- voiceio/__init__.py +1 -1
- voiceio/app.py +74 -1
- voiceio/clipboard_read.py +47 -1
- voiceio/config.py +25 -0
- voiceio/postcorrect.py +75 -6
- voiceio/recorder.py +11 -0
- voiceio/retention.py +65 -1
- voiceio/streaming.py +153 -12
- voiceio/transcriber.py +22 -6
- voiceio/wizard.py +1 -1
- {python_voiceio-0.4.1.dist-info → python_voiceio-0.6.0.dist-info}/WHEEL +0 -0
- {python_voiceio-0.4.1.dist-info → python_voiceio-0.6.0.dist-info}/entry_points.txt +0 -0
- {python_voiceio-0.4.1.dist-info → python_voiceio-0.6.0.dist-info}/licenses/LICENSE +0 -0
- {python_voiceio-0.4.1.dist-info → python_voiceio-0.6.0.dist-info}/top_level.txt +0 -0
|
@@ -1,15 +1,15 @@
|
|
|
1
|
-
python_voiceio-0.
|
|
2
|
-
voiceio/__init__.py,sha256=
|
|
1
|
+
python_voiceio-0.6.0.dist-info/licenses/LICENSE,sha256=Gz61o8jFTAvZUZyB3nWDB3DQQVuipjfPkVu9W8hBHM0,1072
|
|
2
|
+
voiceio/__init__.py,sha256=cID1jLnC_vj48GgMN6Yb1FA3JsQ95zNmCHmRYE8TFhY,22
|
|
3
3
|
voiceio/__main__.py,sha256=xT5QCGGreYMisHO7Lh_Y-xAQ2TOzG2D7npKknGNcWSY,53
|
|
4
|
-
voiceio/app.py,sha256=
|
|
4
|
+
voiceio/app.py,sha256=4o3nOv_Z_NCBr6KTHr1eSGT159ZE98dVBAy5wUOpzAo,64398
|
|
5
5
|
voiceio/audit.py,sha256=vX3x-YAhmt6EZYRZd0_6wvTquZu5sxGMvH45BhSBnCk,14915
|
|
6
6
|
voiceio/autocorrect.py,sha256=jGsglEYwAHWk2uWedeEZzyAMW2LWO_8nQGZgrWYy4Dc,26552
|
|
7
7
|
voiceio/autocorrect_state.py,sha256=BNzdHP5slcOV_-OU4MlaKE7xgVJzrPncsYFK30JWTs8,6858
|
|
8
8
|
voiceio/backends.py,sha256=-rs1YhPblub4NjFb6qCD7Ih9FqBRXtPuvjVv9bQoQ1w,368
|
|
9
9
|
voiceio/cli.py,sha256=_HfcWQt2Jf34YfI4W1RavsyqegQy9jrg0nwLe6DvD6E,71639
|
|
10
|
-
voiceio/clipboard_read.py,sha256=
|
|
10
|
+
voiceio/clipboard_read.py,sha256=CqbHSvOztcQ3u2I2EzFm0jdcDVKv56gl0T1YfWxNpmI,3390
|
|
11
11
|
voiceio/commands.py,sha256=Vhtn8s5G5OcCbRN44YZ-_14Fa85ISXl_3Ix-KqSHs2A,4694
|
|
12
|
-
voiceio/config.py,sha256=
|
|
12
|
+
voiceio/config.py,sha256=AOFaJras9rUPRDz7xeaZd4-WbWLOPUe0f0hYcQNwdJM,14042
|
|
13
13
|
voiceio/consent.py,sha256=jtdp1TkTbicUSTCDmV4qo4SdbS4kQpYzy1Kjc_7pJVw,1883
|
|
14
14
|
voiceio/corrections.py,sha256=7moKhkVvo5CZU1mZobV2-MajkcHiwliKBY1dSBFh0w0,7445
|
|
15
15
|
voiceio/demo.py,sha256=QRNJuNObXCsjwGD-Cz3dh81cfb21xiuSeVyyU720U9Q,8673
|
|
@@ -22,18 +22,18 @@ voiceio/llm_api.py,sha256=ex1brY10a8xNtw4cHCYC0yCth3Z02NkCQaHdB4XQPag,7497
|
|
|
22
22
|
voiceio/numbers.py,sha256=MP8jag4_F2OqUU5H14KzOmR9L8gWg48yQhLoUbGg_YI,8376
|
|
23
23
|
voiceio/pidlock.py,sha256=RInlF_8H7wR0nqkdlhUfxYs7OGzHIvb5p-XDjjQSTNg,619
|
|
24
24
|
voiceio/platform.py,sha256=f_8MmC5q1sCEuEPyKjRqlVvV9FExuGmwn_aJXhkpiEk,11809
|
|
25
|
-
voiceio/postcorrect.py,sha256=
|
|
25
|
+
voiceio/postcorrect.py,sha256=iyO7MPIR4a4XIASEGOs9Y3iUXVTfwtVkhVOWj2vY1lk,11201
|
|
26
26
|
voiceio/postprocess.py,sha256=9grkq3I6XgiQV6ZQbIw2_fOUe57tzC7GBPvLXqhhdTs,2979
|
|
27
27
|
voiceio/prompt.py,sha256=9iza8KoQxvO_U9NPMJ8Ep7j-zkIs9nW8N5sacCRpsEk,2904
|
|
28
|
-
voiceio/recorder.py,sha256=
|
|
29
|
-
voiceio/retention.py,sha256=
|
|
28
|
+
voiceio/recorder.py,sha256=p1r0mH_plBD3BHqBbIGV_m2N4Mj5RAF7LfMu8u6mk5I,12813
|
|
29
|
+
voiceio/retention.py,sha256=i_VWPCyykT4kQ1uKilTplpiIrACTYljn09BU1vL71A0,6019
|
|
30
30
|
voiceio/service.py,sha256=jqX1opZyQh_dOJMK4bVf8Aoei0zOuuk1Ogta9BXTWGY,12106
|
|
31
31
|
voiceio/snapshots.py,sha256=bm-lzIeYucDUDlPBZiFiWd-jyB2I6Ft1jRANZCO6go4,2351
|
|
32
|
-
voiceio/streaming.py,sha256=
|
|
33
|
-
voiceio/transcriber.py,sha256=
|
|
32
|
+
voiceio/streaming.py,sha256=XpF4fALTAgYqRB8zoDL81TuKlKLmUOc5_Cfd3auZFBI,21802
|
|
33
|
+
voiceio/transcriber.py,sha256=2ev_hwdusve7ZZqht9A9wtjjT72rcr7SuhmSIW8aClk,10721
|
|
34
34
|
voiceio/vad.py,sha256=72_ICk4jSuSPYvTHdHYtHzY8W5uaGEOz5FTdqkolJi8,5327
|
|
35
35
|
voiceio/vocabulary.py,sha256=eGV7QZs3tN6Cw5YpyoZsVEk2M5fJvie1n2XFi9CD4hw,6321
|
|
36
|
-
voiceio/wizard.py,sha256=
|
|
36
|
+
voiceio/wizard.py,sha256=AwjwrocrQp0DPSOMZ0m6z6G3hlo9VPceEwPzR9-xXe0,80227
|
|
37
37
|
voiceio/wordfreq.py,sha256=2UMjW1xIFqo5EchXzsXPizUhRb9N18CVgjnXso1wsgw,2382
|
|
38
38
|
voiceio/worker.py,sha256=M6e4s6P0-4UGQzw0j8Mq68nypvwZleOSwuRhjuOqlVE,2563
|
|
39
39
|
voiceio/hotkeys/__init__.py,sha256=rGXSGZLD2mS_Ep2HfVdRdkUKNr4xagySlL1FM00cYgk,735
|
|
@@ -71,8 +71,8 @@ voiceio/typers/pynput_type.py,sha256=DTtkT59M-EKOAlTNkvcZrRvf_Fq52OpI8S5t_8S-QAY
|
|
|
71
71
|
voiceio/typers/wtype.py,sha256=d1wG-HZdYDQxXIysV_Xvk9HixDpaJgl3VEoKbP5vhIs,1786
|
|
72
72
|
voiceio/typers/xdotool.py,sha256=dh1zhPqUT8ihJahJ7ZKm6PtfYY087UAzADx664DvQOM,1356
|
|
73
73
|
voiceio/typers/ydotool.py,sha256=dt0W9ot__W8LQG2so8vZLGIw9leK2Q6g3i-Voo0LXE4,5327
|
|
74
|
-
python_voiceio-0.
|
|
75
|
-
python_voiceio-0.
|
|
76
|
-
python_voiceio-0.
|
|
77
|
-
python_voiceio-0.
|
|
78
|
-
python_voiceio-0.
|
|
74
|
+
python_voiceio-0.6.0.dist-info/METADATA,sha256=HdRa-d2IgSiL7VZxA7lcXavG0dGkqcKGf6Lmzdf9xxk,15877
|
|
75
|
+
python_voiceio-0.6.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
|
|
76
|
+
python_voiceio-0.6.0.dist-info/entry_points.txt,sha256=U64fA65zxzyLoC8bgbn2ztQVWHLsc0o0H0qCk1J9DMc,218
|
|
77
|
+
python_voiceio-0.6.0.dist-info/top_level.txt,sha256=piwtn309lD6uexQyXdZ-efAVBJF9y6Wfr48Z-8zkNhg,8
|
|
78
|
+
python_voiceio-0.6.0.dist-info/RECORD,,
|
voiceio/__init__.py
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
__version__ = "0.
|
|
1
|
+
__version__ = "0.6.0"
|
voiceio/app.py
CHANGED
|
@@ -337,7 +337,31 @@ class VoiceIO:
|
|
|
337
337
|
log.exception("Cannot reopen audio stream, aborting recording")
|
|
338
338
|
return
|
|
339
339
|
|
|
340
|
-
if
|
|
340
|
+
if self.recorder.is_zombie():
|
|
341
|
+
# Full pre-buffer of digital zeros with a "healthy" stream is the
|
|
342
|
+
# post-suspend zombie signature (callbacks fire, capture node
|
|
343
|
+
# gone). Reopening fixes it; a failed reopen means no audio is
|
|
344
|
+
# possible, so abort rather than record silence.
|
|
345
|
+
log.warning("Zombie audio stream (all-zero pre-buffer) — reopening")
|
|
346
|
+
try:
|
|
347
|
+
self.recorder.reopen_stream()
|
|
348
|
+
time.sleep(0.4) # let the new stream fill some pre-buffer
|
|
349
|
+
except Exception:
|
|
350
|
+
log.exception("Audio stream reopen failed, aborting recording")
|
|
351
|
+
from voiceio import feedback
|
|
352
|
+
feedback.notify(
|
|
353
|
+
"VoiceIO: microphone unavailable",
|
|
354
|
+
"Could not reopen the audio stream — check your input device.",
|
|
355
|
+
)
|
|
356
|
+
return
|
|
357
|
+
if not self.recorder.has_signal():
|
|
358
|
+
log.warning("Mic still silent after reopen — muted or wrong device?")
|
|
359
|
+
from voiceio import feedback
|
|
360
|
+
feedback.notify(
|
|
361
|
+
"VoiceIO: microphone appears silent",
|
|
362
|
+
"Check that your mic is unmuted and the right input device is selected.",
|
|
363
|
+
)
|
|
364
|
+
elif not self.recorder.has_signal():
|
|
341
365
|
log.warning("Mic appears silent or muted (pre-buffer is all zeros)")
|
|
342
366
|
|
|
343
367
|
self._state = _State.RECORDING
|
|
@@ -368,6 +392,8 @@ class VoiceIO:
|
|
|
368
392
|
llm=self._llm,
|
|
369
393
|
voice_input_prefix=self._voice_input_prefix,
|
|
370
394
|
on_typer_broken=self._on_typer_broken,
|
|
395
|
+
on_interim=self._on_interim_text,
|
|
396
|
+
freeze_secs=self.cfg.output.streaming_freeze_secs,
|
|
371
397
|
)
|
|
372
398
|
self._session.start()
|
|
373
399
|
log.info("Recording... press [%s] again to stop", self.cfg.hotkey.key)
|
|
@@ -492,12 +518,19 @@ class VoiceIO:
|
|
|
492
518
|
) -> None:
|
|
493
519
|
"""Run final transcription and commit in background thread."""
|
|
494
520
|
t_final = time.monotonic()
|
|
521
|
+
# Snappy clipboard: make the best-so-far text pasteable the moment
|
|
522
|
+
# the user stops, instead of after the (possibly long) final decode.
|
|
523
|
+
if (self.cfg.output.copy_to_clipboard == "live"
|
|
524
|
+
and self._typer.name != "clipboard" and session.interim_text):
|
|
525
|
+
from voiceio import clipboard_read
|
|
526
|
+
clipboard_read.copy_text(session.interim_text)
|
|
495
527
|
extra = self._retention_extra(audio)
|
|
496
528
|
final_text = session.stop(audio)
|
|
497
529
|
if self._generation != gen:
|
|
498
530
|
log.debug("Finalize cancelled (gen %d, current %d)", gen, self._generation)
|
|
499
531
|
return
|
|
500
532
|
if final_text:
|
|
533
|
+
self._copy_result_async(final_text)
|
|
501
534
|
self._play_feedback(final_text)
|
|
502
535
|
stored = self._strip_voice_prefix(final_text)
|
|
503
536
|
self._prompt_builder.add_transcript(stored)
|
|
@@ -513,6 +546,16 @@ class VoiceIO:
|
|
|
513
546
|
duration=elapsed,
|
|
514
547
|
extra=extra,
|
|
515
548
|
)
|
|
549
|
+
from voiceio import retention
|
|
550
|
+
retention.save_trace(self.cfg.data, {
|
|
551
|
+
"ts": time.time(),
|
|
552
|
+
"audio": extra.get("audio"),
|
|
553
|
+
"duration": round(elapsed, 2),
|
|
554
|
+
"latency": latency,
|
|
555
|
+
# Snapshot: a timed-out worker join means the thread may
|
|
556
|
+
# still be appending while we serialize.
|
|
557
|
+
"passes": list(session.trace),
|
|
558
|
+
})
|
|
516
559
|
log.info("Streaming done (%.1fs): '%s'", elapsed, final_text)
|
|
517
560
|
# Release the IBus input source now that the final commit is done.
|
|
518
561
|
# Generation-checked inside: if a newer recording started, it already
|
|
@@ -565,6 +608,7 @@ class VoiceIO:
|
|
|
565
608
|
# Superseded by a newer recording — do not touch the typer.
|
|
566
609
|
if text and self._generation == gen:
|
|
567
610
|
self._type_with_fallback(text)
|
|
611
|
+
self._copy_result_async(text)
|
|
568
612
|
self._play_feedback(text)
|
|
569
613
|
stored = self._strip_voice_prefix(text)
|
|
570
614
|
self._prompt_builder.add_transcript(stored)
|
|
@@ -660,6 +704,35 @@ class VoiceIO:
|
|
|
660
704
|
from voiceio.feedback import play_record_stop
|
|
661
705
|
play_record_stop()
|
|
662
706
|
|
|
707
|
+
def _on_interim_text(self, text: str) -> None:
|
|
708
|
+
"""Streaming update: mirror the best-so-far text to the clipboard.
|
|
709
|
+
|
|
710
|
+
Runs on the streaming worker thread (a copy is a few ms). Skipped for
|
|
711
|
+
the clipboard typer, which pastes FROM the clipboard — overwriting it
|
|
712
|
+
here could race a pending Ctrl+V and paste the full text twice.
|
|
713
|
+
"""
|
|
714
|
+
if self.cfg.output.copy_to_clipboard != "live":
|
|
715
|
+
return
|
|
716
|
+
if self._typer.name == "clipboard":
|
|
717
|
+
return
|
|
718
|
+
from voiceio import clipboard_read
|
|
719
|
+
clipboard_read.copy_text(text)
|
|
720
|
+
|
|
721
|
+
def _copy_result_async(self, text: str) -> None:
|
|
722
|
+
"""Mirror the final text to the clipboard, off the commit hot path."""
|
|
723
|
+
if self.cfg.output.copy_to_clipboard not in ("final", "live"):
|
|
724
|
+
return
|
|
725
|
+
|
|
726
|
+
def _copy() -> None:
|
|
727
|
+
# Give a just-sent paste keystroke time to consume the clipboard
|
|
728
|
+
# before we overwrite it with the full text.
|
|
729
|
+
if self._typer.name == "clipboard":
|
|
730
|
+
time.sleep(0.3)
|
|
731
|
+
from voiceio import clipboard_read
|
|
732
|
+
clipboard_read.copy_text(text)
|
|
733
|
+
|
|
734
|
+
threading.Thread(target=_copy, daemon=True).start()
|
|
735
|
+
|
|
663
736
|
def _play_feedback(self, text: str) -> None:
|
|
664
737
|
if self.cfg.feedback.sound_enabled:
|
|
665
738
|
from voiceio.feedback import play_commit_sound
|
voiceio/clipboard_read.py
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""Read text from the system clipboard or primary selection."""
|
|
1
|
+
"""Read/write text from/to the system clipboard or primary selection."""
|
|
2
2
|
from __future__ import annotations
|
|
3
3
|
|
|
4
4
|
import logging
|
|
@@ -67,3 +67,49 @@ def read_text() -> str | None:
|
|
|
67
67
|
|
|
68
68
|
log.debug("No clipboard tool found")
|
|
69
69
|
return None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _copy_via(cmd: list[str], text: str) -> bool:
|
|
73
|
+
"""Pipe text into a copy command. Returns True on success."""
|
|
74
|
+
try:
|
|
75
|
+
result = subprocess.run(
|
|
76
|
+
cmd, input=text.encode(), capture_output=True, timeout=_TIMEOUT,
|
|
77
|
+
)
|
|
78
|
+
return result.returncode == 0
|
|
79
|
+
except (FileNotFoundError, subprocess.TimeoutExpired, OSError):
|
|
80
|
+
return False
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def copy_text(text: str) -> bool:
|
|
84
|
+
"""Copy text to the system CLIPBOARD selection.
|
|
85
|
+
|
|
86
|
+
Best-effort: returns False (never raises) when no tool is available
|
|
87
|
+
or the copy fails.
|
|
88
|
+
"""
|
|
89
|
+
if not text:
|
|
90
|
+
return False
|
|
91
|
+
p = detect()
|
|
92
|
+
|
|
93
|
+
if p.is_windows:
|
|
94
|
+
try:
|
|
95
|
+
import pyperclip
|
|
96
|
+
pyperclip.copy(text)
|
|
97
|
+
return True
|
|
98
|
+
except Exception:
|
|
99
|
+
log.debug("pyperclip not available for clipboard write")
|
|
100
|
+
return False
|
|
101
|
+
|
|
102
|
+
if p.is_mac:
|
|
103
|
+
return _copy_via(["pbcopy"], text)
|
|
104
|
+
|
|
105
|
+
if p.is_wayland:
|
|
106
|
+
if shutil.which("wl-copy"):
|
|
107
|
+
return _copy_via(["wl-copy", "--"], text)
|
|
108
|
+
else:
|
|
109
|
+
if shutil.which("xclip"):
|
|
110
|
+
return _copy_via(["xclip", "-selection", "clipboard"], text)
|
|
111
|
+
if shutil.which("xsel"):
|
|
112
|
+
return _copy_via(["xsel", "-i", "--clipboard"], text)
|
|
113
|
+
|
|
114
|
+
log.debug("No clipboard tool found for copy")
|
|
115
|
+
return False
|
voiceio/config.py
CHANGED
|
@@ -33,6 +33,11 @@ PID_PATH = LOG_DIR / "voiceio.pid"
|
|
|
33
33
|
# and the retired-rule state consulted by the mining side.
|
|
34
34
|
CORRECTIONS_AUDIT_PATH = LOG_DIR / "corrections_audit.jsonl"
|
|
35
35
|
METRICS_PATH = LOG_DIR / "metrics.jsonl"
|
|
36
|
+
# Intermediate pipeline data kept for debugging/profiling and future training:
|
|
37
|
+
# per-pass streaming decode traces and postcorrect before/after pairs. These
|
|
38
|
+
# live in their own JSONL files (not the rotating log) so they survive.
|
|
39
|
+
TRACES_PATH = LOG_DIR / "streaming_trace.jsonl"
|
|
40
|
+
POSTCORRECT_PAIRS_PATH = LOG_DIR / "postcorrect_pairs.jsonl"
|
|
36
41
|
SNAPSHOTS_DIR = CONFIG_DIR / "snapshots"
|
|
37
42
|
AUDIT_STATE_PATH = CONFIG_DIR / "audit_state.json"
|
|
38
43
|
AUTOCORRECT_STATE_PATH = CONFIG_DIR / "autocorrect_state.json"
|
|
@@ -79,6 +84,17 @@ class OutputConfig:
|
|
|
79
84
|
punctuation_cleanup: bool = True
|
|
80
85
|
number_conversion: bool = True
|
|
81
86
|
voice_input_prefix: str = "" # e.g. "[voice]" — empty disables
|
|
87
|
+
# Incremental finalization: once the un-finalized audio tail grows past
|
|
88
|
+
# this many seconds AND ends in silence, beam-decode and freeze it during
|
|
89
|
+
# recording, so interim passes and the stop-time final decode only cover
|
|
90
|
+
# the short remaining tail. 0 disables (full re-decode at stop).
|
|
91
|
+
streaming_freeze_secs: float = 25.0
|
|
92
|
+
# Mirror transcribed text to the system clipboard so it can be pasted:
|
|
93
|
+
# "off" — never
|
|
94
|
+
# "final" — the corrected final text, once ready
|
|
95
|
+
# "live" — also the best-so-far text at each streaming update and the
|
|
96
|
+
# instant recording stops (pasteable while the final decode runs)
|
|
97
|
+
copy_to_clipboard: str = "final"
|
|
82
98
|
|
|
83
99
|
|
|
84
100
|
@dataclass
|
|
@@ -173,7 +189,14 @@ class DataConfig:
|
|
|
173
189
|
"""
|
|
174
190
|
retain_audio: bool = True
|
|
175
191
|
max_audio_mb: int = 4096 # prune oldest recordings beyond this
|
|
192
|
+
# Never let retention squeeze the disk: skip saving and prune harder
|
|
193
|
+
# when the filesystem has less than this much free space.
|
|
194
|
+
min_free_gb: float = 5.0
|
|
176
195
|
capture_context: bool = True # best-effort active-window title
|
|
196
|
+
# Per-pass streaming decode traces + postcorrect before/after pairs
|
|
197
|
+
# (streaming_trace.jsonl / postcorrect_pairs.jsonl) for debugging,
|
|
198
|
+
# profiling, and future fine-tuning.
|
|
199
|
+
capture_intermediates: bool = True
|
|
177
200
|
|
|
178
201
|
|
|
179
202
|
@dataclass
|
|
@@ -277,6 +300,8 @@ def _content_files() -> list[Path]:
|
|
|
277
300
|
HISTORY_PATH,
|
|
278
301
|
CORRECTIONS_AUDIT_PATH,
|
|
279
302
|
METRICS_PATH,
|
|
303
|
+
TRACES_PATH,
|
|
304
|
+
POSTCORRECT_PAIRS_PATH,
|
|
280
305
|
AUDIT_STATE_PATH,
|
|
281
306
|
AUTOCORRECT_STATE_PATH,
|
|
282
307
|
CONSENT_PATH,
|
voiceio/postcorrect.py
CHANGED
|
@@ -15,6 +15,7 @@ from __future__ import annotations
|
|
|
15
15
|
import dataclasses
|
|
16
16
|
import difflib
|
|
17
17
|
import logging
|
|
18
|
+
import threading
|
|
18
19
|
import time
|
|
19
20
|
|
|
20
21
|
from voiceio.config import AutocorrectConfig, Config
|
|
@@ -90,6 +91,13 @@ class PostCorrector:
|
|
|
90
91
|
self._vocabulary = ""
|
|
91
92
|
self._recent: list[str] = []
|
|
92
93
|
self._context: str | None = None
|
|
94
|
+
# Context actually sent with the in-progress correct() call (what
|
|
95
|
+
# _record must persist — may differ from self._context).
|
|
96
|
+
self._effective_ctx: str | None = None
|
|
97
|
+
# A worker abandoned at the deadline may block indefinitely on a hung
|
|
98
|
+
# endpoint; cap leakage at one thread/socket by skipping new calls
|
|
99
|
+
# while it is still alive.
|
|
100
|
+
self._abandoned: threading.Thread | None = None
|
|
93
101
|
|
|
94
102
|
# ── availability ────────────────────────────────────────────────────
|
|
95
103
|
|
|
@@ -140,6 +148,27 @@ class PostCorrector:
|
|
|
140
148
|
parts.append(f"Transcript to correct:\n{text}")
|
|
141
149
|
return "\n\n".join(parts)
|
|
142
150
|
|
|
151
|
+
def _record(self, before: str, after: str | None, outcome: str) -> None:
|
|
152
|
+
"""Persist one LLM attempt (before/after/outcome) as training data.
|
|
153
|
+
|
|
154
|
+
Pairs land in postcorrect_pairs.jsonl — unlike the rotating log they
|
|
155
|
+
survive, so accepted AND rejected corrections stay available for
|
|
156
|
+
tuning guards or training a local corrector later.
|
|
157
|
+
"""
|
|
158
|
+
if not self._cfg.data.capture_intermediates:
|
|
159
|
+
return
|
|
160
|
+
from voiceio import retention
|
|
161
|
+
from voiceio.config import POSTCORRECT_PAIRS_PATH
|
|
162
|
+
retention.append_jsonl(POSTCORRECT_PAIRS_PATH, {
|
|
163
|
+
"ts": time.time(),
|
|
164
|
+
"before": before,
|
|
165
|
+
"after": after,
|
|
166
|
+
"outcome": outcome,
|
|
167
|
+
"secs": round(self.last_secs, 3) if self.last_secs is not None else None,
|
|
168
|
+
"model": self._pc.model or self._ac.model,
|
|
169
|
+
"context": self._effective_ctx,
|
|
170
|
+
})
|
|
171
|
+
|
|
143
172
|
def correct(
|
|
144
173
|
self, text: str, *, vocabulary: str = "",
|
|
145
174
|
recent: list[str] | None = None, context: str | None = None,
|
|
@@ -161,28 +190,65 @@ class PostCorrector:
|
|
|
161
190
|
vocab = vocabulary or self._vocabulary
|
|
162
191
|
rec = recent if recent is not None else self._recent
|
|
163
192
|
ctx = context if context is not None else self._context
|
|
193
|
+
self._effective_ctx = ctx
|
|
194
|
+
|
|
195
|
+
if self._abandoned is not None:
|
|
196
|
+
if self._abandoned.is_alive():
|
|
197
|
+
log.warning(
|
|
198
|
+
"PostCorrector: previous request still hung — skipping this one",
|
|
199
|
+
)
|
|
200
|
+
self._record(text, None, "skipped_busy")
|
|
201
|
+
return text
|
|
202
|
+
self._abandoned = None
|
|
164
203
|
|
|
165
204
|
from voiceio.llm_api import chat
|
|
166
205
|
user_msg = self._build_user_message(text, vocab, rec, ctx)
|
|
206
|
+
# timeout_secs must bound the WALL CLOCK the user waits, but urllib's
|
|
207
|
+
# timeout is per-socket-read — a slowly streaming response can run
|
|
208
|
+
# far past it (observed ~14s with an 8s config). Run the call in a
|
|
209
|
+
# thread and abandon it at the deadline.
|
|
167
210
|
t0 = time.monotonic()
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
211
|
+
outcome: dict = {}
|
|
212
|
+
|
|
213
|
+
def _call() -> None:
|
|
214
|
+
try:
|
|
215
|
+
outcome["response"] = chat(
|
|
216
|
+
self._client_cfg(), _SYSTEM_PROMPT, user_msg, max_tokens=1024,
|
|
217
|
+
)
|
|
218
|
+
except Exception as e:
|
|
219
|
+
outcome["error"] = e
|
|
220
|
+
|
|
221
|
+
worker = threading.Thread(target=_call, daemon=True)
|
|
222
|
+
worker.start()
|
|
223
|
+
worker.join(self._pc.timeout_secs)
|
|
224
|
+
self.last_secs = time.monotonic() - t0
|
|
225
|
+
if worker.is_alive():
|
|
226
|
+
log.debug(
|
|
227
|
+
"PostCorrector deadline (%.1fs) exceeded — keeping original",
|
|
228
|
+
self._pc.timeout_secs,
|
|
229
|
+
)
|
|
230
|
+
self._abandoned = worker
|
|
231
|
+
self._record(text, None, "timeout")
|
|
232
|
+
return text
|
|
233
|
+
if "error" in outcome:
|
|
234
|
+
log.debug("PostCorrector LLM error: %s — keeping original", outcome["error"])
|
|
235
|
+
self._record(text, None, "error")
|
|
172
236
|
return text
|
|
173
|
-
|
|
174
|
-
self.last_secs = time.monotonic() - t0
|
|
237
|
+
response = outcome.get("response")
|
|
175
238
|
|
|
176
239
|
if not response:
|
|
177
240
|
log.debug("PostCorrector: empty/failed response — keeping original")
|
|
241
|
+
self._record(text, None, "empty")
|
|
178
242
|
return text
|
|
179
243
|
|
|
180
244
|
corrected = _strip_wrapping(response)
|
|
181
245
|
if not corrected:
|
|
182
246
|
log.debug("PostCorrector: response empty after stripping — keeping original")
|
|
247
|
+
self._record(text, None, "empty")
|
|
183
248
|
return text
|
|
184
249
|
|
|
185
250
|
if corrected == text:
|
|
251
|
+
self._record(text, corrected, "unchanged")
|
|
186
252
|
return text
|
|
187
253
|
|
|
188
254
|
# Guard: word-count must not change materially.
|
|
@@ -192,6 +258,7 @@ class PostCorrector:
|
|
|
192
258
|
"PostCorrector reject: word count %d→%d (>%.0f%%) — keeping original",
|
|
193
259
|
orig_wc, new_wc, _MAX_WORDCOUNT_DELTA * 100,
|
|
194
260
|
)
|
|
261
|
+
self._record(text, corrected, "rejected_wordcount")
|
|
195
262
|
return text
|
|
196
263
|
|
|
197
264
|
# Guard: only a small fraction of words may change.
|
|
@@ -201,7 +268,9 @@ class PostCorrector:
|
|
|
201
268
|
"PostCorrector reject: edit ratio %.2f > %.2f — keeping original",
|
|
202
269
|
ratio, _MAX_EDIT_RATIO,
|
|
203
270
|
)
|
|
271
|
+
self._record(text, corrected, "rejected_editratio")
|
|
204
272
|
return text
|
|
205
273
|
|
|
206
274
|
log.info("PostCorrector fixed: %s", ", ".join(_changed_words(text, corrected)))
|
|
275
|
+
self._record(text, corrected, "applied")
|
|
207
276
|
return corrected
|
voiceio/recorder.py
CHANGED
|
@@ -162,6 +162,17 @@ class AudioRecorder:
|
|
|
162
162
|
return False, f"no audio callback for {stale:.1f}s"
|
|
163
163
|
return True, ""
|
|
164
164
|
|
|
165
|
+
def is_zombie(self) -> bool:
|
|
166
|
+
"""Stream firing callbacks but delivering only digital zeros.
|
|
167
|
+
|
|
168
|
+
The post-suspend PortAudio failure mode: the heartbeat stays healthy
|
|
169
|
+
while the capture node is gone. Distinct from a merely-empty ring
|
|
170
|
+
(fresh stream) or a muted mic (noise floor > 0 on real hardware).
|
|
171
|
+
"""
|
|
172
|
+
if self._ring._filled < self._ring._max:
|
|
173
|
+
return False # ring not yet full — can't judge
|
|
174
|
+
return not self.has_signal()
|
|
175
|
+
|
|
165
176
|
def has_signal(self) -> bool:
|
|
166
177
|
"""Check if the pre-buffer ring contains non-silence audio.
|
|
167
178
|
|
voiceio/retention.py
CHANGED
|
@@ -7,17 +7,19 @@ mining measurable and personal fine-tuning possible later.
|
|
|
7
7
|
from __future__ import annotations
|
|
8
8
|
|
|
9
9
|
import functools
|
|
10
|
+
import json
|
|
10
11
|
import logging
|
|
11
12
|
import shutil
|
|
12
13
|
import subprocess
|
|
13
14
|
import time
|
|
14
15
|
import wave
|
|
16
|
+
from pathlib import Path
|
|
15
17
|
from typing import TYPE_CHECKING
|
|
16
18
|
|
|
17
19
|
import numpy as np
|
|
18
20
|
|
|
19
21
|
from voiceio import config
|
|
20
|
-
from voiceio.config import RECORDINGS_DIR
|
|
22
|
+
from voiceio.config import RECORDINGS_DIR, TRACES_PATH
|
|
21
23
|
|
|
22
24
|
if TYPE_CHECKING:
|
|
23
25
|
from voiceio.config import DataConfig
|
|
@@ -27,11 +29,25 @@ log = logging.getLogger(__name__)
|
|
|
27
29
|
_which = functools.lru_cache(maxsize=8)(shutil.which)
|
|
28
30
|
|
|
29
31
|
|
|
32
|
+
def _free_gb(path: Path) -> float:
|
|
33
|
+
"""Free space (GB) on the filesystem holding `path`. Inf on failure."""
|
|
34
|
+
try:
|
|
35
|
+
return shutil.disk_usage(path).free / 1024**3
|
|
36
|
+
except OSError:
|
|
37
|
+
return float("inf")
|
|
38
|
+
|
|
39
|
+
|
|
30
40
|
def save_audio(audio: np.ndarray, ts: float, cfg: DataConfig) -> str | None:
|
|
31
41
|
"""Persist one utterance as 16kHz mono int16 WAV. Returns the filename
|
|
32
42
|
(relative to the recordings dir) or None if disabled/failed."""
|
|
33
43
|
if not cfg.retain_audio or audio is None or len(audio) == 0:
|
|
34
44
|
return None
|
|
45
|
+
if _free_gb(RECORDINGS_DIR.parent) < cfg.min_free_gb:
|
|
46
|
+
log.warning(
|
|
47
|
+
"Disk has <%.0fGB free — skipping audio retention for this utterance",
|
|
48
|
+
cfg.min_free_gb,
|
|
49
|
+
)
|
|
50
|
+
return None
|
|
35
51
|
name = time.strftime("%Y%m%d-%H%M%S", time.localtime(ts)) + f"-{int(ts * 1000) % 1000:03d}.wav"
|
|
36
52
|
path = RECORDINGS_DIR / name
|
|
37
53
|
try:
|
|
@@ -70,6 +86,15 @@ def prune(cfg: DataConfig) -> None:
|
|
|
70
86
|
)
|
|
71
87
|
total = sum(p.stat().st_size for p in files)
|
|
72
88
|
budget = cfg.max_audio_mb * 1024 * 1024
|
|
89
|
+
# Below the free-disk floor, shrink the budget so retention yields
|
|
90
|
+
# space back instead of holding its full cap on a squeezed disk.
|
|
91
|
+
free = _free_gb(RECORDINGS_DIR.parent)
|
|
92
|
+
if free < cfg.min_free_gb:
|
|
93
|
+
budget = min(budget, total // 2)
|
|
94
|
+
log.warning(
|
|
95
|
+
"Disk has %.1fGB free (<%.0fGB floor) — pruning recordings to %.0fMB",
|
|
96
|
+
free, cfg.min_free_gb, budget / 1024 / 1024,
|
|
97
|
+
)
|
|
73
98
|
while total > budget and files:
|
|
74
99
|
oldest = files.pop(0)
|
|
75
100
|
total -= oldest.stat().st_size
|
|
@@ -79,6 +104,45 @@ def prune(cfg: DataConfig) -> None:
|
|
|
79
104
|
log.debug("Recording prune failed", exc_info=True)
|
|
80
105
|
|
|
81
106
|
|
|
107
|
+
_JSONL_MAX_BYTES = 64 * 1024 * 1024 # per capture file; oldest half dropped
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def append_jsonl(path: Path, entry: dict) -> None:
|
|
111
|
+
"""Append one JSON line to a 0600 data file. Best-effort, never raises.
|
|
112
|
+
|
|
113
|
+
Capture files are bounded: past _JSONL_MAX_BYTES the oldest half of the
|
|
114
|
+
lines is dropped, so intermediate-data capture can never grow unbounded
|
|
115
|
+
on a disk the min_free_gb floor is trying to protect.
|
|
116
|
+
"""
|
|
117
|
+
try:
|
|
118
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
119
|
+
config._chmod(path.parent, config._SECURE_DIR)
|
|
120
|
+
newly_created = not path.exists()
|
|
121
|
+
if not newly_created and path.stat().st_size > _JSONL_MAX_BYTES:
|
|
122
|
+
lines = path.read_text(encoding="utf-8").splitlines(keepends=True)
|
|
123
|
+
path.write_text("".join(lines[len(lines) // 2:]), encoding="utf-8")
|
|
124
|
+
log.info("Trimmed %s to newest %d entries", path.name, len(lines) - len(lines) // 2)
|
|
125
|
+
with open(path, "a", encoding="utf-8") as f:
|
|
126
|
+
f.write(json.dumps(entry, ensure_ascii=False) + "\n")
|
|
127
|
+
if newly_created:
|
|
128
|
+
config._chmod(path, config._SECURE_FILE)
|
|
129
|
+
except (OSError, TypeError, ValueError) as e:
|
|
130
|
+
log.warning("Failed to write %s: %s", path.name, e)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def save_trace(cfg: DataConfig, entry: dict) -> None:
|
|
134
|
+
"""Persist one utterance's per-pass streaming decode trace.
|
|
135
|
+
|
|
136
|
+
One line per utterance in streaming_trace.jsonl: pass timings, kinds
|
|
137
|
+
(interim/freeze/final), tail lengths, and raw tail texts — the data
|
|
138
|
+
needed to debug/profile streaming behaviour and to train on interim
|
|
139
|
+
hypotheses later. Linked to history via ts + audio filename.
|
|
140
|
+
"""
|
|
141
|
+
if not cfg.capture_intermediates:
|
|
142
|
+
return
|
|
143
|
+
append_jsonl(TRACES_PATH, entry)
|
|
144
|
+
|
|
145
|
+
|
|
82
146
|
def active_window_title() -> str | None:
|
|
83
147
|
"""Best-effort title of the focused window (dictation target context).
|
|
84
148
|
|
voiceio/streaming.py
CHANGED
|
@@ -8,11 +8,12 @@ import threading
|
|
|
8
8
|
import time
|
|
9
9
|
from typing import TYPE_CHECKING, Callable
|
|
10
10
|
|
|
11
|
+
import numpy as np
|
|
12
|
+
|
|
11
13
|
from voiceio.transcriber import transcribe_timeout
|
|
12
14
|
from voiceio.typers.base import StreamingTyper
|
|
13
15
|
|
|
14
16
|
if TYPE_CHECKING:
|
|
15
|
-
import numpy as np
|
|
16
17
|
from voiceio.commands import CommandProcessor
|
|
17
18
|
from voiceio.corrections import CorrectionDict
|
|
18
19
|
from voiceio.llm import LLMProcessor
|
|
@@ -24,6 +25,32 @@ if TYPE_CHECKING:
|
|
|
24
25
|
log = logging.getLogger(__name__)
|
|
25
26
|
DELETE_SETTLE_SECS = 0.05 # delay between delete and type for ydotool reliability
|
|
26
27
|
|
|
28
|
+
# Incremental finalization: a freeze cut is only safe at a speech pause, so
|
|
29
|
+
# require the tail's trailing window to be quiet before cutting there.
|
|
30
|
+
_FREEZE_SILENCE_WINDOW_SECS = 0.3
|
|
31
|
+
# Quiet is judged RELATIVE to the tail's own level: mic gains vary wildly
|
|
32
|
+
# and the decode path normalizes audio anyway, so an absolute threshold
|
|
33
|
+
# would read a quiet mic's speech as silence (mid-word cuts) and a hot
|
|
34
|
+
# noise floor as speech (freeze never fires).
|
|
35
|
+
_FREEZE_SILENCE_RATIO = 0.15
|
|
36
|
+
_FREEZE_SILENCE_FLOOR = 1e-4 # digital silence is always quiet
|
|
37
|
+
# Whisper conditions on ~224 prompt tokens; more frozen context is wasted.
|
|
38
|
+
_FREEZE_CONTEXT_CHARS = 400
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _rms(samples: np.ndarray) -> float:
|
|
42
|
+
return float(np.sqrt(np.mean(np.square(samples, dtype=np.float64))))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _tail_ends_in_silence(tail: np.ndarray, sample_rate: int) -> bool:
|
|
46
|
+
"""True when the last _FREEZE_SILENCE_WINDOW_SECS of audio are quiet
|
|
47
|
+
relative to the tail's overall level."""
|
|
48
|
+
window = tail[-int(_FREEZE_SILENCE_WINDOW_SECS * sample_rate):]
|
|
49
|
+
if len(window) == 0:
|
|
50
|
+
return False
|
|
51
|
+
threshold = max(_FREEZE_SILENCE_RATIO * _rms(tail), _FREEZE_SILENCE_FLOOR)
|
|
52
|
+
return _rms(window) < threshold
|
|
53
|
+
|
|
27
54
|
|
|
28
55
|
def _common_prefix_len(a: str, b: str) -> int:
|
|
29
56
|
"""Length of the longest common prefix between two strings."""
|
|
@@ -34,6 +61,16 @@ def _common_prefix_len(a: str, b: str) -> int:
|
|
|
34
61
|
return limit
|
|
35
62
|
|
|
36
63
|
|
|
64
|
+
def _join_text(a: str, b: str) -> str:
|
|
65
|
+
"""Join two transcript fragments with a single space."""
|
|
66
|
+
a, b = a.strip(), (b or "").strip()
|
|
67
|
+
if not a:
|
|
68
|
+
return b
|
|
69
|
+
if not b:
|
|
70
|
+
return a
|
|
71
|
+
return a + " " + b
|
|
72
|
+
|
|
73
|
+
|
|
37
74
|
def _clean_word(w: str) -> str:
|
|
38
75
|
"""Strip punctuation for fuzzy word matching."""
|
|
39
76
|
return re.sub(r'[^\w]', '', w).lower()
|
|
@@ -84,6 +121,8 @@ class StreamingSession:
|
|
|
84
121
|
voice_input_prefix: str = "",
|
|
85
122
|
on_typer_broken: Callable[[], None] | None = None,
|
|
86
123
|
is_current: Callable[[], bool] | None = None,
|
|
124
|
+
on_interim: Callable[[str], None] | None = None,
|
|
125
|
+
freeze_secs: float = 25.0,
|
|
87
126
|
):
|
|
88
127
|
self._transcriber = transcriber
|
|
89
128
|
self._typer = typer
|
|
@@ -104,6 +143,7 @@ class StreamingSession:
|
|
|
104
143
|
self._llm = llm
|
|
105
144
|
self._voice_input_prefix = voice_input_prefix
|
|
106
145
|
self._on_typer_broken = on_typer_broken
|
|
146
|
+
self._on_interim = on_interim
|
|
107
147
|
self._typer_fail_count = 0
|
|
108
148
|
self._typer_broken_signalled = False
|
|
109
149
|
self._typed_text = ""
|
|
@@ -111,10 +151,26 @@ class StreamingSession:
|
|
|
111
151
|
self._stop_event = threading.Event()
|
|
112
152
|
self._worker_thread: threading.Thread | None = None
|
|
113
153
|
self._final_audio: np.ndarray | None = None # set on stop
|
|
154
|
+
# Incremental finalization: audio before _frozen_samples has already
|
|
155
|
+
# been beam-decoded into _frozen_raw and is never decoded again.
|
|
156
|
+
# Interim and final passes only decode the tail after this offset.
|
|
157
|
+
self._freeze_secs = freeze_secs
|
|
158
|
+
self._frozen_samples = 0
|
|
159
|
+
self._frozen_raw = ""
|
|
160
|
+
self._frozen_segments: list[dict] = []
|
|
114
161
|
# Raw (pre-pipeline) text + confidence of the final pass, for history
|
|
115
162
|
self.raw_final_text: str | None = None
|
|
116
163
|
self.final_latency: dict = {}
|
|
117
164
|
self.final_segments: list[dict] = []
|
|
165
|
+
# Per-pass decode trace (interim/freeze/final) for debugging,
|
|
166
|
+
# profiling, and future training. Persisted via retention.save_trace.
|
|
167
|
+
self.trace: list[dict] = []
|
|
168
|
+
self._t0 = time.monotonic()
|
|
169
|
+
|
|
170
|
+
@property
|
|
171
|
+
def interim_text(self) -> str:
|
|
172
|
+
"""Best-so-far text (post-pipeline). Safe to read while finalizing."""
|
|
173
|
+
return self._typed_text
|
|
118
174
|
|
|
119
175
|
def set_is_current(self, is_current: Callable[[], bool]) -> None:
|
|
120
176
|
"""Install the output-ownership gate (see __init__).
|
|
@@ -206,10 +262,22 @@ class StreamingSession:
|
|
|
206
262
|
log.exception("Final transcribe/apply error")
|
|
207
263
|
self._final_audio = None # release memory
|
|
208
264
|
|
|
265
|
+
def _frozen_context(self) -> str | None:
|
|
266
|
+
"""Frozen-text suffix used to condition the tail decode."""
|
|
267
|
+
if not self._frozen_raw:
|
|
268
|
+
return None
|
|
269
|
+
return self._frozen_raw[-_FREEZE_CONTEXT_CHARS:]
|
|
270
|
+
|
|
209
271
|
def _transcribe_and_apply(
|
|
210
272
|
self, min_seconds: float = 1.0, final: bool = False,
|
|
211
273
|
) -> None:
|
|
212
|
-
"""Get audio, transcribe, apply correction.
|
|
274
|
+
"""Get audio, transcribe the un-frozen tail, apply correction.
|
|
275
|
+
|
|
276
|
+
Incremental finalization: audio before _frozen_samples was already
|
|
277
|
+
beam-decoded and frozen, so every pass — including the final one —
|
|
278
|
+
only decodes the tail. Long dictations finalize in O(tail), not
|
|
279
|
+
O(recording).
|
|
280
|
+
"""
|
|
213
281
|
if final and self._final_audio is not None:
|
|
214
282
|
# Use the snapshot passed to stop() — recorder may be gone
|
|
215
283
|
audio = self._final_audio
|
|
@@ -223,22 +291,84 @@ class StreamingSession:
|
|
|
223
291
|
if len(audio) < self._sample_rate * min_seconds:
|
|
224
292
|
return
|
|
225
293
|
|
|
294
|
+
tail = audio[self._frozen_samples:]
|
|
295
|
+
if not final and len(tail) < self._sample_rate * min_seconds:
|
|
296
|
+
return # not enough new audio since the last freeze
|
|
297
|
+
|
|
298
|
+
# Freeze when the tail has grown long AND ends at a speech pause
|
|
299
|
+
# (never cut mid-word). The freeze pass gets beam search because its
|
|
300
|
+
# text is never decoded again.
|
|
301
|
+
freeze = (
|
|
302
|
+
not final
|
|
303
|
+
and self._freeze_secs > 0
|
|
304
|
+
and len(tail) >= self._freeze_secs * self._sample_rate
|
|
305
|
+
and _tail_ends_in_silence(tail, self._sample_rate)
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
tail_text = ""
|
|
309
|
+
tail_segments: list[dict] = []
|
|
226
310
|
t0 = time.monotonic()
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
311
|
+
# The final pass decodes ANY remaining audio (a clipped last word
|
|
312
|
+
# must not be dropped); interim/freeze passes skip sub-0.3s tails.
|
|
313
|
+
if len(tail) > (0 if final else self._sample_rate * 0.3):
|
|
314
|
+
try:
|
|
315
|
+
tail_text = self._transcriber.transcribe(
|
|
316
|
+
tail, final=final or freeze, context=self._frozen_context(),
|
|
317
|
+
)
|
|
318
|
+
except Exception:
|
|
319
|
+
# Decode FAILED (timeout/crash) — this is not silence. Never
|
|
320
|
+
# advance the freeze boundary or char-diff against a partial
|
|
321
|
+
# transcript; on final, commit what is already on screen.
|
|
322
|
+
log.exception("Streaming transcription failed")
|
|
323
|
+
if final:
|
|
324
|
+
self.raw_final_text = ""
|
|
325
|
+
self.final_latency = {
|
|
326
|
+
"audio_secs": round(len(audio) / self._sample_rate, 2),
|
|
327
|
+
"frozen_secs": round(self._frozen_samples / self._sample_rate, 2),
|
|
328
|
+
"tail_secs": round(len(tail) / self._sample_rate, 2),
|
|
329
|
+
"transcribe": round(time.monotonic() - t0, 3),
|
|
330
|
+
"decode_failed": True,
|
|
331
|
+
}
|
|
332
|
+
self._commit_interim_on_final()
|
|
333
|
+
return
|
|
334
|
+
tail_segments = list(getattr(self._transcriber, "last_segments", []))
|
|
232
335
|
t_transcribe = time.monotonic() - t0
|
|
233
336
|
|
|
337
|
+
self.trace.append({
|
|
338
|
+
"t": round(time.monotonic() - self._t0, 2),
|
|
339
|
+
"kind": "final" if final else ("freeze" if freeze else "interim"),
|
|
340
|
+
"frozen_secs": round(self._frozen_samples / self._sample_rate, 2),
|
|
341
|
+
"tail_secs": round(len(tail) / self._sample_rate, 2),
|
|
342
|
+
"secs": round(t_transcribe, 3),
|
|
343
|
+
"text": tail_text,
|
|
344
|
+
})
|
|
345
|
+
|
|
346
|
+
if freeze:
|
|
347
|
+
self._frozen_raw = _join_text(self._frozen_raw, tail_text)
|
|
348
|
+
self._frozen_samples = len(audio)
|
|
349
|
+
self._frozen_segments.extend(tail_segments)
|
|
350
|
+
log.debug(
|
|
351
|
+
"Froze %.1fs of audio (frozen text now %d chars)",
|
|
352
|
+
len(tail) / self._sample_rate, len(self._frozen_raw),
|
|
353
|
+
)
|
|
354
|
+
raw = self._frozen_raw if freeze else _join_text(self._frozen_raw, tail_text)
|
|
355
|
+
|
|
356
|
+
# An interim pass that decoded nothing must not touch the display:
|
|
357
|
+
# rewriting to frozen-only text would visibly delete the un-frozen
|
|
358
|
+
# words the previous pass just showed (transient decoder flake).
|
|
359
|
+
if not final and not tail_text:
|
|
360
|
+
return
|
|
361
|
+
|
|
234
362
|
if final:
|
|
235
|
-
self.raw_final_text =
|
|
236
|
-
self.final_segments =
|
|
363
|
+
self.raw_final_text = raw
|
|
364
|
+
self.final_segments = self._frozen_segments + tail_segments
|
|
237
365
|
self.final_latency = {
|
|
238
366
|
"audio_secs": round(len(audio) / self._sample_rate, 2),
|
|
367
|
+
"frozen_secs": round(self._frozen_samples / self._sample_rate, 2),
|
|
368
|
+
"tail_secs": round(len(tail) / self._sample_rate, 2),
|
|
239
369
|
"transcribe": round(t_transcribe, 3),
|
|
240
370
|
}
|
|
241
|
-
if not
|
|
371
|
+
if not raw:
|
|
242
372
|
# The final pass produced nothing. If we already have interim
|
|
243
373
|
# text (e.g. the final transcription timed out on a long
|
|
244
374
|
# dictation), commit that instead of silently dropping it —
|
|
@@ -246,11 +376,11 @@ class StreamingSession:
|
|
|
246
376
|
self._commit_interim_on_final()
|
|
247
377
|
return
|
|
248
378
|
|
|
249
|
-
if
|
|
379
|
+
if raw:
|
|
250
380
|
from voiceio.postprocess import apply_pipeline
|
|
251
381
|
t1 = time.monotonic()
|
|
252
382
|
text, abort = apply_pipeline(
|
|
253
|
-
|
|
383
|
+
raw,
|
|
254
384
|
do_cleanup=self._cleanup,
|
|
255
385
|
number_conversion=self._number_conversion,
|
|
256
386
|
language=self._language,
|
|
@@ -267,6 +397,11 @@ class StreamingSession:
|
|
|
267
397
|
if pc_secs is not None:
|
|
268
398
|
self.final_latency["postcorrect"] = round(pc_secs, 3)
|
|
269
399
|
if abort:
|
|
400
|
+
# A voice command wiped the utterance — discard frozen text
|
|
401
|
+
# too, or the next pass would resurrect it.
|
|
402
|
+
self._frozen_raw = ""
|
|
403
|
+
self._frozen_segments = []
|
|
404
|
+
self._frozen_samples = len(audio)
|
|
270
405
|
# A superseded session must not touch the typer on the final
|
|
271
406
|
# path (would corrupt the newer session's output).
|
|
272
407
|
if final and not self._is_current():
|
|
@@ -280,7 +415,13 @@ class StreamingSession:
|
|
|
280
415
|
return
|
|
281
416
|
|
|
282
417
|
if text:
|
|
418
|
+
prev = self._typed_text
|
|
283
419
|
self._apply_correction(text, final=final)
|
|
420
|
+
if not final and self._on_interim and self._typed_text != prev:
|
|
421
|
+
try:
|
|
422
|
+
self._on_interim(self._typed_text)
|
|
423
|
+
except Exception:
|
|
424
|
+
log.debug("on_interim callback failed", exc_info=True)
|
|
284
425
|
|
|
285
426
|
def _commit_interim_on_final(self) -> None:
|
|
286
427
|
"""Commit whatever interim text we have when the final pass yields none.
|
voiceio/transcriber.py
CHANGED
|
@@ -18,6 +18,12 @@ if TYPE_CHECKING:
|
|
|
18
18
|
log = logging.getLogger(__name__)
|
|
19
19
|
|
|
20
20
|
TRANSCRIBE_TIMEOUT = 30 # seconds (floor; scaled up for long audio)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class TranscriptionError(RuntimeError):
|
|
24
|
+
"""Decode failed (timeout / worker crash). Distinct from a legitimate
|
|
25
|
+
empty result ("" = silence) so callers never treat lost audio as decoded
|
|
26
|
+
silence — the streaming freeze/final paths must not advance state on it."""
|
|
21
27
|
# Realtime-factor headroom for the read timeout. Whisper decodes well faster
|
|
22
28
|
# than realtime, so 1.5x the audio duration is a generous ceiling that still
|
|
23
29
|
# never kills a long dictation mid-decode.
|
|
@@ -152,10 +158,17 @@ class Transcriber:
|
|
|
152
158
|
log.warning("Worker died, restarting (attempt %d/%d)", self._restarts, MAX_RESTARTS)
|
|
153
159
|
self._start_worker()
|
|
154
160
|
|
|
155
|
-
def transcribe(
|
|
161
|
+
def transcribe(
|
|
162
|
+
self, audio: np.ndarray, final: bool = False, context: str | None = None,
|
|
163
|
+
) -> str:
|
|
156
164
|
"""Transcribe audio. `final=True` marks the pass whose text the user
|
|
157
165
|
keeps (streaming final / batch): it gets beam search; interim streaming
|
|
158
|
-
passes stay greedy for latency.
|
|
166
|
+
passes stay greedy for latency.
|
|
167
|
+
|
|
168
|
+
`context` is per-call text appended to the initial_prompt — used by
|
|
169
|
+
incremental finalization to condition a tail decode on the already-
|
|
170
|
+
frozen transcript so sentences stay coherent across the cut.
|
|
171
|
+
"""
|
|
159
172
|
with self._lock:
|
|
160
173
|
self._ensure_worker()
|
|
161
174
|
|
|
@@ -165,8 +178,9 @@ class Transcriber:
|
|
|
165
178
|
audio = normalize_audio(audio)
|
|
166
179
|
audio_b64 = base64.b64encode(audio.tobytes()).decode("ascii")
|
|
167
180
|
req = {"audio_b64": audio_b64, "options": {"beam_size": 5 if final else 1}}
|
|
168
|
-
|
|
169
|
-
|
|
181
|
+
prompt = "\n".join(p for p in (self._initial_prompt, context) if p)
|
|
182
|
+
if prompt:
|
|
183
|
+
req["initial_prompt"] = prompt
|
|
170
184
|
if self._hotwords:
|
|
171
185
|
req["hotwords"] = self._hotwords
|
|
172
186
|
try:
|
|
@@ -184,16 +198,18 @@ class Transcriber:
|
|
|
184
198
|
timeout = transcribe_timeout(duration)
|
|
185
199
|
result_line = self._read_with_timeout(timeout)
|
|
186
200
|
if result_line is None:
|
|
201
|
+
self.last_segments = []
|
|
187
202
|
log.warning("Transcription timed out after %.0fs, restarting worker", timeout)
|
|
188
203
|
self._kill_worker()
|
|
189
204
|
self._ensure_worker()
|
|
190
|
-
|
|
205
|
+
raise TranscriptionError(f"decode timed out after {timeout:.0f}s")
|
|
191
206
|
|
|
192
207
|
try:
|
|
193
208
|
result = json.loads(result_line)
|
|
194
209
|
except (json.JSONDecodeError, TypeError):
|
|
210
|
+
self.last_segments = []
|
|
195
211
|
log.warning("Invalid response from worker: %s", repr(result_line)[:100])
|
|
196
|
-
|
|
212
|
+
raise TranscriptionError("invalid worker response")
|
|
197
213
|
text = result.get("text", "")
|
|
198
214
|
self.last_segments = result.get("segments", [])
|
|
199
215
|
|
voiceio/wizard.py
CHANGED
|
@@ -680,7 +680,7 @@ def _write_config(
|
|
|
680
680
|
"model": autocorrect_model or "moonshotai/kimi-k2-0905",
|
|
681
681
|
})
|
|
682
682
|
|
|
683
|
-
CONFIG_PATH.write_text(_dump_toml(cfg))
|
|
683
|
+
CONFIG_PATH.write_text(_dump_toml(cfg), encoding="utf-8")
|
|
684
684
|
_secure_config_permissions()
|
|
685
685
|
if not quiet:
|
|
686
686
|
print(f"\n {GREEN}✓{RESET} Config saved to {DIM}{CONFIG_PATH}{RESET}")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|