python-voiceio 0.4.1__py3-none-any.whl → 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: python-voiceio
3
- Version: 0.4.1
3
+ Version: 0.6.0
4
4
  Summary: Voice dictation for Linux. Speak → text, locally, instantly.
5
5
  Author: Hugo Montenegro
6
6
  License-Expression: MIT
@@ -1,15 +1,15 @@
1
- python_voiceio-0.4.1.dist-info/licenses/LICENSE,sha256=Gz61o8jFTAvZUZyB3nWDB3DQQVuipjfPkVu9W8hBHM0,1072
2
- voiceio/__init__.py,sha256=pMtTmSUht-XtbR_7Doz6bsQqopJJd8rZ8I8zy2HwwoA,22
1
+ python_voiceio-0.6.0.dist-info/licenses/LICENSE,sha256=Gz61o8jFTAvZUZyB3nWDB3DQQVuipjfPkVu9W8hBHM0,1072
2
+ voiceio/__init__.py,sha256=cID1jLnC_vj48GgMN6Yb1FA3JsQ95zNmCHmRYE8TFhY,22
3
3
  voiceio/__main__.py,sha256=xT5QCGGreYMisHO7Lh_Y-xAQ2TOzG2D7npKknGNcWSY,53
4
- voiceio/app.py,sha256=ZVuvITD76lYauIB68XvItyOsB1I3kMSqy9YISp0gl9E,60833
4
+ voiceio/app.py,sha256=4o3nOv_Z_NCBr6KTHr1eSGT159ZE98dVBAy5wUOpzAo,64398
5
5
  voiceio/audit.py,sha256=vX3x-YAhmt6EZYRZd0_6wvTquZu5sxGMvH45BhSBnCk,14915
6
6
  voiceio/autocorrect.py,sha256=jGsglEYwAHWk2uWedeEZzyAMW2LWO_8nQGZgrWYy4Dc,26552
7
7
  voiceio/autocorrect_state.py,sha256=BNzdHP5slcOV_-OU4MlaKE7xgVJzrPncsYFK30JWTs8,6858
8
8
  voiceio/backends.py,sha256=-rs1YhPblub4NjFb6qCD7Ih9FqBRXtPuvjVv9bQoQ1w,368
9
9
  voiceio/cli.py,sha256=_HfcWQt2Jf34YfI4W1RavsyqegQy9jrg0nwLe6DvD6E,71639
10
- voiceio/clipboard_read.py,sha256=266u_QoJb_n3iLuy5fbccPZJfpYpPkjsNtZxjDNmqAE,2062
10
+ voiceio/clipboard_read.py,sha256=CqbHSvOztcQ3u2I2EzFm0jdcDVKv56gl0T1YfWxNpmI,3390
11
11
  voiceio/commands.py,sha256=Vhtn8s5G5OcCbRN44YZ-_14Fa85ISXl_3Ix-KqSHs2A,4694
12
- voiceio/config.py,sha256=PF_G8xnzsFTFLee-tXReJiWUpACg5Li6HM9k97dWjM8,12569
12
+ voiceio/config.py,sha256=AOFaJras9rUPRDz7xeaZd4-WbWLOPUe0f0hYcQNwdJM,14042
13
13
  voiceio/consent.py,sha256=jtdp1TkTbicUSTCDmV4qo4SdbS4kQpYzy1Kjc_7pJVw,1883
14
14
  voiceio/corrections.py,sha256=7moKhkVvo5CZU1mZobV2-MajkcHiwliKBY1dSBFh0w0,7445
15
15
  voiceio/demo.py,sha256=QRNJuNObXCsjwGD-Cz3dh81cfb21xiuSeVyyU720U9Q,8673
@@ -22,18 +22,18 @@ voiceio/llm_api.py,sha256=ex1brY10a8xNtw4cHCYC0yCth3Z02NkCQaHdB4XQPag,7497
22
22
  voiceio/numbers.py,sha256=MP8jag4_F2OqUU5H14KzOmR9L8gWg48yQhLoUbGg_YI,8376
23
23
  voiceio/pidlock.py,sha256=RInlF_8H7wR0nqkdlhUfxYs7OGzHIvb5p-XDjjQSTNg,619
24
24
  voiceio/platform.py,sha256=f_8MmC5q1sCEuEPyKjRqlVvV9FExuGmwn_aJXhkpiEk,11809
25
- voiceio/postcorrect.py,sha256=lxm3zxdz3yzoR6qpf0bKRycBrF5y91bVnU4DqyTRbxU,8163
25
+ voiceio/postcorrect.py,sha256=iyO7MPIR4a4XIASEGOs9Y3iUXVTfwtVkhVOWj2vY1lk,11201
26
26
  voiceio/postprocess.py,sha256=9grkq3I6XgiQV6ZQbIw2_fOUe57tzC7GBPvLXqhhdTs,2979
27
27
  voiceio/prompt.py,sha256=9iza8KoQxvO_U9NPMJ8Ep7j-zkIs9nW8N5sacCRpsEk,2904
28
- voiceio/recorder.py,sha256=aWCZC8sKYnUNnrzGxFlYzJlnUVIk8PNpkahVXeAUULE,12323
29
- voiceio/retention.py,sha256=VrN261qL4C2_mSbjI6yPnMNiTVMxpo16wYua27TQb_Q,3330
28
+ voiceio/recorder.py,sha256=p1r0mH_plBD3BHqBbIGV_m2N4Mj5RAF7LfMu8u6mk5I,12813
29
+ voiceio/retention.py,sha256=i_VWPCyykT4kQ1uKilTplpiIrACTYljn09BU1vL71A0,6019
30
30
  voiceio/service.py,sha256=jqX1opZyQh_dOJMK4bVf8Aoei0zOuuk1Ogta9BXTWGY,12106
31
31
  voiceio/snapshots.py,sha256=bm-lzIeYucDUDlPBZiFiWd-jyB2I6Ft1jRANZCO6go4,2351
32
- voiceio/streaming.py,sha256=BBIU-Uub5z8ByYi8tXEUykdhTBgMhwvLHS87-_pZeaQ,15339
33
- voiceio/transcriber.py,sha256=D6_RLO8K3bGUV4CsS2q_k5rxkHTMB_6DqnzDka5mEb0,9939
32
+ voiceio/streaming.py,sha256=XpF4fALTAgYqRB8zoDL81TuKlKLmUOc5_Cfd3auZFBI,21802
33
+ voiceio/transcriber.py,sha256=2ev_hwdusve7ZZqht9A9wtjjT72rcr7SuhmSIW8aClk,10721
34
34
  voiceio/vad.py,sha256=72_ICk4jSuSPYvTHdHYtHzY8W5uaGEOz5FTdqkolJi8,5327
35
35
  voiceio/vocabulary.py,sha256=eGV7QZs3tN6Cw5YpyoZsVEk2M5fJvie1n2XFi9CD4hw,6321
36
- voiceio/wizard.py,sha256=cqNKMMzzfFy0Ncg4LJ0TWX77B84-ZvXm_1ibbarQFng,80209
36
+ voiceio/wizard.py,sha256=AwjwrocrQp0DPSOMZ0m6z6G3hlo9VPceEwPzR9-xXe0,80227
37
37
  voiceio/wordfreq.py,sha256=2UMjW1xIFqo5EchXzsXPizUhRb9N18CVgjnXso1wsgw,2382
38
38
  voiceio/worker.py,sha256=M6e4s6P0-4UGQzw0j8Mq68nypvwZleOSwuRhjuOqlVE,2563
39
39
  voiceio/hotkeys/__init__.py,sha256=rGXSGZLD2mS_Ep2HfVdRdkUKNr4xagySlL1FM00cYgk,735
@@ -71,8 +71,8 @@ voiceio/typers/pynput_type.py,sha256=DTtkT59M-EKOAlTNkvcZrRvf_Fq52OpI8S5t_8S-QAY
71
71
  voiceio/typers/wtype.py,sha256=d1wG-HZdYDQxXIysV_Xvk9HixDpaJgl3VEoKbP5vhIs,1786
72
72
  voiceio/typers/xdotool.py,sha256=dh1zhPqUT8ihJahJ7ZKm6PtfYY087UAzADx664DvQOM,1356
73
73
  voiceio/typers/ydotool.py,sha256=dt0W9ot__W8LQG2so8vZLGIw9leK2Q6g3i-Voo0LXE4,5327
74
- python_voiceio-0.4.1.dist-info/METADATA,sha256=eCo6CC2sRMLl4xKLDyp0N7W7a_qVKvBf6vARmNZYJaY,15877
75
- python_voiceio-0.4.1.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
76
- python_voiceio-0.4.1.dist-info/entry_points.txt,sha256=U64fA65zxzyLoC8bgbn2ztQVWHLsc0o0H0qCk1J9DMc,218
77
- python_voiceio-0.4.1.dist-info/top_level.txt,sha256=piwtn309lD6uexQyXdZ-efAVBJF9y6Wfr48Z-8zkNhg,8
78
- python_voiceio-0.4.1.dist-info/RECORD,,
74
+ python_voiceio-0.6.0.dist-info/METADATA,sha256=HdRa-d2IgSiL7VZxA7lcXavG0dGkqcKGf6Lmzdf9xxk,15877
75
+ python_voiceio-0.6.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
76
+ python_voiceio-0.6.0.dist-info/entry_points.txt,sha256=U64fA65zxzyLoC8bgbn2ztQVWHLsc0o0H0qCk1J9DMc,218
77
+ python_voiceio-0.6.0.dist-info/top_level.txt,sha256=piwtn309lD6uexQyXdZ-efAVBJF9y6Wfr48Z-8zkNhg,8
78
+ python_voiceio-0.6.0.dist-info/RECORD,,
voiceio/__init__.py CHANGED
@@ -1 +1 @@
1
- __version__ = "0.4.1"
1
+ __version__ = "0.6.0"
voiceio/app.py CHANGED
@@ -337,7 +337,31 @@ class VoiceIO:
337
337
  log.exception("Cannot reopen audio stream, aborting recording")
338
338
  return
339
339
 
340
- if not self.recorder.has_signal():
340
+ if self.recorder.is_zombie():
341
+ # Full pre-buffer of digital zeros with a "healthy" stream is the
342
+ # post-suspend zombie signature (callbacks fire, capture node
343
+ # gone). Reopening fixes it; a failed reopen means no audio is
344
+ # possible, so abort rather than record silence.
345
+ log.warning("Zombie audio stream (all-zero pre-buffer) — reopening")
346
+ try:
347
+ self.recorder.reopen_stream()
348
+ time.sleep(0.4) # let the new stream fill some pre-buffer
349
+ except Exception:
350
+ log.exception("Audio stream reopen failed, aborting recording")
351
+ from voiceio import feedback
352
+ feedback.notify(
353
+ "VoiceIO: microphone unavailable",
354
+ "Could not reopen the audio stream — check your input device.",
355
+ )
356
+ return
357
+ if not self.recorder.has_signal():
358
+ log.warning("Mic still silent after reopen — muted or wrong device?")
359
+ from voiceio import feedback
360
+ feedback.notify(
361
+ "VoiceIO: microphone appears silent",
362
+ "Check that your mic is unmuted and the right input device is selected.",
363
+ )
364
+ elif not self.recorder.has_signal():
341
365
  log.warning("Mic appears silent or muted (pre-buffer is all zeros)")
342
366
 
343
367
  self._state = _State.RECORDING
@@ -368,6 +392,8 @@ class VoiceIO:
368
392
  llm=self._llm,
369
393
  voice_input_prefix=self._voice_input_prefix,
370
394
  on_typer_broken=self._on_typer_broken,
395
+ on_interim=self._on_interim_text,
396
+ freeze_secs=self.cfg.output.streaming_freeze_secs,
371
397
  )
372
398
  self._session.start()
373
399
  log.info("Recording... press [%s] again to stop", self.cfg.hotkey.key)
@@ -492,12 +518,19 @@ class VoiceIO:
492
518
  ) -> None:
493
519
  """Run final transcription and commit in background thread."""
494
520
  t_final = time.monotonic()
521
+ # Snappy clipboard: make the best-so-far text pasteable the moment
522
+ # the user stops, instead of after the (possibly long) final decode.
523
+ if (self.cfg.output.copy_to_clipboard == "live"
524
+ and self._typer.name != "clipboard" and session.interim_text):
525
+ from voiceio import clipboard_read
526
+ clipboard_read.copy_text(session.interim_text)
495
527
  extra = self._retention_extra(audio)
496
528
  final_text = session.stop(audio)
497
529
  if self._generation != gen:
498
530
  log.debug("Finalize cancelled (gen %d, current %d)", gen, self._generation)
499
531
  return
500
532
  if final_text:
533
+ self._copy_result_async(final_text)
501
534
  self._play_feedback(final_text)
502
535
  stored = self._strip_voice_prefix(final_text)
503
536
  self._prompt_builder.add_transcript(stored)
@@ -513,6 +546,16 @@ class VoiceIO:
513
546
  duration=elapsed,
514
547
  extra=extra,
515
548
  )
549
+ from voiceio import retention
550
+ retention.save_trace(self.cfg.data, {
551
+ "ts": time.time(),
552
+ "audio": extra.get("audio"),
553
+ "duration": round(elapsed, 2),
554
+ "latency": latency,
555
+ # Snapshot: a timed-out worker join means the thread may
556
+ # still be appending while we serialize.
557
+ "passes": list(session.trace),
558
+ })
516
559
  log.info("Streaming done (%.1fs): '%s'", elapsed, final_text)
517
560
  # Release the IBus input source now that the final commit is done.
518
561
  # Generation-checked inside: if a newer recording started, it already
@@ -565,6 +608,7 @@ class VoiceIO:
565
608
  # Superseded by a newer recording — do not touch the typer.
566
609
  if text and self._generation == gen:
567
610
  self._type_with_fallback(text)
611
+ self._copy_result_async(text)
568
612
  self._play_feedback(text)
569
613
  stored = self._strip_voice_prefix(text)
570
614
  self._prompt_builder.add_transcript(stored)
@@ -660,6 +704,35 @@ class VoiceIO:
660
704
  from voiceio.feedback import play_record_stop
661
705
  play_record_stop()
662
706
 
707
+ def _on_interim_text(self, text: str) -> None:
708
+ """Streaming update: mirror the best-so-far text to the clipboard.
709
+
710
+ Runs on the streaming worker thread (a copy is a few ms). Skipped for
711
+ the clipboard typer, which pastes FROM the clipboard — overwriting it
712
+ here could race a pending Ctrl+V and paste the full text twice.
713
+ """
714
+ if self.cfg.output.copy_to_clipboard != "live":
715
+ return
716
+ if self._typer.name == "clipboard":
717
+ return
718
+ from voiceio import clipboard_read
719
+ clipboard_read.copy_text(text)
720
+
721
+ def _copy_result_async(self, text: str) -> None:
722
+ """Mirror the final text to the clipboard, off the commit hot path."""
723
+ if self.cfg.output.copy_to_clipboard not in ("final", "live"):
724
+ return
725
+
726
+ def _copy() -> None:
727
+ # Give a just-sent paste keystroke time to consume the clipboard
728
+ # before we overwrite it with the full text.
729
+ if self._typer.name == "clipboard":
730
+ time.sleep(0.3)
731
+ from voiceio import clipboard_read
732
+ clipboard_read.copy_text(text)
733
+
734
+ threading.Thread(target=_copy, daemon=True).start()
735
+
663
736
  def _play_feedback(self, text: str) -> None:
664
737
  if self.cfg.feedback.sound_enabled:
665
738
  from voiceio.feedback import play_commit_sound
voiceio/clipboard_read.py CHANGED
@@ -1,4 +1,4 @@
1
- """Read text from the system clipboard or primary selection."""
1
+ """Read/write text from/to the system clipboard or primary selection."""
2
2
  from __future__ import annotations
3
3
 
4
4
  import logging
@@ -67,3 +67,49 @@ def read_text() -> str | None:
67
67
 
68
68
  log.debug("No clipboard tool found")
69
69
  return None
70
+
71
+
72
+ def _copy_via(cmd: list[str], text: str) -> bool:
73
+ """Pipe text into a copy command. Returns True on success."""
74
+ try:
75
+ result = subprocess.run(
76
+ cmd, input=text.encode(), capture_output=True, timeout=_TIMEOUT,
77
+ )
78
+ return result.returncode == 0
79
+ except (FileNotFoundError, subprocess.TimeoutExpired, OSError):
80
+ return False
81
+
82
+
83
+ def copy_text(text: str) -> bool:
84
+ """Copy text to the system CLIPBOARD selection.
85
+
86
+ Best-effort: returns False (never raises) when no tool is available
87
+ or the copy fails.
88
+ """
89
+ if not text:
90
+ return False
91
+ p = detect()
92
+
93
+ if p.is_windows:
94
+ try:
95
+ import pyperclip
96
+ pyperclip.copy(text)
97
+ return True
98
+ except Exception:
99
+ log.debug("pyperclip not available for clipboard write")
100
+ return False
101
+
102
+ if p.is_mac:
103
+ return _copy_via(["pbcopy"], text)
104
+
105
+ if p.is_wayland:
106
+ if shutil.which("wl-copy"):
107
+ return _copy_via(["wl-copy", "--"], text)
108
+ else:
109
+ if shutil.which("xclip"):
110
+ return _copy_via(["xclip", "-selection", "clipboard"], text)
111
+ if shutil.which("xsel"):
112
+ return _copy_via(["xsel", "-i", "--clipboard"], text)
113
+
114
+ log.debug("No clipboard tool found for copy")
115
+ return False
voiceio/config.py CHANGED
@@ -33,6 +33,11 @@ PID_PATH = LOG_DIR / "voiceio.pid"
33
33
  # and the retired-rule state consulted by the mining side.
34
34
  CORRECTIONS_AUDIT_PATH = LOG_DIR / "corrections_audit.jsonl"
35
35
  METRICS_PATH = LOG_DIR / "metrics.jsonl"
36
+ # Intermediate pipeline data kept for debugging/profiling and future training:
37
+ # per-pass streaming decode traces and postcorrect before/after pairs. These
38
+ # live in their own JSONL files (not the rotating log) so they survive.
39
+ TRACES_PATH = LOG_DIR / "streaming_trace.jsonl"
40
+ POSTCORRECT_PAIRS_PATH = LOG_DIR / "postcorrect_pairs.jsonl"
36
41
  SNAPSHOTS_DIR = CONFIG_DIR / "snapshots"
37
42
  AUDIT_STATE_PATH = CONFIG_DIR / "audit_state.json"
38
43
  AUTOCORRECT_STATE_PATH = CONFIG_DIR / "autocorrect_state.json"
@@ -79,6 +84,17 @@ class OutputConfig:
79
84
  punctuation_cleanup: bool = True
80
85
  number_conversion: bool = True
81
86
  voice_input_prefix: str = "" # e.g. "[voice]" — empty disables
87
+ # Incremental finalization: once the un-finalized audio tail grows past
88
+ # this many seconds AND ends in silence, beam-decode and freeze it during
89
+ # recording, so interim passes and the stop-time final decode only cover
90
+ # the short remaining tail. 0 disables (full re-decode at stop).
91
+ streaming_freeze_secs: float = 25.0
92
+ # Mirror transcribed text to the system clipboard so it can be pasted:
93
+ # "off" — never
94
+ # "final" — the corrected final text, once ready
95
+ # "live" — also the best-so-far text at each streaming update and the
96
+ # instant recording stops (pasteable while the final decode runs)
97
+ copy_to_clipboard: str = "final"
82
98
 
83
99
 
84
100
  @dataclass
@@ -173,7 +189,14 @@ class DataConfig:
173
189
  """
174
190
  retain_audio: bool = True
175
191
  max_audio_mb: int = 4096 # prune oldest recordings beyond this
192
+ # Never let retention squeeze the disk: skip saving and prune harder
193
+ # when the filesystem has less than this much free space.
194
+ min_free_gb: float = 5.0
176
195
  capture_context: bool = True # best-effort active-window title
196
+ # Per-pass streaming decode traces + postcorrect before/after pairs
197
+ # (streaming_trace.jsonl / postcorrect_pairs.jsonl) for debugging,
198
+ # profiling, and future fine-tuning.
199
+ capture_intermediates: bool = True
177
200
 
178
201
 
179
202
  @dataclass
@@ -277,6 +300,8 @@ def _content_files() -> list[Path]:
277
300
  HISTORY_PATH,
278
301
  CORRECTIONS_AUDIT_PATH,
279
302
  METRICS_PATH,
303
+ TRACES_PATH,
304
+ POSTCORRECT_PAIRS_PATH,
280
305
  AUDIT_STATE_PATH,
281
306
  AUTOCORRECT_STATE_PATH,
282
307
  CONSENT_PATH,
voiceio/postcorrect.py CHANGED
@@ -15,6 +15,7 @@ from __future__ import annotations
15
15
  import dataclasses
16
16
  import difflib
17
17
  import logging
18
+ import threading
18
19
  import time
19
20
 
20
21
  from voiceio.config import AutocorrectConfig, Config
@@ -90,6 +91,13 @@ class PostCorrector:
90
91
  self._vocabulary = ""
91
92
  self._recent: list[str] = []
92
93
  self._context: str | None = None
94
+ # Context actually sent with the in-progress correct() call (what
95
+ # _record must persist — may differ from self._context).
96
+ self._effective_ctx: str | None = None
97
+ # A worker abandoned at the deadline may block indefinitely on a hung
98
+ # endpoint; cap leakage at one thread/socket by skipping new calls
99
+ # while it is still alive.
100
+ self._abandoned: threading.Thread | None = None
93
101
 
94
102
  # ── availability ────────────────────────────────────────────────────
95
103
 
@@ -140,6 +148,27 @@ class PostCorrector:
140
148
  parts.append(f"Transcript to correct:\n{text}")
141
149
  return "\n\n".join(parts)
142
150
 
151
+ def _record(self, before: str, after: str | None, outcome: str) -> None:
152
+ """Persist one LLM attempt (before/after/outcome) as training data.
153
+
154
+ Pairs land in postcorrect_pairs.jsonl — unlike the rotating log they
155
+ survive, so accepted AND rejected corrections stay available for
156
+ tuning guards or training a local corrector later.
157
+ """
158
+ if not self._cfg.data.capture_intermediates:
159
+ return
160
+ from voiceio import retention
161
+ from voiceio.config import POSTCORRECT_PAIRS_PATH
162
+ retention.append_jsonl(POSTCORRECT_PAIRS_PATH, {
163
+ "ts": time.time(),
164
+ "before": before,
165
+ "after": after,
166
+ "outcome": outcome,
167
+ "secs": round(self.last_secs, 3) if self.last_secs is not None else None,
168
+ "model": self._pc.model or self._ac.model,
169
+ "context": self._effective_ctx,
170
+ })
171
+
143
172
  def correct(
144
173
  self, text: str, *, vocabulary: str = "",
145
174
  recent: list[str] | None = None, context: str | None = None,
@@ -161,28 +190,65 @@ class PostCorrector:
161
190
  vocab = vocabulary or self._vocabulary
162
191
  rec = recent if recent is not None else self._recent
163
192
  ctx = context if context is not None else self._context
193
+ self._effective_ctx = ctx
194
+
195
+ if self._abandoned is not None:
196
+ if self._abandoned.is_alive():
197
+ log.warning(
198
+ "PostCorrector: previous request still hung — skipping this one",
199
+ )
200
+ self._record(text, None, "skipped_busy")
201
+ return text
202
+ self._abandoned = None
164
203
 
165
204
  from voiceio.llm_api import chat
166
205
  user_msg = self._build_user_message(text, vocab, rec, ctx)
206
+ # timeout_secs must bound the WALL CLOCK the user waits, but urllib's
207
+ # timeout is per-socket-read — a slowly streaming response can run
208
+ # far past it (observed ~14s with an 8s config). Run the call in a
209
+ # thread and abandon it at the deadline.
167
210
  t0 = time.monotonic()
168
- try:
169
- response = chat(self._client_cfg(), _SYSTEM_PROMPT, user_msg, max_tokens=1024)
170
- except Exception as e:
171
- log.debug("PostCorrector LLM error: %s — keeping original", e)
211
+ outcome: dict = {}
212
+
213
+ def _call() -> None:
214
+ try:
215
+ outcome["response"] = chat(
216
+ self._client_cfg(), _SYSTEM_PROMPT, user_msg, max_tokens=1024,
217
+ )
218
+ except Exception as e:
219
+ outcome["error"] = e
220
+
221
+ worker = threading.Thread(target=_call, daemon=True)
222
+ worker.start()
223
+ worker.join(self._pc.timeout_secs)
224
+ self.last_secs = time.monotonic() - t0
225
+ if worker.is_alive():
226
+ log.debug(
227
+ "PostCorrector deadline (%.1fs) exceeded — keeping original",
228
+ self._pc.timeout_secs,
229
+ )
230
+ self._abandoned = worker
231
+ self._record(text, None, "timeout")
232
+ return text
233
+ if "error" in outcome:
234
+ log.debug("PostCorrector LLM error: %s — keeping original", outcome["error"])
235
+ self._record(text, None, "error")
172
236
  return text
173
- finally:
174
- self.last_secs = time.monotonic() - t0
237
+ response = outcome.get("response")
175
238
 
176
239
  if not response:
177
240
  log.debug("PostCorrector: empty/failed response — keeping original")
241
+ self._record(text, None, "empty")
178
242
  return text
179
243
 
180
244
  corrected = _strip_wrapping(response)
181
245
  if not corrected:
182
246
  log.debug("PostCorrector: response empty after stripping — keeping original")
247
+ self._record(text, None, "empty")
183
248
  return text
184
249
 
185
250
  if corrected == text:
251
+ self._record(text, corrected, "unchanged")
186
252
  return text
187
253
 
188
254
  # Guard: word-count must not change materially.
@@ -192,6 +258,7 @@ class PostCorrector:
192
258
  "PostCorrector reject: word count %d→%d (>%.0f%%) — keeping original",
193
259
  orig_wc, new_wc, _MAX_WORDCOUNT_DELTA * 100,
194
260
  )
261
+ self._record(text, corrected, "rejected_wordcount")
195
262
  return text
196
263
 
197
264
  # Guard: only a small fraction of words may change.
@@ -201,7 +268,9 @@ class PostCorrector:
201
268
  "PostCorrector reject: edit ratio %.2f > %.2f — keeping original",
202
269
  ratio, _MAX_EDIT_RATIO,
203
270
  )
271
+ self._record(text, corrected, "rejected_editratio")
204
272
  return text
205
273
 
206
274
  log.info("PostCorrector fixed: %s", ", ".join(_changed_words(text, corrected)))
275
+ self._record(text, corrected, "applied")
207
276
  return corrected
voiceio/recorder.py CHANGED
@@ -162,6 +162,17 @@ class AudioRecorder:
162
162
  return False, f"no audio callback for {stale:.1f}s"
163
163
  return True, ""
164
164
 
165
+ def is_zombie(self) -> bool:
166
+ """Stream firing callbacks but delivering only digital zeros.
167
+
168
+ The post-suspend PortAudio failure mode: the heartbeat stays healthy
169
+ while the capture node is gone. Distinct from a merely-empty ring
170
+ (fresh stream) or a muted mic (noise floor > 0 on real hardware).
171
+ """
172
+ if self._ring._filled < self._ring._max:
173
+ return False # ring not yet full — can't judge
174
+ return not self.has_signal()
175
+
165
176
  def has_signal(self) -> bool:
166
177
  """Check if the pre-buffer ring contains non-silence audio.
167
178
 
voiceio/retention.py CHANGED
@@ -7,17 +7,19 @@ mining measurable and personal fine-tuning possible later.
7
7
  from __future__ import annotations
8
8
 
9
9
  import functools
10
+ import json
10
11
  import logging
11
12
  import shutil
12
13
  import subprocess
13
14
  import time
14
15
  import wave
16
+ from pathlib import Path
15
17
  from typing import TYPE_CHECKING
16
18
 
17
19
  import numpy as np
18
20
 
19
21
  from voiceio import config
20
- from voiceio.config import RECORDINGS_DIR
22
+ from voiceio.config import RECORDINGS_DIR, TRACES_PATH
21
23
 
22
24
  if TYPE_CHECKING:
23
25
  from voiceio.config import DataConfig
@@ -27,11 +29,25 @@ log = logging.getLogger(__name__)
27
29
  _which = functools.lru_cache(maxsize=8)(shutil.which)
28
30
 
29
31
 
32
+ def _free_gb(path: Path) -> float:
33
+ """Free space (GB) on the filesystem holding `path`. Inf on failure."""
34
+ try:
35
+ return shutil.disk_usage(path).free / 1024**3
36
+ except OSError:
37
+ return float("inf")
38
+
39
+
30
40
  def save_audio(audio: np.ndarray, ts: float, cfg: DataConfig) -> str | None:
31
41
  """Persist one utterance as 16kHz mono int16 WAV. Returns the filename
32
42
  (relative to the recordings dir) or None if disabled/failed."""
33
43
  if not cfg.retain_audio or audio is None or len(audio) == 0:
34
44
  return None
45
+ if _free_gb(RECORDINGS_DIR.parent) < cfg.min_free_gb:
46
+ log.warning(
47
+ "Disk has <%.0fGB free — skipping audio retention for this utterance",
48
+ cfg.min_free_gb,
49
+ )
50
+ return None
35
51
  name = time.strftime("%Y%m%d-%H%M%S", time.localtime(ts)) + f"-{int(ts * 1000) % 1000:03d}.wav"
36
52
  path = RECORDINGS_DIR / name
37
53
  try:
@@ -70,6 +86,15 @@ def prune(cfg: DataConfig) -> None:
70
86
  )
71
87
  total = sum(p.stat().st_size for p in files)
72
88
  budget = cfg.max_audio_mb * 1024 * 1024
89
+ # Below the free-disk floor, shrink the budget so retention yields
90
+ # space back instead of holding its full cap on a squeezed disk.
91
+ free = _free_gb(RECORDINGS_DIR.parent)
92
+ if free < cfg.min_free_gb:
93
+ budget = min(budget, total // 2)
94
+ log.warning(
95
+ "Disk has %.1fGB free (<%.0fGB floor) — pruning recordings to %.0fMB",
96
+ free, cfg.min_free_gb, budget / 1024 / 1024,
97
+ )
73
98
  while total > budget and files:
74
99
  oldest = files.pop(0)
75
100
  total -= oldest.stat().st_size
@@ -79,6 +104,45 @@ def prune(cfg: DataConfig) -> None:
79
104
  log.debug("Recording prune failed", exc_info=True)
80
105
 
81
106
 
107
+ _JSONL_MAX_BYTES = 64 * 1024 * 1024 # per capture file; oldest half dropped
108
+
109
+
110
+ def append_jsonl(path: Path, entry: dict) -> None:
111
+ """Append one JSON line to a 0600 data file. Best-effort, never raises.
112
+
113
+ Capture files are bounded: past _JSONL_MAX_BYTES the oldest half of the
114
+ lines is dropped, so intermediate-data capture can never grow unbounded
115
+ on a disk the min_free_gb floor is trying to protect.
116
+ """
117
+ try:
118
+ path.parent.mkdir(parents=True, exist_ok=True)
119
+ config._chmod(path.parent, config._SECURE_DIR)
120
+ newly_created = not path.exists()
121
+ if not newly_created and path.stat().st_size > _JSONL_MAX_BYTES:
122
+ lines = path.read_text(encoding="utf-8").splitlines(keepends=True)
123
+ path.write_text("".join(lines[len(lines) // 2:]), encoding="utf-8")
124
+ log.info("Trimmed %s to newest %d entries", path.name, len(lines) - len(lines) // 2)
125
+ with open(path, "a", encoding="utf-8") as f:
126
+ f.write(json.dumps(entry, ensure_ascii=False) + "\n")
127
+ if newly_created:
128
+ config._chmod(path, config._SECURE_FILE)
129
+ except (OSError, TypeError, ValueError) as e:
130
+ log.warning("Failed to write %s: %s", path.name, e)
131
+
132
+
133
+ def save_trace(cfg: DataConfig, entry: dict) -> None:
134
+ """Persist one utterance's per-pass streaming decode trace.
135
+
136
+ One line per utterance in streaming_trace.jsonl: pass timings, kinds
137
+ (interim/freeze/final), tail lengths, and raw tail texts — the data
138
+ needed to debug/profile streaming behaviour and to train on interim
139
+ hypotheses later. Linked to history via ts + audio filename.
140
+ """
141
+ if not cfg.capture_intermediates:
142
+ return
143
+ append_jsonl(TRACES_PATH, entry)
144
+
145
+
82
146
  def active_window_title() -> str | None:
83
147
  """Best-effort title of the focused window (dictation target context).
84
148
 
voiceio/streaming.py CHANGED
@@ -8,11 +8,12 @@ import threading
8
8
  import time
9
9
  from typing import TYPE_CHECKING, Callable
10
10
 
11
+ import numpy as np
12
+
11
13
  from voiceio.transcriber import transcribe_timeout
12
14
  from voiceio.typers.base import StreamingTyper
13
15
 
14
16
  if TYPE_CHECKING:
15
- import numpy as np
16
17
  from voiceio.commands import CommandProcessor
17
18
  from voiceio.corrections import CorrectionDict
18
19
  from voiceio.llm import LLMProcessor
@@ -24,6 +25,32 @@ if TYPE_CHECKING:
24
25
  log = logging.getLogger(__name__)
25
26
  DELETE_SETTLE_SECS = 0.05 # delay between delete and type for ydotool reliability
26
27
 
28
+ # Incremental finalization: a freeze cut is only safe at a speech pause, so
29
+ # require the tail's trailing window to be quiet before cutting there.
30
+ _FREEZE_SILENCE_WINDOW_SECS = 0.3
31
+ # Quiet is judged RELATIVE to the tail's own level: mic gains vary wildly
32
+ # and the decode path normalizes audio anyway, so an absolute threshold
33
+ # would read a quiet mic's speech as silence (mid-word cuts) and a hot
34
+ # noise floor as speech (freeze never fires).
35
+ _FREEZE_SILENCE_RATIO = 0.15
36
+ _FREEZE_SILENCE_FLOOR = 1e-4 # digital silence is always quiet
37
+ # Whisper conditions on ~224 prompt tokens; more frozen context is wasted.
38
+ _FREEZE_CONTEXT_CHARS = 400
39
+
40
+
41
+ def _rms(samples: np.ndarray) -> float:
42
+ return float(np.sqrt(np.mean(np.square(samples, dtype=np.float64))))
43
+
44
+
45
+ def _tail_ends_in_silence(tail: np.ndarray, sample_rate: int) -> bool:
46
+ """True when the last _FREEZE_SILENCE_WINDOW_SECS of audio are quiet
47
+ relative to the tail's overall level."""
48
+ window = tail[-int(_FREEZE_SILENCE_WINDOW_SECS * sample_rate):]
49
+ if len(window) == 0:
50
+ return False
51
+ threshold = max(_FREEZE_SILENCE_RATIO * _rms(tail), _FREEZE_SILENCE_FLOOR)
52
+ return _rms(window) < threshold
53
+
27
54
 
28
55
  def _common_prefix_len(a: str, b: str) -> int:
29
56
  """Length of the longest common prefix between two strings."""
@@ -34,6 +61,16 @@ def _common_prefix_len(a: str, b: str) -> int:
34
61
  return limit
35
62
 
36
63
 
64
+ def _join_text(a: str, b: str) -> str:
65
+ """Join two transcript fragments with a single space."""
66
+ a, b = a.strip(), (b or "").strip()
67
+ if not a:
68
+ return b
69
+ if not b:
70
+ return a
71
+ return a + " " + b
72
+
73
+
37
74
  def _clean_word(w: str) -> str:
38
75
  """Strip punctuation for fuzzy word matching."""
39
76
  return re.sub(r'[^\w]', '', w).lower()
@@ -84,6 +121,8 @@ class StreamingSession:
84
121
  voice_input_prefix: str = "",
85
122
  on_typer_broken: Callable[[], None] | None = None,
86
123
  is_current: Callable[[], bool] | None = None,
124
+ on_interim: Callable[[str], None] | None = None,
125
+ freeze_secs: float = 25.0,
87
126
  ):
88
127
  self._transcriber = transcriber
89
128
  self._typer = typer
@@ -104,6 +143,7 @@ class StreamingSession:
104
143
  self._llm = llm
105
144
  self._voice_input_prefix = voice_input_prefix
106
145
  self._on_typer_broken = on_typer_broken
146
+ self._on_interim = on_interim
107
147
  self._typer_fail_count = 0
108
148
  self._typer_broken_signalled = False
109
149
  self._typed_text = ""
@@ -111,10 +151,26 @@ class StreamingSession:
111
151
  self._stop_event = threading.Event()
112
152
  self._worker_thread: threading.Thread | None = None
113
153
  self._final_audio: np.ndarray | None = None # set on stop
154
+ # Incremental finalization: audio before _frozen_samples has already
155
+ # been beam-decoded into _frozen_raw and is never decoded again.
156
+ # Interim and final passes only decode the tail after this offset.
157
+ self._freeze_secs = freeze_secs
158
+ self._frozen_samples = 0
159
+ self._frozen_raw = ""
160
+ self._frozen_segments: list[dict] = []
114
161
  # Raw (pre-pipeline) text + confidence of the final pass, for history
115
162
  self.raw_final_text: str | None = None
116
163
  self.final_latency: dict = {}
117
164
  self.final_segments: list[dict] = []
165
+ # Per-pass decode trace (interim/freeze/final) for debugging,
166
+ # profiling, and future training. Persisted via retention.save_trace.
167
+ self.trace: list[dict] = []
168
+ self._t0 = time.monotonic()
169
+
170
+ @property
171
+ def interim_text(self) -> str:
172
+ """Best-so-far text (post-pipeline). Safe to read while finalizing."""
173
+ return self._typed_text
118
174
 
119
175
  def set_is_current(self, is_current: Callable[[], bool]) -> None:
120
176
  """Install the output-ownership gate (see __init__).
@@ -206,10 +262,22 @@ class StreamingSession:
206
262
  log.exception("Final transcribe/apply error")
207
263
  self._final_audio = None # release memory
208
264
 
265
+ def _frozen_context(self) -> str | None:
266
+ """Frozen-text suffix used to condition the tail decode."""
267
+ if not self._frozen_raw:
268
+ return None
269
+ return self._frozen_raw[-_FREEZE_CONTEXT_CHARS:]
270
+
209
271
  def _transcribe_and_apply(
210
272
  self, min_seconds: float = 1.0, final: bool = False,
211
273
  ) -> None:
212
- """Get audio, transcribe, apply correction."""
274
+ """Get audio, transcribe the un-frozen tail, apply correction.
275
+
276
+ Incremental finalization: audio before _frozen_samples was already
277
+ beam-decoded and frozen, so every pass — including the final one —
278
+ only decodes the tail. Long dictations finalize in O(tail), not
279
+ O(recording).
280
+ """
213
281
  if final and self._final_audio is not None:
214
282
  # Use the snapshot passed to stop() — recorder may be gone
215
283
  audio = self._final_audio
@@ -223,22 +291,84 @@ class StreamingSession:
223
291
  if len(audio) < self._sample_rate * min_seconds:
224
292
  return
225
293
 
294
+ tail = audio[self._frozen_samples:]
295
+ if not final and len(tail) < self._sample_rate * min_seconds:
296
+ return # not enough new audio since the last freeze
297
+
298
+ # Freeze when the tail has grown long AND ends at a speech pause
299
+ # (never cut mid-word). The freeze pass gets beam search because its
300
+ # text is never decoded again.
301
+ freeze = (
302
+ not final
303
+ and self._freeze_secs > 0
304
+ and len(tail) >= self._freeze_secs * self._sample_rate
305
+ and _tail_ends_in_silence(tail, self._sample_rate)
306
+ )
307
+
308
+ tail_text = ""
309
+ tail_segments: list[dict] = []
226
310
  t0 = time.monotonic()
227
- try:
228
- text = self._transcriber.transcribe(audio, final=final)
229
- except Exception:
230
- log.exception("Streaming transcription failed")
231
- return
311
+ # The final pass decodes ANY remaining audio (a clipped last word
312
+ # must not be dropped); interim/freeze passes skip sub-0.3s tails.
313
+ if len(tail) > (0 if final else self._sample_rate * 0.3):
314
+ try:
315
+ tail_text = self._transcriber.transcribe(
316
+ tail, final=final or freeze, context=self._frozen_context(),
317
+ )
318
+ except Exception:
319
+ # Decode FAILED (timeout/crash) — this is not silence. Never
320
+ # advance the freeze boundary or char-diff against a partial
321
+ # transcript; on final, commit what is already on screen.
322
+ log.exception("Streaming transcription failed")
323
+ if final:
324
+ self.raw_final_text = ""
325
+ self.final_latency = {
326
+ "audio_secs": round(len(audio) / self._sample_rate, 2),
327
+ "frozen_secs": round(self._frozen_samples / self._sample_rate, 2),
328
+ "tail_secs": round(len(tail) / self._sample_rate, 2),
329
+ "transcribe": round(time.monotonic() - t0, 3),
330
+ "decode_failed": True,
331
+ }
332
+ self._commit_interim_on_final()
333
+ return
334
+ tail_segments = list(getattr(self._transcriber, "last_segments", []))
232
335
  t_transcribe = time.monotonic() - t0
233
336
 
337
+ self.trace.append({
338
+ "t": round(time.monotonic() - self._t0, 2),
339
+ "kind": "final" if final else ("freeze" if freeze else "interim"),
340
+ "frozen_secs": round(self._frozen_samples / self._sample_rate, 2),
341
+ "tail_secs": round(len(tail) / self._sample_rate, 2),
342
+ "secs": round(t_transcribe, 3),
343
+ "text": tail_text,
344
+ })
345
+
346
+ if freeze:
347
+ self._frozen_raw = _join_text(self._frozen_raw, tail_text)
348
+ self._frozen_samples = len(audio)
349
+ self._frozen_segments.extend(tail_segments)
350
+ log.debug(
351
+ "Froze %.1fs of audio (frozen text now %d chars)",
352
+ len(tail) / self._sample_rate, len(self._frozen_raw),
353
+ )
354
+ raw = self._frozen_raw if freeze else _join_text(self._frozen_raw, tail_text)
355
+
356
+ # An interim pass that decoded nothing must not touch the display:
357
+ # rewriting to frozen-only text would visibly delete the un-frozen
358
+ # words the previous pass just showed (transient decoder flake).
359
+ if not final and not tail_text:
360
+ return
361
+
234
362
  if final:
235
- self.raw_final_text = text
236
- self.final_segments = getattr(self._transcriber, "last_segments", [])
363
+ self.raw_final_text = raw
364
+ self.final_segments = self._frozen_segments + tail_segments
237
365
  self.final_latency = {
238
366
  "audio_secs": round(len(audio) / self._sample_rate, 2),
367
+ "frozen_secs": round(self._frozen_samples / self._sample_rate, 2),
368
+ "tail_secs": round(len(tail) / self._sample_rate, 2),
239
369
  "transcribe": round(t_transcribe, 3),
240
370
  }
241
- if not text:
371
+ if not raw:
242
372
  # The final pass produced nothing. If we already have interim
243
373
  # text (e.g. the final transcription timed out on a long
244
374
  # dictation), commit that instead of silently dropping it —
@@ -246,11 +376,11 @@ class StreamingSession:
246
376
  self._commit_interim_on_final()
247
377
  return
248
378
 
249
- if text and isinstance(text, str):
379
+ if raw:
250
380
  from voiceio.postprocess import apply_pipeline
251
381
  t1 = time.monotonic()
252
382
  text, abort = apply_pipeline(
253
- text,
383
+ raw,
254
384
  do_cleanup=self._cleanup,
255
385
  number_conversion=self._number_conversion,
256
386
  language=self._language,
@@ -267,6 +397,11 @@ class StreamingSession:
267
397
  if pc_secs is not None:
268
398
  self.final_latency["postcorrect"] = round(pc_secs, 3)
269
399
  if abort:
400
+ # A voice command wiped the utterance — discard frozen text
401
+ # too, or the next pass would resurrect it.
402
+ self._frozen_raw = ""
403
+ self._frozen_segments = []
404
+ self._frozen_samples = len(audio)
270
405
  # A superseded session must not touch the typer on the final
271
406
  # path (would corrupt the newer session's output).
272
407
  if final and not self._is_current():
@@ -280,7 +415,13 @@ class StreamingSession:
280
415
  return
281
416
 
282
417
  if text:
418
+ prev = self._typed_text
283
419
  self._apply_correction(text, final=final)
420
+ if not final and self._on_interim and self._typed_text != prev:
421
+ try:
422
+ self._on_interim(self._typed_text)
423
+ except Exception:
424
+ log.debug("on_interim callback failed", exc_info=True)
284
425
 
285
426
  def _commit_interim_on_final(self) -> None:
286
427
  """Commit whatever interim text we have when the final pass yields none.
voiceio/transcriber.py CHANGED
@@ -18,6 +18,12 @@ if TYPE_CHECKING:
18
18
  log = logging.getLogger(__name__)
19
19
 
20
20
  TRANSCRIBE_TIMEOUT = 30 # seconds (floor; scaled up for long audio)
21
+
22
+
23
+ class TranscriptionError(RuntimeError):
24
+ """Decode failed (timeout / worker crash). Distinct from a legitimate
25
+ empty result ("" = silence) so callers never treat lost audio as decoded
26
+ silence — the streaming freeze/final paths must not advance state on it."""
21
27
  # Realtime-factor headroom for the read timeout. Whisper decodes well faster
22
28
  # than realtime, so 1.5x the audio duration is a generous ceiling that still
23
29
  # never kills a long dictation mid-decode.
@@ -152,10 +158,17 @@ class Transcriber:
152
158
  log.warning("Worker died, restarting (attempt %d/%d)", self._restarts, MAX_RESTARTS)
153
159
  self._start_worker()
154
160
 
155
- def transcribe(self, audio: np.ndarray, final: bool = False) -> str:
161
+ def transcribe(
162
+ self, audio: np.ndarray, final: bool = False, context: str | None = None,
163
+ ) -> str:
156
164
  """Transcribe audio. `final=True` marks the pass whose text the user
157
165
  keeps (streaming final / batch): it gets beam search; interim streaming
158
- passes stay greedy for latency."""
166
+ passes stay greedy for latency.
167
+
168
+ `context` is per-call text appended to the initial_prompt — used by
169
+ incremental finalization to condition a tail decode on the already-
170
+ frozen transcript so sentences stay coherent across the cut.
171
+ """
159
172
  with self._lock:
160
173
  self._ensure_worker()
161
174
 
@@ -165,8 +178,9 @@ class Transcriber:
165
178
  audio = normalize_audio(audio)
166
179
  audio_b64 = base64.b64encode(audio.tobytes()).decode("ascii")
167
180
  req = {"audio_b64": audio_b64, "options": {"beam_size": 5 if final else 1}}
168
- if self._initial_prompt:
169
- req["initial_prompt"] = self._initial_prompt
181
+ prompt = "\n".join(p for p in (self._initial_prompt, context) if p)
182
+ if prompt:
183
+ req["initial_prompt"] = prompt
170
184
  if self._hotwords:
171
185
  req["hotwords"] = self._hotwords
172
186
  try:
@@ -184,16 +198,18 @@ class Transcriber:
184
198
  timeout = transcribe_timeout(duration)
185
199
  result_line = self._read_with_timeout(timeout)
186
200
  if result_line is None:
201
+ self.last_segments = []
187
202
  log.warning("Transcription timed out after %.0fs, restarting worker", timeout)
188
203
  self._kill_worker()
189
204
  self._ensure_worker()
190
- return ""
205
+ raise TranscriptionError(f"decode timed out after {timeout:.0f}s")
191
206
 
192
207
  try:
193
208
  result = json.loads(result_line)
194
209
  except (json.JSONDecodeError, TypeError):
210
+ self.last_segments = []
195
211
  log.warning("Invalid response from worker: %s", repr(result_line)[:100])
196
- return ""
212
+ raise TranscriptionError("invalid worker response")
197
213
  text = result.get("text", "")
198
214
  self.last_segments = result.get("segments", [])
199
215
 
voiceio/wizard.py CHANGED
@@ -680,7 +680,7 @@ def _write_config(
680
680
  "model": autocorrect_model or "moonshotai/kimi-k2-0905",
681
681
  })
682
682
 
683
- CONFIG_PATH.write_text(_dump_toml(cfg))
683
+ CONFIG_PATH.write_text(_dump_toml(cfg), encoding="utf-8")
684
684
  _secure_config_permissions()
685
685
  if not quiet:
686
686
  print(f"\n {GREEN}✓{RESET} Config saved to {DIM}{CONFIG_PATH}{RESET}")