python-voiceio 0.8.2__py3-none-any.whl → 0.9.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: python-voiceio
3
- Version: 0.8.2
3
+ Version: 0.9.0
4
4
  Summary: Voice dictation for Linux. Speak → text, locally, instantly.
5
5
  Author: Hugo Montenegro
6
6
  License-Expression: MIT
@@ -1,18 +1,19 @@
1
- python_voiceio-0.8.2.dist-info/licenses/LICENSE,sha256=Gz61o8jFTAvZUZyB3nWDB3DQQVuipjfPkVu9W8hBHM0,1072
2
- voiceio/__init__.py,sha256=B7GiO0rd49YwtLYjvPg4lmCZEDlMTonslQKdSImaMJk,22
1
+ python_voiceio-0.9.0.dist-info/licenses/LICENSE,sha256=Gz61o8jFTAvZUZyB3nWDB3DQQVuipjfPkVu9W8hBHM0,1072
2
+ voiceio/__init__.py,sha256=H9NWRZb7NbeRRPLP_V1fARmLNXranorVM-OOY-8_2ug,22
3
3
  voiceio/__main__.py,sha256=xT5QCGGreYMisHO7Lh_Y-xAQ2TOzG2D7npKknGNcWSY,53
4
- voiceio/app.py,sha256=vO5Sk12CxT2aFtt14XP1Vyen_leUQcqYftswURjN58U,71360
5
- voiceio/audit.py,sha256=vX3x-YAhmt6EZYRZd0_6wvTquZu5sxGMvH45BhSBnCk,14915
4
+ voiceio/app.py,sha256=ScrUVDQQ_YbAFKKSMOhaRPX_WTcPuenhnRdSzWMxThE,73785
5
+ voiceio/audit.py,sha256=p4F2BBX4wEC9eS8MSplzfXL52XC2dyIKbgrRvR-WbTU,15802
6
6
  voiceio/autocorrect.py,sha256=jGsglEYwAHWk2uWedeEZzyAMW2LWO_8nQGZgrWYy4Dc,26552
7
7
  voiceio/autocorrect_state.py,sha256=BNzdHP5slcOV_-OU4MlaKE7xgVJzrPncsYFK30JWTs8,6858
8
8
  voiceio/backends.py,sha256=-rs1YhPblub4NjFb6qCD7Ih9FqBRXtPuvjVv9bQoQ1w,368
9
- voiceio/cli.py,sha256=_HfcWQt2Jf34YfI4W1RavsyqegQy9jrg0nwLe6DvD6E,71639
9
+ voiceio/cli.py,sha256=QE8N7N3d1O7r9-u_jbYMAH9gw0ILX7dSxDZZYj83lIs,78969
10
10
  voiceio/clipboard_read.py,sha256=lz75CkRGh7pyvO6xvoUmxJDl-Gwie1xlXQHK70PO84I,4261
11
11
  voiceio/commands.py,sha256=Vhtn8s5G5OcCbRN44YZ-_14Fa85ISXl_3Ix-KqSHs2A,4694
12
- voiceio/config.py,sha256=kpKileJk5mAp4eB7tXBblzOxmlxIfMvRDzY2tjwTkSQ,14465
12
+ voiceio/config.py,sha256=DA1PGIanhrPjWkVT6jj0NSFslW9vuegzV6eMJNuKBzM,15273
13
13
  voiceio/consent.py,sha256=jtdp1TkTbicUSTCDmV4qo4SdbS4kQpYzy1Kjc_7pJVw,1883
14
14
  voiceio/corrections.py,sha256=7moKhkVvo5CZU1mZobV2-MajkcHiwliKBY1dSBFh0w0,7445
15
15
  voiceio/demo.py,sha256=QRNJuNObXCsjwGD-Cz3dh81cfb21xiuSeVyyU720U9Q,8673
16
+ voiceio/evaluate.py,sha256=_4AZRBsa9fqIiYY1CASmzliy6-RmIjyEjiduFpINXRs,12019
16
17
  voiceio/feedback.py,sha256=TyUy_wy5HNTO0u3opx58aaPXz7mAqeN98UbNdLfnoiw,5136
17
18
  voiceio/health.py,sha256=5YNu1mMzwTo6I_Z_Mu3Bs46ZHDpVmxP0AXyMNAbaczc,15707
18
19
  voiceio/hints.py,sha256=dmoiMAJ0YGTWwFc9kIPvEs_BQJN7UEUiwYuLUD_GoQQ,1415
@@ -30,12 +31,14 @@ voiceio/retention.py,sha256=i_VWPCyykT4kQ1uKilTplpiIrACTYljn09BU1vL71A0,6019
30
31
  voiceio/service.py,sha256=e3j5OxhuwEqcgXlmQ9AhgdNrrZPwCNfzj-BmABuBYRw,12145
31
32
  voiceio/snapshots.py,sha256=bm-lzIeYucDUDlPBZiFiWd-jyB2I6Ft1jRANZCO6go4,2351
32
33
  voiceio/streaming.py,sha256=mXkQEEMGMbABSRNG2H7MYObKUdCwfqMvNIZISGZ5uic,25360
33
- voiceio/transcriber.py,sha256=NWYMZpIhYccUBKzcFyMzjH1MTMNv4bG9op5Qw5reOhI,11608
34
+ voiceio/tokens.py,sha256=BbOXnsDIfWYMVbSp25PYG1UtK7-7OtZ7q7l4kOIm9OY,5292
35
+ voiceio/transcriber.py,sha256=12o3POqS8nqJdXXQH-riZlSVChZPP_UP1bNLd7qNfTs,12303
34
36
  voiceio/vad.py,sha256=72_ICk4jSuSPYvTHdHYtHzY8W5uaGEOz5FTdqkolJi8,5327
35
- voiceio/vocabulary.py,sha256=eGV7QZs3tN6Cw5YpyoZsVEk2M5fJvie1n2XFi9CD4hw,6321
36
- voiceio/wizard.py,sha256=AwjwrocrQp0DPSOMZ0m6z6G3hlo9VPceEwPzR9-xXe0,80227
37
+ voiceio/vocab_stats.py,sha256=obz_mdUqlSoAD15wkIQM-1X3AXNum74VKDX_bu1mEZw,5403
38
+ voiceio/vocabulary.py,sha256=_7LjCS5Mqqh7PLKl8LMPIPsIlyfNKTaaLlfJ2BDfwtM,11355
39
+ voiceio/wizard.py,sha256=1zIYGixPsyu2OhNBnxk-DiErUgmVUvxhWvGoauiQXFY,80349
37
40
  voiceio/wordfreq.py,sha256=2UMjW1xIFqo5EchXzsXPizUhRb9N18CVgjnXso1wsgw,2382
38
- voiceio/worker.py,sha256=Ec_OChl6CCX1XHdNYdmx_4l4_sIRexyxih64-PBNhjs,3233
41
+ voiceio/worker.py,sha256=AkEYu2OzAup00ew24DYO7SfA91xTBnFvbtioGE0nsK4,3520
39
42
  voiceio/hotkeys/__init__.py,sha256=rGXSGZLD2mS_Ep2HfVdRdkUKNr4xagySlL1FM00cYgk,735
40
43
  voiceio/hotkeys/base.py,sha256=L3iBh368sjWpt64JZpocH9DibYEZ0V16-rYflLYbpZY,699
41
44
  voiceio/hotkeys/chain.py,sha256=8pvPhipmxWwmGG9DRXy8dlEKTPFILxNzYO-adcvw5Ps,2800
@@ -65,14 +68,14 @@ voiceio/tts/player.py,sha256=tyKeJzDHKCpIzlPzQGTWIuDHQJT4DDGbFD0hK11knuY,1698
65
68
  voiceio/typers/__init__.py,sha256=lFIMEjfPGw8tfYGELhsq0mYM849uKLx_SZhbHqGJrlo,1071
66
69
  voiceio/typers/base.py,sha256=VVMgARlpaSVqen-zFUssnbUMYDrWryacRE8MKvlLk-s,1180
67
70
  voiceio/typers/chain.py,sha256=F4QcAMTzYvrxeC-_Er4CbjnqQ1ONYjt7Fp11y8PNpqU,2964
68
- voiceio/typers/clipboard.py,sha256=cxeOL_a_P2HeJ-nr-klG94nHjbd9rWu3nxGDXmyfzys,6110
69
- voiceio/typers/ibus.py,sha256=ciaB5OgW5RqgLVdGZqoBhU4q1ttkh7o2H4aAoXSCqbA,18531
71
+ voiceio/typers/clipboard.py,sha256=8MJp3fEWCe3bvagqwevd3SgVVD830--AfEY0k87dkW0,7432
72
+ voiceio/typers/ibus.py,sha256=q2zXdjn68YUGSdHRvnflAhPN4-mE4Op2zKlOx_4FXYQ,20080
70
73
  voiceio/typers/pynput_type.py,sha256=DTtkT59M-EKOAlTNkvcZrRvf_Fq52OpI8S5t_8S-QAY,1887
71
74
  voiceio/typers/wtype.py,sha256=d1wG-HZdYDQxXIysV_Xvk9HixDpaJgl3VEoKbP5vhIs,1786
72
75
  voiceio/typers/xdotool.py,sha256=dh1zhPqUT8ihJahJ7ZKm6PtfYY087UAzADx664DvQOM,1356
73
76
  voiceio/typers/ydotool.py,sha256=dt0W9ot__W8LQG2so8vZLGIw9leK2Q6g3i-Voo0LXE4,5327
74
- python_voiceio-0.8.2.dist-info/METADATA,sha256=tHzCPzNpT-x1-89kZF547QCd-Qo8_XqiL7NIsia6gNw,15877
75
- python_voiceio-0.8.2.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
76
- python_voiceio-0.8.2.dist-info/entry_points.txt,sha256=U64fA65zxzyLoC8bgbn2ztQVWHLsc0o0H0qCk1J9DMc,218
77
- python_voiceio-0.8.2.dist-info/top_level.txt,sha256=piwtn309lD6uexQyXdZ-efAVBJF9y6Wfr48Z-8zkNhg,8
78
- python_voiceio-0.8.2.dist-info/RECORD,,
77
+ python_voiceio-0.9.0.dist-info/METADATA,sha256=5Wkd2wpiUlDzxWnslJI6HdpHLRr1zWMdmZcbvb8CS1o,15877
78
+ python_voiceio-0.9.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
79
+ python_voiceio-0.9.0.dist-info/entry_points.txt,sha256=U64fA65zxzyLoC8bgbn2ztQVWHLsc0o0H0qCk1J9DMc,218
80
+ python_voiceio-0.9.0.dist-info/top_level.txt,sha256=piwtn309lD6uexQyXdZ-efAVBJF9y6Wfr48Z-8zkNhg,8
81
+ python_voiceio-0.9.0.dist-info/RECORD,,
voiceio/__init__.py CHANGED
@@ -1 +1 @@
1
- __version__ = "0.8.2"
1
+ __version__ = "0.9.0"
voiceio/app.py CHANGED
@@ -97,10 +97,20 @@ def _import_graphical_env() -> None:
97
97
  _HEALTH_CHECK_INTERVAL = 10 # seconds between health checks
98
98
 
99
99
  # Whisper's decoder has a hard 448-token sequence budget shared by hotwords,
100
- # initial_prompt AND the transcription output. Hotwords + prompt must stay
101
- # well under half of it or output gets truncated mid-utterance.
102
- # ~600 chars ≈ 150 tokens.
103
- _HOTWORDS_MAX_CHARS = 600
100
+ # initial_prompt AND the transcription output. Hotwords + prompt must stay well
101
+ # under half of it or output gets truncated mid-utterance.
102
+ #
103
+ # This is a TOKEN budget, counted for real (voiceio.tokens) — the old 600-char
104
+ # proxy assumed ~4 chars/token, but proper nouns (the whole point of a custom
105
+ # vocabulary) run ~2.6, so "600 chars" was really 238 tokens: already past
106
+ # faster-whisper's own 223-token hotwords cap, which then silently dropped the
107
+ # tail.
108
+ #
109
+ # Derived, not guessed: 448 total - 200 output reserve - 120 prompt - 4 overhead
110
+ # = 124. At ~3.5 tokens/term that is ~34 terms. Fewer terms than the old config
111
+ # pretended to carry, but they are real: the old 223-token list left ~50 tokens
112
+ # to transcribe with and truncated freeze chunks mid-sentence.
113
+ _HOTWORDS_TOKEN_BUDGET = 120
104
114
 
105
115
 
106
116
  class _State(enum.Enum):
@@ -309,30 +319,56 @@ class VoiceIO:
309
319
  self._do_start()
310
320
 
311
321
  def _refresh_hotwords(self) -> None:
312
- """Rebuild Whisper hotwords from vocabulary + correction targets.
322
+ """Rebuild Whisper hotwords: rank by usage, fill the token budget.
313
323
 
314
- The vocabulary read is mtime-cached (only a `stat` when unchanged), and
315
- `set_hotwords` is called only when the merged string actually changed,
316
- so this stays cheap enough to run on every recording start.
324
+ Selection is mandatory and permanent the budget holds ~40-60 terms and
325
+ the vocabulary only grows — so the slots go to the terms actually being
326
+ dictated (voiceio.vocab_stats), highest-scoring first.
327
+
328
+ Both reads are mtime-cached and `set_hotwords` only fires when the string
329
+ changed, so the steady-state cost here is a couple of `stat` calls. That
330
+ matters: this runs synchronously before `recorder.start()`.
317
331
  """
318
- vocab = self._vocab_loader.get()
332
+ terms = self._vocab_loader.get_selected(token_budget=_HOTWORDS_TOKEN_BUDGET)
319
333
  # Only merge RARE correction targets (proper nouns, technical terms):
320
334
  # common-word targets ("review", "company") add nothing to decoder bias
321
335
  # and hundreds of them blow Whisper's 448-token prompt+output budget,
322
336
  # which truncates transcriptions mid-utterance.
323
- from voiceio.wordfreq import is_common
324
- vocab_terms = [
325
- t for t in self._corrections.vocabulary_terms()
326
- if not is_common(t, self.cfg.model.language)
327
- ]
328
- if vocab_terms:
329
- extra = ", ".join(vocab_terms)
330
- vocab = f"{vocab}, {extra}" if vocab else extra
331
- vocab = vocab[:_HOTWORDS_MAX_CHARS]
337
+ vocab = ", ".join(terms + self._rare_correction_targets())
332
338
  if vocab != self._hotwords:
333
339
  self._hotwords = vocab
334
- if vocab:
335
- self.transcriber.set_hotwords(vocab)
340
+ # Set unconditionally: an emptied vocabulary must CLEAR the hotwords,
341
+ # not leave the previous string biasing every future decode.
342
+ self.transcriber.set_hotwords(vocab or None)
343
+
344
+ def _record_vocab_usage(self, text: str) -> None:
345
+ """Record which vocabulary terms this utterance used (finalize path only).
346
+
347
+ This is the feedback signal that decides who gets the hotword budget next
348
+ time. Deliberately not on the recording-start path, and never fatal — a
349
+ usage counter must not be able to break a dictation.
350
+ """
351
+ try:
352
+ from voiceio import vocab_stats
353
+ vocab_stats.update_from_text(text, self._vocab_loader.get_all())
354
+ except Exception: # noqa: BLE001
355
+ log.debug("vocab usage update failed", exc_info=True)
356
+
357
+ def _rare_correction_targets(self) -> list[str]:
358
+ """Correction targets worth biasing, cached against corrections mtime.
359
+
360
+ `is_common` over every target used to be recomputed on each recording
361
+ start; the result only changes when corrections.json does.
362
+ """
363
+ mtime = getattr(self._corrections, "_mtime", None)
364
+ if getattr(self, "_rare_targets_mtime", object()) != mtime:
365
+ from voiceio.wordfreq import is_common
366
+ self._rare_targets = [
367
+ t for t in self._corrections.vocabulary_terms()
368
+ if not is_common(t, self.cfg.model.language)
369
+ ]
370
+ self._rare_targets_mtime = mtime
371
+ return self._rare_targets
336
372
 
337
373
  def _do_start(self) -> None:
338
374
  """Transition to RECORDING."""
@@ -432,8 +468,13 @@ class VoiceIO:
432
468
  if self._postcorrect is not None:
433
469
  # Freshest context for the finalize pass: this recording's window
434
470
  # title (captured at start) and the latest transcripts/vocabulary.
471
+ #
472
+ # The FULL vocabulary, not the hotwords string: the decoder can only
473
+ # be biased with ~40-60 terms, but this is an LLM with no 448-token
474
+ # budget. The long tail that can never fit in hotwords gets its only
475
+ # chance to be corrected here.
435
476
  self._postcorrect.set_context(
436
- vocabulary=self._hotwords,
477
+ vocabulary=", ".join(self._vocab_loader.get_all()),
437
478
  recent=self._prompt_builder.recent(3),
438
479
  title=self._context_title,
439
480
  )
@@ -552,6 +593,7 @@ class VoiceIO:
552
593
  self._play_feedback(final_text)
553
594
  stored = self._strip_voice_prefix(final_text)
554
595
  self._prompt_builder.add_transcript(stored)
596
+ self._record_vocab_usage(stored)
555
597
  latency = dict(session.final_latency)
556
598
  # stop-to-commit: what the user actually waits for
557
599
  latency["finalize_total"] = round(time.monotonic() - t_final, 3)
voiceio/audit.py CHANGED
@@ -169,9 +169,11 @@ def _read_metrics(path: Path | None = None) -> list[dict]:
169
169
 
170
170
  def _default_teacher():
171
171
  """Lazily build the offline teacher transcriber (distil-large-v3)."""
172
- from faster_whisper import WhisperModel
172
+ from voiceio.worker import load_model
173
173
 
174
- model = WhisperModel("distil-large-v3", device="cpu", compute_type="int8")
174
+ # Cache-first: the module docstring's "nothing here uses the network" only
175
+ # holds if the load itself can't reach out and hang (see load_model).
176
+ model = load_model("distil-large-v3")
175
177
 
176
178
  def _transcribe(wav_path: Path) -> str:
177
179
  segments, _ = model.transcribe(
@@ -323,14 +325,42 @@ def run_audit(
323
325
  return report
324
326
 
325
327
 
328
+ def _last_seen_map(terms: list[str]) -> dict[str, float]:
329
+ """When each term was last dictated.
330
+
331
+ Prefers vocab_stats.json, which the runtime maintains per utterance. Falls
332
+ back to scanning history only when no stats exist yet (fresh install, or a
333
+ daemon that hasn't finalized a dictation since stats landed) — that scan is
334
+ O(entries x terms) over the whole history, which at 6k entries and 300+
335
+ terms is not worth running when the answer is already recorded.
336
+ """
337
+ from voiceio.vocab_stats import VocabStats
338
+
339
+ stats = VocabStats()
340
+ stats.load()
341
+ if stats.as_dict():
342
+ return {t: float(stats.get(t).get("last_seen_ts", 0.0) or 0.0) for t in terms}
343
+
344
+ from voiceio import history
345
+
346
+ last_seen: dict[str, float] = {}
347
+ for e in history.read(limit=0):
348
+ text = (e.get("text") or "").lower()
349
+ ts = e.get("ts", 0.0)
350
+ for term in terms:
351
+ if term.lower() in text and ts > last_seen.get(term, 0.0):
352
+ last_seen[term] = ts
353
+ return last_seen
354
+
355
+
326
356
  def _audit_vocabulary(cfg, teacher_texts, now, report) -> None:
327
357
  """Confirm vocab terms the teacher heard; age out terms unseen for 90 days.
328
358
 
329
- Aging moves a term to the bottom of vocabulary.txt rather than deleting it:
330
- the 800-char loader truncates the tail, so stale terms fall out of the
331
- hotword budget but can rejoin if used again.
359
+ Aging moves a term to the bottom of vocabulary.txt rather than deleting it.
360
+ File order is now only the tie-break for terms with no usage signal —
361
+ vocabulary.select_terms ranks by vocab_stats first but keeping stale terms
362
+ at the bottom still means they lose ties, and they rejoin on next use.
332
363
  """
333
- from voiceio import history
334
364
  from voiceio.vocabulary import _read_terms, resolve_vocab_path
335
365
 
336
366
  path = resolve_vocab_path(cfg.model)
@@ -341,16 +371,7 @@ def _audit_vocabulary(cfg, teacher_texts, now, report) -> None:
341
371
  return
342
372
 
343
373
  teacher_blob = "\n".join(teacher_texts)
344
- # Last time each term appeared in a live transcript.
345
- entries = history.read(limit=0)
346
- last_seen: dict[str, float] = {}
347
- for e in entries:
348
- text = e.get("text", "")
349
- ts = e.get("ts", 0.0)
350
- low = text.lower()
351
- for term in terms:
352
- if term.lower() in low and ts > last_seen.get(term, 0.0):
353
- last_seen[term] = ts
374
+ last_seen = _last_seen_map(terms)
354
375
 
355
376
  fresh: list[str] = []
356
377
  aged: list[str] = []
voiceio/cli.py CHANGED
@@ -83,6 +83,8 @@ def main() -> None:
83
83
  p_correct.add_argument("wrong", nargs="?", help="Misheard word/phrase")
84
84
  p_correct.add_argument("right", nargs="?", help="Correct replacement")
85
85
  p_correct.add_argument("--list", action="store_true", help="List all corrections")
86
+ p_correct.add_argument("--vocab", action="store_true",
87
+ help="Show vocabulary and which terms fit the decoder budget")
86
88
  p_correct.add_argument("--remove", metavar="WORD", help="Remove a correction")
87
89
  p_correct.add_argument("--flagged", action="store_true",
88
90
  help="Show words flagged by 'correct that'")
@@ -108,6 +110,14 @@ def main() -> None:
108
110
  p_history.add_argument("--path", action="store_true",
109
111
  help="Print history file path")
110
112
 
113
+ # ── voiceio eval ───────────────────────────────────────────────────
114
+ p_eval = sub.add_parser(
115
+ "eval", help="Replay your retained audio across decoder configs and score them")
116
+ p_eval.add_argument("--max-audio-secs", type=float, default=900,
117
+ help="Audio budget to replay (default: 900 = 15min)")
118
+ p_eval.add_argument("--json", action="store_true",
119
+ help="Emit raw results as JSON")
120
+
111
121
  # ── voiceio demo ──────────────────────────────────────────────────
112
122
  sub.add_parser("demo", help="Interactive guided tour of voiceio features")
113
123
 
@@ -134,6 +144,8 @@ def main() -> None:
134
144
  _cmd_correct(args)
135
145
  elif args.command == "history":
136
146
  _cmd_history(args)
147
+ elif args.command == "eval":
148
+ _cmd_eval(args)
137
149
  elif args.command == "demo":
138
150
  _cmd_demo()
139
151
  elif args.command == "logs":
@@ -718,6 +730,138 @@ def _cmd_uninstall() -> None:
718
730
  print("\nvoiceio fully removed.")
719
731
 
720
732
 
733
+ def _cmd_eval(args: argparse.Namespace) -> None:
734
+ """Score decoder configs against a teacher model on your own audio."""
735
+ import json as _json
736
+ import sys
737
+ import time
738
+
739
+ from voiceio.config import load
740
+ from voiceio.evaluate import default_matrix, evaluate
741
+ from voiceio.wizard import BOLD, DIM, GREEN, RESET, YELLOW
742
+
743
+ cfg = load()
744
+ matrix = default_matrix()
745
+
746
+ print(f"{BOLD}Evaluating {len(matrix)} decoder configs{RESET} "
747
+ f"{DIM}on up to {args.max_audio_secs:.0f}s of your retained audio{RESET}")
748
+ print(f"{DIM}Scored against a teacher model (distil-large-v3) — a proxy for")
749
+ print(f"ground truth, good for ranking configs, not for absolute accuracy.{RESET}\n")
750
+
751
+ last = [""]
752
+
753
+ def progress(name, i, n):
754
+ if name != last[0]:
755
+ last[0] = name
756
+ sys.stderr.write(f"\n {name}: ")
757
+ sys.stderr.write(".")
758
+ sys.stderr.flush()
759
+
760
+ t0 = time.time()
761
+ scores = evaluate(cfg, matrix, max_audio_secs=args.max_audio_secs,
762
+ on_progress=progress)
763
+ sys.stderr.write("\n\n")
764
+
765
+ if not scores:
766
+ print("No retained audio to evaluate. Dictate a few notes first "
767
+ "(audio is kept under [data] retain_audio).")
768
+ return
769
+
770
+ if args.json:
771
+ print(_json.dumps([{
772
+ "config": s.config.name, **s.config.as_dict(),
773
+ "clips": s.clips, "audio_secs": round(s.audio_secs, 1),
774
+ "wer": s.wer, "hallucinations": s.hallucinations,
775
+ "repetitions": s.repetitions, "empties": s.empties,
776
+ "crashes": s.crashes, "decode_secs": round(s.decode_secs, 1),
777
+ } for s in scores], indent=2))
778
+ return
779
+
780
+ best = scores[0]
781
+ shipped = next((s for s in scores if s.config.name == "shipped"), None)
782
+
783
+ print(f" {'config':<24} {'WER':>7} {'halluc':>7} {'repeat':>7} "
784
+ f"{'empty':>6} {'crash':>6}")
785
+ print(f" {'-'*24} {'-'*7} {'-'*7} {'-'*7} {'-'*6} {'-'*6}")
786
+ for s in scores:
787
+ mark = f" {GREEN}<- best{RESET}" if s is best else ""
788
+ if s is shipped and s is not best:
789
+ mark = f" {DIM}<- shipped{RESET}"
790
+ elif s is shipped:
791
+ mark = f" {GREEN}<- best (shipped){RESET}"
792
+ print(f" {s.config.name:<24} {s.wer:>7.3f} {s.hallucinations:>7} "
793
+ f"{s.repetitions:>7} {s.empties:>6} {s.crashes:>6}{mark}")
794
+
795
+ print(f"\n {DIM}{best.clips} clips, {best.audio_secs:.0f}s audio, "
796
+ f"{time.time()-t0:.0f}s total{RESET}")
797
+
798
+ if shipped and best is not shipped:
799
+ delta = (shipped.wer - best.wer) / shipped.wer * 100 if shipped.wer else 0
800
+ print(f"\n {YELLOW}'{best.config.name}' beats the shipped config by "
801
+ f"{delta:.0f}% WER.{RESET}")
802
+ # A WER win means nothing if it comes from the failure mode the knob
803
+ # exists to prevent, so make that trade explicit rather than implied.
804
+ if best.hallucinations > shipped.hallucinations:
805
+ print(f" {YELLOW}But it hallucinates more "
806
+ f"({best.hallucinations} vs {shipped.hallucinations}) — "
807
+ f"that's what vad_filter is for.{RESET}")
808
+ if best.repetitions > shipped.repetitions:
809
+ print(f" {YELLOW}But it loops more "
810
+ f"({best.repetitions} vs {shipped.repetitions}) — "
811
+ f"that's what condition_on_previous_text=False is for.{RESET}")
812
+ print(f" {DIM}The teacher is a proxy, not truth. Spot-check before "
813
+ f"changing a default.{RESET}")
814
+ elif shipped:
815
+ print(f"\n {GREEN}The shipped config wins. No change indicated.{RESET}")
816
+
817
+
818
+ def _show_vocab() -> None:
819
+ """Show the vocabulary split by what actually reaches the decoder.
820
+
821
+ Whisper can only be biased with ~35 terms (a hard 223-token hotwords cap,
822
+ and proper nouns cost 5-7 tokens each). Everything past that is not wasted —
823
+ it still reaches the post-correction LLM — but it does NOT bias decoding,
824
+ and until now there was no way to see which side of the line a term fell on.
825
+ """
826
+ from voiceio.app import _HOTWORDS_TOKEN_BUDGET
827
+ from voiceio.config import load
828
+ from voiceio.tokens import count_tokens
829
+ from voiceio.vocab_stats import VocabStats
830
+ from voiceio.vocabulary import load_terms, select_terms
831
+ from voiceio.wizard import BOLD, DIM, GREEN, RESET
832
+
833
+ cfg = load()
834
+ terms = load_terms(cfg.model)
835
+ if not terms:
836
+ print("No vocabulary configured.")
837
+ print("\nTerms are learned automatically: voiceio correct --auto")
838
+ return
839
+
840
+ stats = VocabStats()
841
+ stats.load()
842
+ selected = select_terms(terms, token_budget=_HOTWORDS_TOKEN_BUDGET,
843
+ model_name=cfg.model.name, stats=stats)
844
+ used = count_tokens(", ".join(selected), cfg.model.name)
845
+ sel = set(selected)
846
+
847
+ print(f"{BOLD}Biasing the decoder{RESET} "
848
+ f"{DIM}({used}/{_HOTWORDS_TOKEN_BUDGET} tokens){RESET}")
849
+ for t in selected:
850
+ rec = stats.get(t)
851
+ hits = rec.get("hits", 0)
852
+ print(f" {GREEN}●{RESET} {t}" + (f" {DIM}({hits} uses){RESET}" if hits else ""))
853
+
854
+ tail = [t for t in terms if t not in sel]
855
+ if tail:
856
+ print(f"\n{BOLD}Post-correction only{RESET} "
857
+ f"{DIM}({len(tail)} terms — past the decoder's token budget,{RESET}")
858
+ print(f" {DIM}still used by the correction pass){RESET}")
859
+ shown = ", ".join(tail[:12])
860
+ print(f" {DIM}{shown}{'...' if len(tail) > 12 else ''}{RESET}")
861
+
862
+ print(f"\n{len(terms)} term(s): {len(selected)} biasing, {len(tail)} tail")
863
+
864
+
721
865
  def _cmd_correct(args: argparse.Namespace) -> None:
722
866
  """Manage the corrections dictionary."""
723
867
  from voiceio.corrections import CorrectionDict
@@ -734,6 +878,10 @@ def _cmd_correct(args: argparse.Namespace) -> None:
734
878
  print(f"\n{len(corrections)} correction(s)")
735
879
  return
736
880
 
881
+ if args.vocab:
882
+ _show_vocab()
883
+ return
884
+
737
885
  if args.remove:
738
886
  if cd.remove(args.remove):
739
887
  print(f"Removed: {args.remove}")
@@ -936,12 +1084,20 @@ def _cmd_correct_auto(cd, *, full: bool = False, batch: bool = False) -> None:
936
1084
  # Build lookup for O(1) access to suspicious word metadata
937
1085
  sw_by_word = {sw.word: sw for sw in suspicious}
938
1086
 
1087
+ # Correction mining is retired by default ([autocorrect] mine_corrections).
1088
+ # The vocabulary bucket below is unaffected: those terms bias the decoder
1089
+ # and measurably help, whereas mined find/replace rules fired 4 times in 387
1090
+ # while postcorrect applied 288 edits over the same span. Skipping the
1091
+ # correction buckets also skips adjudication — the expensive part, at
1092
+ # votes x ceil(N/25) LLM calls per run.
1093
+ mine_corrections = cfg.autocorrect.mine_corrections
1094
+
939
1095
  # ── Bucket 1: Auto-fix (high confidence, safety-gated) ──────────────
940
1096
  # Pairs failing the gate (real word being "corrected", or target that is
941
1097
  # itself junk) are downgraded to manual review rather than persisted.
942
1098
  auto_fixed = 0
943
1099
  downgraded: list[dict] = []
944
- if result.auto_fix:
1100
+ if result.auto_fix and mine_corrections:
945
1101
  gated = []
946
1102
  for fix in result.auto_fix:
947
1103
  reason = gate_correction(
@@ -975,10 +1131,13 @@ def _cmd_correct_auto(cd, *, full: bool = False, batch: bool = False) -> None:
975
1131
  )
976
1132
 
977
1133
  # ── Bucket 2: Ask user (ambiguous) ──────────────────────────────────
978
- to_review = downgraded + list(result.ask_user)
1134
+ # Empty when correction mining is retired — nothing to adjudicate or review.
1135
+ to_review = (downgraded + list(result.ask_user)) if mine_corrections else []
979
1136
 
980
1137
  # If no LLM was used, put all suspicious words into manual review
981
- if not has_api and not has_ollama:
1138
+ if not mine_corrections:
1139
+ pass # correction mining retired — nothing to review or adjudicate
1140
+ elif not has_api and not has_ollama:
982
1141
  for sw in suspicious:
983
1142
  to_review.append({
984
1143
  "wrong": sw.word,
@@ -1034,8 +1193,13 @@ def _cmd_correct_auto(cd, *, full: bool = False, batch: bool = False) -> None:
1034
1193
  print(f" {DIM}{len(capped)} lower-frequency item(s) deferred "
1035
1194
  f"to next run{RESET}")
1036
1195
 
1037
- adj = adjudicate(cfg, to_review, sw_by_word,
1038
- vocabulary=vocab_words, language=language)
1196
+ if to_review:
1197
+ adj = adjudicate(cfg, to_review, sw_by_word,
1198
+ vocabulary=vocab_words, language=language)
1199
+ else:
1200
+ # Nothing to weigh — don't spend votes x ceil(N/25) LLM calls proving it.
1201
+ from voiceio.autocorrect import AdjudicationResult
1202
+ adj = AdjudicationResult()
1039
1203
 
1040
1204
  # Unanimous corrections (already gate-passed inside adjudicate).
1041
1205
  adj_applied = 0
voiceio/config.py CHANGED
@@ -138,6 +138,17 @@ class AutocorrectConfig:
138
138
  # Languages you also dictate in: mined corrections never rewrite words
139
139
  # that are real in these (e.g. ["es"] protects Spanish "harina").
140
140
  protect_languages: list[str] = field(default_factory=list)
141
+ # Mine find/replace CORRECTION rules from history. Off by default: measured
142
+ # over months of real use, mining produced 387 rules of which 4 ever fired
143
+ # (1%), while the runtime postcorrect pass applied 288 edits over the same
144
+ # period. Rules are exact-string matches, so they only fire when Whisper
145
+ # repeats a misrecognition verbatim — and the errors worth fixing are
146
+ # common words ("nuance" for "neurons") that the safety gate must refuse
147
+ # anyway. Meanwhile a bad rule is a silent regression: one run learned rules
148
+ # destroying real Spanish words. Poor odds, real downside, and postcorrect
149
+ # already does this contextually. Vocabulary mining is unaffected and stays
150
+ # on — those terms bias the decoder and demonstrably help.
151
+ mine_corrections: bool = False
141
152
 
142
153
 
143
154
  @dataclass