python-voiceio 0.8.2__py3-none-any.whl → 0.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {python_voiceio-0.8.2.dist-info → python_voiceio-0.9.0.dist-info}/METADATA +1 -1
- {python_voiceio-0.8.2.dist-info → python_voiceio-0.9.0.dist-info}/RECORD +20 -17
- voiceio/__init__.py +1 -1
- voiceio/app.py +63 -21
- voiceio/audit.py +37 -16
- voiceio/cli.py +169 -5
- voiceio/config.py +11 -0
- voiceio/evaluate.py +322 -0
- voiceio/tokens.py +138 -0
- voiceio/transcriber.py +10 -0
- voiceio/typers/clipboard.py +26 -1
- voiceio/typers/ibus.py +34 -4
- voiceio/vocab_stats.py +159 -0
- voiceio/vocabulary.py +167 -37
- voiceio/wizard.py +4 -2
- voiceio/worker.py +17 -9
- {python_voiceio-0.8.2.dist-info → python_voiceio-0.9.0.dist-info}/WHEEL +0 -0
- {python_voiceio-0.8.2.dist-info → python_voiceio-0.9.0.dist-info}/entry_points.txt +0 -0
- {python_voiceio-0.8.2.dist-info → python_voiceio-0.9.0.dist-info}/licenses/LICENSE +0 -0
- {python_voiceio-0.8.2.dist-info → python_voiceio-0.9.0.dist-info}/top_level.txt +0 -0
|
@@ -1,18 +1,19 @@
|
|
|
1
|
-
python_voiceio-0.
|
|
2
|
-
voiceio/__init__.py,sha256=
|
|
1
|
+
python_voiceio-0.9.0.dist-info/licenses/LICENSE,sha256=Gz61o8jFTAvZUZyB3nWDB3DQQVuipjfPkVu9W8hBHM0,1072
|
|
2
|
+
voiceio/__init__.py,sha256=H9NWRZb7NbeRRPLP_V1fARmLNXranorVM-OOY-8_2ug,22
|
|
3
3
|
voiceio/__main__.py,sha256=xT5QCGGreYMisHO7Lh_Y-xAQ2TOzG2D7npKknGNcWSY,53
|
|
4
|
-
voiceio/app.py,sha256=
|
|
5
|
-
voiceio/audit.py,sha256=
|
|
4
|
+
voiceio/app.py,sha256=ScrUVDQQ_YbAFKKSMOhaRPX_WTcPuenhnRdSzWMxThE,73785
|
|
5
|
+
voiceio/audit.py,sha256=p4F2BBX4wEC9eS8MSplzfXL52XC2dyIKbgrRvR-WbTU,15802
|
|
6
6
|
voiceio/autocorrect.py,sha256=jGsglEYwAHWk2uWedeEZzyAMW2LWO_8nQGZgrWYy4Dc,26552
|
|
7
7
|
voiceio/autocorrect_state.py,sha256=BNzdHP5slcOV_-OU4MlaKE7xgVJzrPncsYFK30JWTs8,6858
|
|
8
8
|
voiceio/backends.py,sha256=-rs1YhPblub4NjFb6qCD7Ih9FqBRXtPuvjVv9bQoQ1w,368
|
|
9
|
-
voiceio/cli.py,sha256=
|
|
9
|
+
voiceio/cli.py,sha256=QE8N7N3d1O7r9-u_jbYMAH9gw0ILX7dSxDZZYj83lIs,78969
|
|
10
10
|
voiceio/clipboard_read.py,sha256=lz75CkRGh7pyvO6xvoUmxJDl-Gwie1xlXQHK70PO84I,4261
|
|
11
11
|
voiceio/commands.py,sha256=Vhtn8s5G5OcCbRN44YZ-_14Fa85ISXl_3Ix-KqSHs2A,4694
|
|
12
|
-
voiceio/config.py,sha256=
|
|
12
|
+
voiceio/config.py,sha256=DA1PGIanhrPjWkVT6jj0NSFslW9vuegzV6eMJNuKBzM,15273
|
|
13
13
|
voiceio/consent.py,sha256=jtdp1TkTbicUSTCDmV4qo4SdbS4kQpYzy1Kjc_7pJVw,1883
|
|
14
14
|
voiceio/corrections.py,sha256=7moKhkVvo5CZU1mZobV2-MajkcHiwliKBY1dSBFh0w0,7445
|
|
15
15
|
voiceio/demo.py,sha256=QRNJuNObXCsjwGD-Cz3dh81cfb21xiuSeVyyU720U9Q,8673
|
|
16
|
+
voiceio/evaluate.py,sha256=_4AZRBsa9fqIiYY1CASmzliy6-RmIjyEjiduFpINXRs,12019
|
|
16
17
|
voiceio/feedback.py,sha256=TyUy_wy5HNTO0u3opx58aaPXz7mAqeN98UbNdLfnoiw,5136
|
|
17
18
|
voiceio/health.py,sha256=5YNu1mMzwTo6I_Z_Mu3Bs46ZHDpVmxP0AXyMNAbaczc,15707
|
|
18
19
|
voiceio/hints.py,sha256=dmoiMAJ0YGTWwFc9kIPvEs_BQJN7UEUiwYuLUD_GoQQ,1415
|
|
@@ -30,12 +31,14 @@ voiceio/retention.py,sha256=i_VWPCyykT4kQ1uKilTplpiIrACTYljn09BU1vL71A0,6019
|
|
|
30
31
|
voiceio/service.py,sha256=e3j5OxhuwEqcgXlmQ9AhgdNrrZPwCNfzj-BmABuBYRw,12145
|
|
31
32
|
voiceio/snapshots.py,sha256=bm-lzIeYucDUDlPBZiFiWd-jyB2I6Ft1jRANZCO6go4,2351
|
|
32
33
|
voiceio/streaming.py,sha256=mXkQEEMGMbABSRNG2H7MYObKUdCwfqMvNIZISGZ5uic,25360
|
|
33
|
-
voiceio/
|
|
34
|
+
voiceio/tokens.py,sha256=BbOXnsDIfWYMVbSp25PYG1UtK7-7OtZ7q7l4kOIm9OY,5292
|
|
35
|
+
voiceio/transcriber.py,sha256=12o3POqS8nqJdXXQH-riZlSVChZPP_UP1bNLd7qNfTs,12303
|
|
34
36
|
voiceio/vad.py,sha256=72_ICk4jSuSPYvTHdHYtHzY8W5uaGEOz5FTdqkolJi8,5327
|
|
35
|
-
voiceio/
|
|
36
|
-
voiceio/
|
|
37
|
+
voiceio/vocab_stats.py,sha256=obz_mdUqlSoAD15wkIQM-1X3AXNum74VKDX_bu1mEZw,5403
|
|
38
|
+
voiceio/vocabulary.py,sha256=_7LjCS5Mqqh7PLKl8LMPIPsIlyfNKTaaLlfJ2BDfwtM,11355
|
|
39
|
+
voiceio/wizard.py,sha256=1zIYGixPsyu2OhNBnxk-DiErUgmVUvxhWvGoauiQXFY,80349
|
|
37
40
|
voiceio/wordfreq.py,sha256=2UMjW1xIFqo5EchXzsXPizUhRb9N18CVgjnXso1wsgw,2382
|
|
38
|
-
voiceio/worker.py,sha256=
|
|
41
|
+
voiceio/worker.py,sha256=AkEYu2OzAup00ew24DYO7SfA91xTBnFvbtioGE0nsK4,3520
|
|
39
42
|
voiceio/hotkeys/__init__.py,sha256=rGXSGZLD2mS_Ep2HfVdRdkUKNr4xagySlL1FM00cYgk,735
|
|
40
43
|
voiceio/hotkeys/base.py,sha256=L3iBh368sjWpt64JZpocH9DibYEZ0V16-rYflLYbpZY,699
|
|
41
44
|
voiceio/hotkeys/chain.py,sha256=8pvPhipmxWwmGG9DRXy8dlEKTPFILxNzYO-adcvw5Ps,2800
|
|
@@ -65,14 +68,14 @@ voiceio/tts/player.py,sha256=tyKeJzDHKCpIzlPzQGTWIuDHQJT4DDGbFD0hK11knuY,1698
|
|
|
65
68
|
voiceio/typers/__init__.py,sha256=lFIMEjfPGw8tfYGELhsq0mYM849uKLx_SZhbHqGJrlo,1071
|
|
66
69
|
voiceio/typers/base.py,sha256=VVMgARlpaSVqen-zFUssnbUMYDrWryacRE8MKvlLk-s,1180
|
|
67
70
|
voiceio/typers/chain.py,sha256=F4QcAMTzYvrxeC-_Er4CbjnqQ1ONYjt7Fp11y8PNpqU,2964
|
|
68
|
-
voiceio/typers/clipboard.py,sha256=
|
|
69
|
-
voiceio/typers/ibus.py,sha256=
|
|
71
|
+
voiceio/typers/clipboard.py,sha256=8MJp3fEWCe3bvagqwevd3SgVVD830--AfEY0k87dkW0,7432
|
|
72
|
+
voiceio/typers/ibus.py,sha256=q2zXdjn68YUGSdHRvnflAhPN4-mE4Op2zKlOx_4FXYQ,20080
|
|
70
73
|
voiceio/typers/pynput_type.py,sha256=DTtkT59M-EKOAlTNkvcZrRvf_Fq52OpI8S5t_8S-QAY,1887
|
|
71
74
|
voiceio/typers/wtype.py,sha256=d1wG-HZdYDQxXIysV_Xvk9HixDpaJgl3VEoKbP5vhIs,1786
|
|
72
75
|
voiceio/typers/xdotool.py,sha256=dh1zhPqUT8ihJahJ7ZKm6PtfYY087UAzADx664DvQOM,1356
|
|
73
76
|
voiceio/typers/ydotool.py,sha256=dt0W9ot__W8LQG2so8vZLGIw9leK2Q6g3i-Voo0LXE4,5327
|
|
74
|
-
python_voiceio-0.
|
|
75
|
-
python_voiceio-0.
|
|
76
|
-
python_voiceio-0.
|
|
77
|
-
python_voiceio-0.
|
|
78
|
-
python_voiceio-0.
|
|
77
|
+
python_voiceio-0.9.0.dist-info/METADATA,sha256=5Wkd2wpiUlDzxWnslJI6HdpHLRr1zWMdmZcbvb8CS1o,15877
|
|
78
|
+
python_voiceio-0.9.0.dist-info/WHEEL,sha256=K260EYznzXsJYBQGqmI8VTxEdiZYNvDZwW9cBh9-_MA,91
|
|
79
|
+
python_voiceio-0.9.0.dist-info/entry_points.txt,sha256=U64fA65zxzyLoC8bgbn2ztQVWHLsc0o0H0qCk1J9DMc,218
|
|
80
|
+
python_voiceio-0.9.0.dist-info/top_level.txt,sha256=piwtn309lD6uexQyXdZ-efAVBJF9y6Wfr48Z-8zkNhg,8
|
|
81
|
+
python_voiceio-0.9.0.dist-info/RECORD,,
|
voiceio/__init__.py
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
__version__ = "0.
|
|
1
|
+
__version__ = "0.9.0"
|
voiceio/app.py
CHANGED
|
@@ -97,10 +97,20 @@ def _import_graphical_env() -> None:
|
|
|
97
97
|
_HEALTH_CHECK_INTERVAL = 10 # seconds between health checks
|
|
98
98
|
|
|
99
99
|
# Whisper's decoder has a hard 448-token sequence budget shared by hotwords,
|
|
100
|
-
# initial_prompt AND the transcription output. Hotwords + prompt must stay
|
|
101
|
-
#
|
|
102
|
-
#
|
|
103
|
-
|
|
100
|
+
# initial_prompt AND the transcription output. Hotwords + prompt must stay well
|
|
101
|
+
# under half of it or output gets truncated mid-utterance.
|
|
102
|
+
#
|
|
103
|
+
# This is a TOKEN budget, counted for real (voiceio.tokens) — the old 600-char
|
|
104
|
+
# proxy assumed ~4 chars/token, but proper nouns (the whole point of a custom
|
|
105
|
+
# vocabulary) run ~2.6, so "600 chars" was really 238 tokens: already past
|
|
106
|
+
# faster-whisper's own 223-token hotwords cap, which then silently dropped the
|
|
107
|
+
# tail.
|
|
108
|
+
#
|
|
109
|
+
# Derived, not guessed: 448 total - 200 output reserve - 120 prompt - 4 overhead
|
|
110
|
+
# = 124. At ~3.5 tokens/term that is ~34 terms. Fewer terms than the old config
|
|
111
|
+
# pretended to carry, but they are real: the old 223-token list left ~50 tokens
|
|
112
|
+
# to transcribe with and truncated freeze chunks mid-sentence.
|
|
113
|
+
_HOTWORDS_TOKEN_BUDGET = 120
|
|
104
114
|
|
|
105
115
|
|
|
106
116
|
class _State(enum.Enum):
|
|
@@ -309,30 +319,56 @@ class VoiceIO:
|
|
|
309
319
|
self._do_start()
|
|
310
320
|
|
|
311
321
|
def _refresh_hotwords(self) -> None:
|
|
312
|
-
"""Rebuild Whisper hotwords
|
|
322
|
+
"""Rebuild Whisper hotwords: rank by usage, fill the token budget.
|
|
313
323
|
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
324
|
+
Selection is mandatory and permanent — the budget holds ~40-60 terms and
|
|
325
|
+
the vocabulary only grows — so the slots go to the terms actually being
|
|
326
|
+
dictated (voiceio.vocab_stats), highest-scoring first.
|
|
327
|
+
|
|
328
|
+
Both reads are mtime-cached and `set_hotwords` only fires when the string
|
|
329
|
+
changed, so the steady-state cost here is a couple of `stat` calls. That
|
|
330
|
+
matters: this runs synchronously before `recorder.start()`.
|
|
317
331
|
"""
|
|
318
|
-
|
|
332
|
+
terms = self._vocab_loader.get_selected(token_budget=_HOTWORDS_TOKEN_BUDGET)
|
|
319
333
|
# Only merge RARE correction targets (proper nouns, technical terms):
|
|
320
334
|
# common-word targets ("review", "company") add nothing to decoder bias
|
|
321
335
|
# and hundreds of them blow Whisper's 448-token prompt+output budget,
|
|
322
336
|
# which truncates transcriptions mid-utterance.
|
|
323
|
-
|
|
324
|
-
vocab_terms = [
|
|
325
|
-
t for t in self._corrections.vocabulary_terms()
|
|
326
|
-
if not is_common(t, self.cfg.model.language)
|
|
327
|
-
]
|
|
328
|
-
if vocab_terms:
|
|
329
|
-
extra = ", ".join(vocab_terms)
|
|
330
|
-
vocab = f"{vocab}, {extra}" if vocab else extra
|
|
331
|
-
vocab = vocab[:_HOTWORDS_MAX_CHARS]
|
|
337
|
+
vocab = ", ".join(terms + self._rare_correction_targets())
|
|
332
338
|
if vocab != self._hotwords:
|
|
333
339
|
self._hotwords = vocab
|
|
334
|
-
|
|
335
|
-
|
|
340
|
+
# Set unconditionally: an emptied vocabulary must CLEAR the hotwords,
|
|
341
|
+
# not leave the previous string biasing every future decode.
|
|
342
|
+
self.transcriber.set_hotwords(vocab or None)
|
|
343
|
+
|
|
344
|
+
def _record_vocab_usage(self, text: str) -> None:
|
|
345
|
+
"""Record which vocabulary terms this utterance used (finalize path only).
|
|
346
|
+
|
|
347
|
+
This is the feedback signal that decides who gets the hotword budget next
|
|
348
|
+
time. Deliberately not on the recording-start path, and never fatal — a
|
|
349
|
+
usage counter must not be able to break a dictation.
|
|
350
|
+
"""
|
|
351
|
+
try:
|
|
352
|
+
from voiceio import vocab_stats
|
|
353
|
+
vocab_stats.update_from_text(text, self._vocab_loader.get_all())
|
|
354
|
+
except Exception: # noqa: BLE001
|
|
355
|
+
log.debug("vocab usage update failed", exc_info=True)
|
|
356
|
+
|
|
357
|
+
def _rare_correction_targets(self) -> list[str]:
|
|
358
|
+
"""Correction targets worth biasing, cached against corrections mtime.
|
|
359
|
+
|
|
360
|
+
`is_common` over every target used to be recomputed on each recording
|
|
361
|
+
start; the result only changes when corrections.json does.
|
|
362
|
+
"""
|
|
363
|
+
mtime = getattr(self._corrections, "_mtime", None)
|
|
364
|
+
if getattr(self, "_rare_targets_mtime", object()) != mtime:
|
|
365
|
+
from voiceio.wordfreq import is_common
|
|
366
|
+
self._rare_targets = [
|
|
367
|
+
t for t in self._corrections.vocabulary_terms()
|
|
368
|
+
if not is_common(t, self.cfg.model.language)
|
|
369
|
+
]
|
|
370
|
+
self._rare_targets_mtime = mtime
|
|
371
|
+
return self._rare_targets
|
|
336
372
|
|
|
337
373
|
def _do_start(self) -> None:
|
|
338
374
|
"""Transition to RECORDING."""
|
|
@@ -432,8 +468,13 @@ class VoiceIO:
|
|
|
432
468
|
if self._postcorrect is not None:
|
|
433
469
|
# Freshest context for the finalize pass: this recording's window
|
|
434
470
|
# title (captured at start) and the latest transcripts/vocabulary.
|
|
471
|
+
#
|
|
472
|
+
# The FULL vocabulary, not the hotwords string: the decoder can only
|
|
473
|
+
# be biased with ~40-60 terms, but this is an LLM with no 448-token
|
|
474
|
+
# budget. The long tail that can never fit in hotwords gets its only
|
|
475
|
+
# chance to be corrected here.
|
|
435
476
|
self._postcorrect.set_context(
|
|
436
|
-
vocabulary=self.
|
|
477
|
+
vocabulary=", ".join(self._vocab_loader.get_all()),
|
|
437
478
|
recent=self._prompt_builder.recent(3),
|
|
438
479
|
title=self._context_title,
|
|
439
480
|
)
|
|
@@ -552,6 +593,7 @@ class VoiceIO:
|
|
|
552
593
|
self._play_feedback(final_text)
|
|
553
594
|
stored = self._strip_voice_prefix(final_text)
|
|
554
595
|
self._prompt_builder.add_transcript(stored)
|
|
596
|
+
self._record_vocab_usage(stored)
|
|
555
597
|
latency = dict(session.final_latency)
|
|
556
598
|
# stop-to-commit: what the user actually waits for
|
|
557
599
|
latency["finalize_total"] = round(time.monotonic() - t_final, 3)
|
voiceio/audit.py
CHANGED
|
@@ -169,9 +169,11 @@ def _read_metrics(path: Path | None = None) -> list[dict]:
|
|
|
169
169
|
|
|
170
170
|
def _default_teacher():
|
|
171
171
|
"""Lazily build the offline teacher transcriber (distil-large-v3)."""
|
|
172
|
-
from
|
|
172
|
+
from voiceio.worker import load_model
|
|
173
173
|
|
|
174
|
-
|
|
174
|
+
# Cache-first: the module docstring's "nothing here uses the network" only
|
|
175
|
+
# holds if the load itself can't reach out and hang (see load_model).
|
|
176
|
+
model = load_model("distil-large-v3")
|
|
175
177
|
|
|
176
178
|
def _transcribe(wav_path: Path) -> str:
|
|
177
179
|
segments, _ = model.transcribe(
|
|
@@ -323,14 +325,42 @@ def run_audit(
|
|
|
323
325
|
return report
|
|
324
326
|
|
|
325
327
|
|
|
328
|
+
def _last_seen_map(terms: list[str]) -> dict[str, float]:
|
|
329
|
+
"""When each term was last dictated.
|
|
330
|
+
|
|
331
|
+
Prefers vocab_stats.json, which the runtime maintains per utterance. Falls
|
|
332
|
+
back to scanning history only when no stats exist yet (fresh install, or a
|
|
333
|
+
daemon that hasn't finalized a dictation since stats landed) — that scan is
|
|
334
|
+
O(entries x terms) over the whole history, which at 6k entries and 300+
|
|
335
|
+
terms is not worth running when the answer is already recorded.
|
|
336
|
+
"""
|
|
337
|
+
from voiceio.vocab_stats import VocabStats
|
|
338
|
+
|
|
339
|
+
stats = VocabStats()
|
|
340
|
+
stats.load()
|
|
341
|
+
if stats.as_dict():
|
|
342
|
+
return {t: float(stats.get(t).get("last_seen_ts", 0.0) or 0.0) for t in terms}
|
|
343
|
+
|
|
344
|
+
from voiceio import history
|
|
345
|
+
|
|
346
|
+
last_seen: dict[str, float] = {}
|
|
347
|
+
for e in history.read(limit=0):
|
|
348
|
+
text = (e.get("text") or "").lower()
|
|
349
|
+
ts = e.get("ts", 0.0)
|
|
350
|
+
for term in terms:
|
|
351
|
+
if term.lower() in text and ts > last_seen.get(term, 0.0):
|
|
352
|
+
last_seen[term] = ts
|
|
353
|
+
return last_seen
|
|
354
|
+
|
|
355
|
+
|
|
326
356
|
def _audit_vocabulary(cfg, teacher_texts, now, report) -> None:
|
|
327
357
|
"""Confirm vocab terms the teacher heard; age out terms unseen for 90 days.
|
|
328
358
|
|
|
329
|
-
Aging moves a term to the bottom of vocabulary.txt rather than deleting it
|
|
330
|
-
|
|
331
|
-
|
|
359
|
+
Aging moves a term to the bottom of vocabulary.txt rather than deleting it.
|
|
360
|
+
File order is now only the tie-break for terms with no usage signal —
|
|
361
|
+
vocabulary.select_terms ranks by vocab_stats first — but keeping stale terms
|
|
362
|
+
at the bottom still means they lose ties, and they rejoin on next use.
|
|
332
363
|
"""
|
|
333
|
-
from voiceio import history
|
|
334
364
|
from voiceio.vocabulary import _read_terms, resolve_vocab_path
|
|
335
365
|
|
|
336
366
|
path = resolve_vocab_path(cfg.model)
|
|
@@ -341,16 +371,7 @@ def _audit_vocabulary(cfg, teacher_texts, now, report) -> None:
|
|
|
341
371
|
return
|
|
342
372
|
|
|
343
373
|
teacher_blob = "\n".join(teacher_texts)
|
|
344
|
-
|
|
345
|
-
entries = history.read(limit=0)
|
|
346
|
-
last_seen: dict[str, float] = {}
|
|
347
|
-
for e in entries:
|
|
348
|
-
text = e.get("text", "")
|
|
349
|
-
ts = e.get("ts", 0.0)
|
|
350
|
-
low = text.lower()
|
|
351
|
-
for term in terms:
|
|
352
|
-
if term.lower() in low and ts > last_seen.get(term, 0.0):
|
|
353
|
-
last_seen[term] = ts
|
|
374
|
+
last_seen = _last_seen_map(terms)
|
|
354
375
|
|
|
355
376
|
fresh: list[str] = []
|
|
356
377
|
aged: list[str] = []
|
voiceio/cli.py
CHANGED
|
@@ -83,6 +83,8 @@ def main() -> None:
|
|
|
83
83
|
p_correct.add_argument("wrong", nargs="?", help="Misheard word/phrase")
|
|
84
84
|
p_correct.add_argument("right", nargs="?", help="Correct replacement")
|
|
85
85
|
p_correct.add_argument("--list", action="store_true", help="List all corrections")
|
|
86
|
+
p_correct.add_argument("--vocab", action="store_true",
|
|
87
|
+
help="Show vocabulary and which terms fit the decoder budget")
|
|
86
88
|
p_correct.add_argument("--remove", metavar="WORD", help="Remove a correction")
|
|
87
89
|
p_correct.add_argument("--flagged", action="store_true",
|
|
88
90
|
help="Show words flagged by 'correct that'")
|
|
@@ -108,6 +110,14 @@ def main() -> None:
|
|
|
108
110
|
p_history.add_argument("--path", action="store_true",
|
|
109
111
|
help="Print history file path")
|
|
110
112
|
|
|
113
|
+
# ── voiceio eval ───────────────────────────────────────────────────
|
|
114
|
+
p_eval = sub.add_parser(
|
|
115
|
+
"eval", help="Replay your retained audio across decoder configs and score them")
|
|
116
|
+
p_eval.add_argument("--max-audio-secs", type=float, default=900,
|
|
117
|
+
help="Audio budget to replay (default: 900 = 15min)")
|
|
118
|
+
p_eval.add_argument("--json", action="store_true",
|
|
119
|
+
help="Emit raw results as JSON")
|
|
120
|
+
|
|
111
121
|
# ── voiceio demo ──────────────────────────────────────────────────
|
|
112
122
|
sub.add_parser("demo", help="Interactive guided tour of voiceio features")
|
|
113
123
|
|
|
@@ -134,6 +144,8 @@ def main() -> None:
|
|
|
134
144
|
_cmd_correct(args)
|
|
135
145
|
elif args.command == "history":
|
|
136
146
|
_cmd_history(args)
|
|
147
|
+
elif args.command == "eval":
|
|
148
|
+
_cmd_eval(args)
|
|
137
149
|
elif args.command == "demo":
|
|
138
150
|
_cmd_demo()
|
|
139
151
|
elif args.command == "logs":
|
|
@@ -718,6 +730,138 @@ def _cmd_uninstall() -> None:
|
|
|
718
730
|
print("\nvoiceio fully removed.")
|
|
719
731
|
|
|
720
732
|
|
|
733
|
+
def _cmd_eval(args: argparse.Namespace) -> None:
|
|
734
|
+
"""Score decoder configs against a teacher model on your own audio."""
|
|
735
|
+
import json as _json
|
|
736
|
+
import sys
|
|
737
|
+
import time
|
|
738
|
+
|
|
739
|
+
from voiceio.config import load
|
|
740
|
+
from voiceio.evaluate import default_matrix, evaluate
|
|
741
|
+
from voiceio.wizard import BOLD, DIM, GREEN, RESET, YELLOW
|
|
742
|
+
|
|
743
|
+
cfg = load()
|
|
744
|
+
matrix = default_matrix()
|
|
745
|
+
|
|
746
|
+
print(f"{BOLD}Evaluating {len(matrix)} decoder configs{RESET} "
|
|
747
|
+
f"{DIM}on up to {args.max_audio_secs:.0f}s of your retained audio{RESET}")
|
|
748
|
+
print(f"{DIM}Scored against a teacher model (distil-large-v3) — a proxy for")
|
|
749
|
+
print(f"ground truth, good for ranking configs, not for absolute accuracy.{RESET}\n")
|
|
750
|
+
|
|
751
|
+
last = [""]
|
|
752
|
+
|
|
753
|
+
def progress(name, i, n):
|
|
754
|
+
if name != last[0]:
|
|
755
|
+
last[0] = name
|
|
756
|
+
sys.stderr.write(f"\n {name}: ")
|
|
757
|
+
sys.stderr.write(".")
|
|
758
|
+
sys.stderr.flush()
|
|
759
|
+
|
|
760
|
+
t0 = time.time()
|
|
761
|
+
scores = evaluate(cfg, matrix, max_audio_secs=args.max_audio_secs,
|
|
762
|
+
on_progress=progress)
|
|
763
|
+
sys.stderr.write("\n\n")
|
|
764
|
+
|
|
765
|
+
if not scores:
|
|
766
|
+
print("No retained audio to evaluate. Dictate a few notes first "
|
|
767
|
+
"(audio is kept under [data] retain_audio).")
|
|
768
|
+
return
|
|
769
|
+
|
|
770
|
+
if args.json:
|
|
771
|
+
print(_json.dumps([{
|
|
772
|
+
"config": s.config.name, **s.config.as_dict(),
|
|
773
|
+
"clips": s.clips, "audio_secs": round(s.audio_secs, 1),
|
|
774
|
+
"wer": s.wer, "hallucinations": s.hallucinations,
|
|
775
|
+
"repetitions": s.repetitions, "empties": s.empties,
|
|
776
|
+
"crashes": s.crashes, "decode_secs": round(s.decode_secs, 1),
|
|
777
|
+
} for s in scores], indent=2))
|
|
778
|
+
return
|
|
779
|
+
|
|
780
|
+
best = scores[0]
|
|
781
|
+
shipped = next((s for s in scores if s.config.name == "shipped"), None)
|
|
782
|
+
|
|
783
|
+
print(f" {'config':<24} {'WER':>7} {'halluc':>7} {'repeat':>7} "
|
|
784
|
+
f"{'empty':>6} {'crash':>6}")
|
|
785
|
+
print(f" {'-'*24} {'-'*7} {'-'*7} {'-'*7} {'-'*6} {'-'*6}")
|
|
786
|
+
for s in scores:
|
|
787
|
+
mark = f" {GREEN}<- best{RESET}" if s is best else ""
|
|
788
|
+
if s is shipped and s is not best:
|
|
789
|
+
mark = f" {DIM}<- shipped{RESET}"
|
|
790
|
+
elif s is shipped:
|
|
791
|
+
mark = f" {GREEN}<- best (shipped){RESET}"
|
|
792
|
+
print(f" {s.config.name:<24} {s.wer:>7.3f} {s.hallucinations:>7} "
|
|
793
|
+
f"{s.repetitions:>7} {s.empties:>6} {s.crashes:>6}{mark}")
|
|
794
|
+
|
|
795
|
+
print(f"\n {DIM}{best.clips} clips, {best.audio_secs:.0f}s audio, "
|
|
796
|
+
f"{time.time()-t0:.0f}s total{RESET}")
|
|
797
|
+
|
|
798
|
+
if shipped and best is not shipped:
|
|
799
|
+
delta = (shipped.wer - best.wer) / shipped.wer * 100 if shipped.wer else 0
|
|
800
|
+
print(f"\n {YELLOW}'{best.config.name}' beats the shipped config by "
|
|
801
|
+
f"{delta:.0f}% WER.{RESET}")
|
|
802
|
+
# A WER win means nothing if it comes from the failure mode the knob
|
|
803
|
+
# exists to prevent, so make that trade explicit rather than implied.
|
|
804
|
+
if best.hallucinations > shipped.hallucinations:
|
|
805
|
+
print(f" {YELLOW}But it hallucinates more "
|
|
806
|
+
f"({best.hallucinations} vs {shipped.hallucinations}) — "
|
|
807
|
+
f"that's what vad_filter is for.{RESET}")
|
|
808
|
+
if best.repetitions > shipped.repetitions:
|
|
809
|
+
print(f" {YELLOW}But it loops more "
|
|
810
|
+
f"({best.repetitions} vs {shipped.repetitions}) — "
|
|
811
|
+
f"that's what condition_on_previous_text=False is for.{RESET}")
|
|
812
|
+
print(f" {DIM}The teacher is a proxy, not truth. Spot-check before "
|
|
813
|
+
f"changing a default.{RESET}")
|
|
814
|
+
elif shipped:
|
|
815
|
+
print(f"\n {GREEN}The shipped config wins. No change indicated.{RESET}")
|
|
816
|
+
|
|
817
|
+
|
|
818
|
+
def _show_vocab() -> None:
|
|
819
|
+
"""Show the vocabulary split by what actually reaches the decoder.
|
|
820
|
+
|
|
821
|
+
Whisper can only be biased with ~35 terms (a hard 223-token hotwords cap,
|
|
822
|
+
and proper nouns cost 5-7 tokens each). Everything past that is not wasted —
|
|
823
|
+
it still reaches the post-correction LLM — but it does NOT bias decoding,
|
|
824
|
+
and until now there was no way to see which side of the line a term fell on.
|
|
825
|
+
"""
|
|
826
|
+
from voiceio.app import _HOTWORDS_TOKEN_BUDGET
|
|
827
|
+
from voiceio.config import load
|
|
828
|
+
from voiceio.tokens import count_tokens
|
|
829
|
+
from voiceio.vocab_stats import VocabStats
|
|
830
|
+
from voiceio.vocabulary import load_terms, select_terms
|
|
831
|
+
from voiceio.wizard import BOLD, DIM, GREEN, RESET
|
|
832
|
+
|
|
833
|
+
cfg = load()
|
|
834
|
+
terms = load_terms(cfg.model)
|
|
835
|
+
if not terms:
|
|
836
|
+
print("No vocabulary configured.")
|
|
837
|
+
print("\nTerms are learned automatically: voiceio correct --auto")
|
|
838
|
+
return
|
|
839
|
+
|
|
840
|
+
stats = VocabStats()
|
|
841
|
+
stats.load()
|
|
842
|
+
selected = select_terms(terms, token_budget=_HOTWORDS_TOKEN_BUDGET,
|
|
843
|
+
model_name=cfg.model.name, stats=stats)
|
|
844
|
+
used = count_tokens(", ".join(selected), cfg.model.name)
|
|
845
|
+
sel = set(selected)
|
|
846
|
+
|
|
847
|
+
print(f"{BOLD}Biasing the decoder{RESET} "
|
|
848
|
+
f"{DIM}({used}/{_HOTWORDS_TOKEN_BUDGET} tokens){RESET}")
|
|
849
|
+
for t in selected:
|
|
850
|
+
rec = stats.get(t)
|
|
851
|
+
hits = rec.get("hits", 0)
|
|
852
|
+
print(f" {GREEN}●{RESET} {t}" + (f" {DIM}({hits} uses){RESET}" if hits else ""))
|
|
853
|
+
|
|
854
|
+
tail = [t for t in terms if t not in sel]
|
|
855
|
+
if tail:
|
|
856
|
+
print(f"\n{BOLD}Post-correction only{RESET} "
|
|
857
|
+
f"{DIM}({len(tail)} terms — past the decoder's token budget,{RESET}")
|
|
858
|
+
print(f" {DIM}still used by the correction pass){RESET}")
|
|
859
|
+
shown = ", ".join(tail[:12])
|
|
860
|
+
print(f" {DIM}{shown}{'...' if len(tail) > 12 else ''}{RESET}")
|
|
861
|
+
|
|
862
|
+
print(f"\n{len(terms)} term(s): {len(selected)} biasing, {len(tail)} tail")
|
|
863
|
+
|
|
864
|
+
|
|
721
865
|
def _cmd_correct(args: argparse.Namespace) -> None:
|
|
722
866
|
"""Manage the corrections dictionary."""
|
|
723
867
|
from voiceio.corrections import CorrectionDict
|
|
@@ -734,6 +878,10 @@ def _cmd_correct(args: argparse.Namespace) -> None:
|
|
|
734
878
|
print(f"\n{len(corrections)} correction(s)")
|
|
735
879
|
return
|
|
736
880
|
|
|
881
|
+
if args.vocab:
|
|
882
|
+
_show_vocab()
|
|
883
|
+
return
|
|
884
|
+
|
|
737
885
|
if args.remove:
|
|
738
886
|
if cd.remove(args.remove):
|
|
739
887
|
print(f"Removed: {args.remove}")
|
|
@@ -936,12 +1084,20 @@ def _cmd_correct_auto(cd, *, full: bool = False, batch: bool = False) -> None:
|
|
|
936
1084
|
# Build lookup for O(1) access to suspicious word metadata
|
|
937
1085
|
sw_by_word = {sw.word: sw for sw in suspicious}
|
|
938
1086
|
|
|
1087
|
+
# Correction mining is retired by default ([autocorrect] mine_corrections).
|
|
1088
|
+
# The vocabulary bucket below is unaffected: those terms bias the decoder
|
|
1089
|
+
# and measurably help, whereas mined find/replace rules fired 4 times in 387
|
|
1090
|
+
# while postcorrect applied 288 edits over the same span. Skipping the
|
|
1091
|
+
# correction buckets also skips adjudication — the expensive part, at
|
|
1092
|
+
# votes x ceil(N/25) LLM calls per run.
|
|
1093
|
+
mine_corrections = cfg.autocorrect.mine_corrections
|
|
1094
|
+
|
|
939
1095
|
# ── Bucket 1: Auto-fix (high confidence, safety-gated) ──────────────
|
|
940
1096
|
# Pairs failing the gate (real word being "corrected", or target that is
|
|
941
1097
|
# itself junk) are downgraded to manual review rather than persisted.
|
|
942
1098
|
auto_fixed = 0
|
|
943
1099
|
downgraded: list[dict] = []
|
|
944
|
-
if result.auto_fix:
|
|
1100
|
+
if result.auto_fix and mine_corrections:
|
|
945
1101
|
gated = []
|
|
946
1102
|
for fix in result.auto_fix:
|
|
947
1103
|
reason = gate_correction(
|
|
@@ -975,10 +1131,13 @@ def _cmd_correct_auto(cd, *, full: bool = False, batch: bool = False) -> None:
|
|
|
975
1131
|
)
|
|
976
1132
|
|
|
977
1133
|
# ── Bucket 2: Ask user (ambiguous) ──────────────────────────────────
|
|
978
|
-
|
|
1134
|
+
# Empty when correction mining is retired — nothing to adjudicate or review.
|
|
1135
|
+
to_review = (downgraded + list(result.ask_user)) if mine_corrections else []
|
|
979
1136
|
|
|
980
1137
|
# If no LLM was used, put all suspicious words into manual review
|
|
981
|
-
if not
|
|
1138
|
+
if not mine_corrections:
|
|
1139
|
+
pass # correction mining retired — nothing to review or adjudicate
|
|
1140
|
+
elif not has_api and not has_ollama:
|
|
982
1141
|
for sw in suspicious:
|
|
983
1142
|
to_review.append({
|
|
984
1143
|
"wrong": sw.word,
|
|
@@ -1034,8 +1193,13 @@ def _cmd_correct_auto(cd, *, full: bool = False, batch: bool = False) -> None:
|
|
|
1034
1193
|
print(f" {DIM}{len(capped)} lower-frequency item(s) deferred "
|
|
1035
1194
|
f"to next run{RESET}")
|
|
1036
1195
|
|
|
1037
|
-
|
|
1038
|
-
|
|
1196
|
+
if to_review:
|
|
1197
|
+
adj = adjudicate(cfg, to_review, sw_by_word,
|
|
1198
|
+
vocabulary=vocab_words, language=language)
|
|
1199
|
+
else:
|
|
1200
|
+
# Nothing to weigh — don't spend votes x ceil(N/25) LLM calls proving it.
|
|
1201
|
+
from voiceio.autocorrect import AdjudicationResult
|
|
1202
|
+
adj = AdjudicationResult()
|
|
1039
1203
|
|
|
1040
1204
|
# Unanimous corrections (already gate-passed inside adjudicate).
|
|
1041
1205
|
adj_applied = 0
|
voiceio/config.py
CHANGED
|
@@ -138,6 +138,17 @@ class AutocorrectConfig:
|
|
|
138
138
|
# Languages you also dictate in: mined corrections never rewrite words
|
|
139
139
|
# that are real in these (e.g. ["es"] protects Spanish "harina").
|
|
140
140
|
protect_languages: list[str] = field(default_factory=list)
|
|
141
|
+
# Mine find/replace CORRECTION rules from history. Off by default: measured
|
|
142
|
+
# over months of real use, mining produced 387 rules of which 4 ever fired
|
|
143
|
+
# (1%), while the runtime postcorrect pass applied 288 edits over the same
|
|
144
|
+
# period. Rules are exact-string matches, so they only fire when Whisper
|
|
145
|
+
# repeats a misrecognition verbatim — and the errors worth fixing are
|
|
146
|
+
# common words ("nuance" for "neurons") that the safety gate must refuse
|
|
147
|
+
# anyway. Meanwhile a bad rule is a silent regression: one run learned rules
|
|
148
|
+
# destroying real Spanish words. Poor odds, real downside, and postcorrect
|
|
149
|
+
# already does this contextually. Vocabulary mining is unaffected and stays
|
|
150
|
+
# on — those terms bias the decoder and demonstrably help.
|
|
151
|
+
mine_corrections: bool = False
|
|
141
152
|
|
|
142
153
|
|
|
143
154
|
@dataclass
|