vocalize-cli 0.10.1__tar.gz → 0.10.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/.gitignore +1 -0
  2. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/CHANGELOG.md +26 -0
  3. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/PKG-INFO +5 -1
  4. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/README.md +4 -0
  5. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/dictation.md +14 -1
  6. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/roadmap.md +2 -1
  7. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_cli.py +5 -2
  8. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_config.py +24 -0
  9. vocalize_cli-0.10.2/tests/test_cue_assets.py +28 -0
  10. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_dictate.py +85 -0
  11. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/__init__.py +1 -1
  12. vocalize_cli-0.10.2/vocalize/assets/cues/README.md +11 -0
  13. vocalize_cli-0.10.2/vocalize/assets/cues/ready.wav +0 -0
  14. vocalize_cli-0.10.2/vocalize/assets/cues/start.wav +0 -0
  15. vocalize_cli-0.10.2/vocalize/assets/cues/stopped.wav +0 -0
  16. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/cli.py +1 -0
  17. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/config.py +12 -0
  18. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/dictate.py +39 -4
  19. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/.env.example +0 -0
  20. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/.github/workflows/ci.yml +0 -0
  21. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/LICENSE +0 -0
  22. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/installation.md +0 -0
  23. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/next-features-analysis.md +0 -0
  24. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/choreography.md +0 -0
  25. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/decisions.md +0 -0
  26. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/design.md +0 -0
  27. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/plan.md +0 -0
  28. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/review-0.10.0.md +0 -0
  29. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-1-status/project-plan.md +0 -0
  30. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-1-status/report.md +0 -0
  31. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-1-status/validate-exit.sh +0 -0
  32. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-10-release-0-11-0/project-plan.md +0 -0
  33. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-10-release-0-11-0/validate-exit.sh +0 -0
  34. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-2-stt-runtime/project-plan.md +0 -0
  35. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-2-stt-runtime/report.md +0 -0
  36. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-2-stt-runtime/validate-exit.sh +0 -0
  37. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-3-recorder/project-plan.md +0 -0
  38. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-3-recorder/report.md +0 -0
  39. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-3-recorder/validate-exit.sh +0 -0
  40. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-4-dictation/project-plan.md +0 -0
  41. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-4-dictation/report.md +0 -0
  42. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-4-dictation/validate-exit.sh +0 -0
  43. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-5-resume/project-plan.md +0 -0
  44. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-5-resume/report.md +0 -0
  45. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-5-resume/validate-exit.sh +0 -0
  46. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-6-release-0-10-0/project-plan.md +0 -0
  47. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-6-release-0-10-0/report.md +0 -0
  48. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-6-release-0-10-0/validate-exit.sh +0 -0
  49. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-7-portal-read/project-plan.md +0 -0
  50. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-7-portal-read/validate-exit.sh +0 -0
  51. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-8-portal-write/project-plan.md +0 -0
  52. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-8-portal-write/validate-exit.sh +0 -0
  53. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-9-portal-page/project-plan.md +0 -0
  54. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-9-portal-page/validate-exit.sh +0 -0
  55. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/spike-2026-09-01.md +0 -0
  56. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/split-assessment.md +0 -0
  57. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/verification.md +0 -0
  58. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/provider-credentials.md +0 -0
  59. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/research/2026-09-01-config-portal-design.md +0 -0
  60. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/research/2026-09-01-dictation-design.md +0 -0
  61. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/docs/research/2026-09-01-voicebox-findings.md +0 -0
  62. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/claude_stop_hook.py +0 -0
  63. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/install_hook.py +0 -0
  64. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/install_quick_action.py +0 -0
  65. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/quick_actions/Dictate with Vocalize.workflow/Contents/Info.plist +0 -0
  66. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/quick_actions/Dictate with Vocalize.workflow/Contents/Resources/document.wflow +0 -0
  67. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/quick_actions/Speak Latest Plan.workflow/Contents/Info.plist +0 -0
  68. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/quick_actions/Speak Latest Plan.workflow/Contents/Resources/document.wflow +0 -0
  69. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/quick_actions/Speak with Vocalize.workflow/Contents/Info.plist +0 -0
  70. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/quick_actions/Speak with Vocalize.workflow/Contents/Resources/document.wflow +0 -0
  71. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/quick_actions/Stop Vocalize.workflow/Contents/Info.plist +0 -0
  72. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/quick_actions/Stop Vocalize.workflow/Contents/Resources/document.wflow +0 -0
  73. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/speak_options.py +0 -0
  74. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/hooks/speak_url_gate.py +0 -0
  75. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/pyproject.toml +0 -0
  76. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/conftest.py +0 -0
  77. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_audio.py +0 -0
  78. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_auth.py +0 -0
  79. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_cache.py +0 -0
  80. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_chain.py +0 -0
  81. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_claude_stop_hook.py +0 -0
  82. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_clipboard.py +0 -0
  83. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_elevenlabs_provider.py +0 -0
  84. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_exceptions.py +0 -0
  85. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_google_provider.py +0 -0
  86. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_http.py +0 -0
  87. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_install_hook.py +0 -0
  88. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_install_quick_action.py +0 -0
  89. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_kokoro_manifest.py +0 -0
  90. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_kokoro_provider.py +0 -0
  91. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_kokoro_worker.py +0 -0
  92. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_ledger.py +0 -0
  93. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_listen_check.py +0 -0
  94. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_local_install.py +0 -0
  95. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_openai_provider.py +0 -0
  96. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_polly_provider.py +0 -0
  97. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_preprocess.py +0 -0
  98. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_providers_registry.py +0 -0
  99. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_readiness.py +0 -0
  100. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_recorder_build.py +0 -0
  101. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_say_provider.py +0 -0
  102. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_speak_options.py +0 -0
  103. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_speak_url_gate.py +0 -0
  104. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_tts.py +0 -0
  105. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_uv_path.py +0 -0
  106. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_whisper_manifest.py +0 -0
  107. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_whisper_worker.py +0 -0
  108. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/tests/test_wizard.py +0 -0
  109. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/__main__.py +0 -0
  110. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/audio.py +0 -0
  111. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/auth.py +0 -0
  112. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/cache.py +0 -0
  113. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/chain.py +0 -0
  114. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/clipboard.py +0 -0
  115. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/exceptions.py +0 -0
  116. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/interrupted.py +0 -0
  117. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/ledger.py +0 -0
  118. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/local/__init__.py +0 -0
  119. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/local/install.py +0 -0
  120. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/local/kokoro_manifest.py +0 -0
  121. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/local/kokoro_worker.py +0 -0
  122. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/local/whisper_manifest.py +0 -0
  123. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/local/whisper_worker.py +0 -0
  124. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/preprocess.py +0 -0
  125. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/providers/__init__.py +0 -0
  126. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/providers/_http.py +0 -0
  127. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/providers/elevenlabs.py +0 -0
  128. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/providers/google.py +0 -0
  129. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/providers/kokoro.py +0 -0
  130. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/providers/openai.py +0 -0
  131. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/providers/polly.py +0 -0
  132. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/providers/say.py +0 -0
  133. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/readiness.py +0 -0
  134. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/recorder/Info.plist.in +0 -0
  135. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/recorder/Recorder.entitlements +0 -0
  136. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/recorder/VocalizeRecorder.swift +0 -0
  137. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/tts.py +0 -0
  138. {vocalize_cli-0.10.1 → vocalize_cli-0.10.2}/vocalize/wizard.py +0 -0
@@ -11,5 +11,6 @@ dist/
11
11
  venv/
12
12
  *.mp3
13
13
  *.wav
14
+ !vocalize/assets/cues/*.wav
14
15
  .claude/
15
16
  uv.lock
@@ -3,6 +3,32 @@
3
3
  All notable changes to this project are documented here. Format follows
4
4
  [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
5
5
 
6
+ ## 0.10.2 - 2026-09-02
7
+
8
+ ### Added
9
+
10
+ - **`[stt] cues`** picks what a dictation's feedback sounds — say instead
11
+ of, or alongside, the Tink/Pop/Glass system sounds. `"sounds"` (default)
12
+ is unchanged; `"words"` speaks "Start.", "Stopped.", "Ready." in their
13
+ place; `"both"` speaks the word and then plays the sound. The word files
14
+ ship in `vocalize/assets/cues/`, generated with the local Kokoro voice.
15
+ A spoken "Start." plays *before* the recorder launches — played once the
16
+ microphone was open it would be recorded and transcribed along with the
17
+ dictation. In `"both"` mode the Tink still plays *after* the microphone
18
+ opens, so the two cues keep distinct meanings: the word is "get ready",
19
+ the sound is "talk now". The plain Tink is unaffected.
20
+
21
+ ### Known issue
22
+
23
+ - In `"words"` mode there is no cue for the moment the microphone actually
24
+ opens, which is a second or so after "Start." finishes (LaunchServices
25
+ start-up plus the input device switching on). People start talking too
26
+ soon and lose their first word. The fix — open and warm the microphone
27
+ first, play the cue, and only then capture — is tracked in
28
+ [#2](https://github.com/matthager12-collab/vocalize/issues/2). Until
29
+ then: in `"words"` mode, wait a beat after "Start."; in `"both"` mode,
30
+ talk after the Tink.
31
+
6
32
  ## 0.10.1 - 2026-09-02
7
33
 
8
34
  Three fixes found in the first owner-present run of 0.10.0's dictation.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: vocalize-cli
3
- Version: 0.10.1
3
+ Version: 0.10.2
4
4
  Summary: A CLI that turns text, markdown, or piped stdin into speech via the ElevenLabs API, with markdown-table-aware preprocessing.
5
5
  Project-URL: Homepage, https://github.com/matthager12-collab/vocalize
6
6
  Project-URL: Repository, https://github.com/matthager12-collab/vocalize
@@ -485,6 +485,8 @@ terminal, if you'd rather trigger it that way.
485
485
  - **A third press while transcribing is refused**: a Pop, and "Still
486
486
  transcribing the last dictation." Wait for the clipboard notification, or
487
487
  `--cancel`, before dictating again.
488
+ - Can't tell Tink from Pop from Glass yet? Set `[stt] cues = "words"` and
489
+ vocalize says "Start.", "Stopped.", "Ready." instead.
488
490
 
489
491
  ### `vocalize listen`
490
492
 
@@ -519,6 +521,7 @@ input_device = "" # "" = system default; else an exact name from --list-dev
519
521
  cleanup = false # send the transcript (never audio) to Claude first
520
522
  max_seconds = 120 # 1-600; the recorder self-stops here, dictate backstops it
521
523
  sounds = true # the Tink/Pop/Glass feedback sounds
524
+ cues = "sounds" # "sounds" | "words" | "both" — speak "Start."/"Stopped."/"Ready." instead
522
525
  ```
523
526
 
524
527
  | Key | Allowed values | Default |
@@ -530,6 +533,7 @@ sounds = true # the Tink/Pop/Glass feedback sounds
530
533
  | `paste` | reserved — not implemented in 0.10.0 | `false` |
531
534
  | `max_seconds` | integer, 1–600 | `120` |
532
535
  | `sounds` | `true` / `false` | `true` |
536
+ | `cues` | `sounds`, `words`, `both` | `sounds` |
533
537
 
534
538
  An unknown key warns on stderr; a bad value is a `ConfigError` naming it —
535
539
  every one of these becomes a subprocess argument eventually, so nothing
@@ -449,6 +449,8 @@ terminal, if you'd rather trigger it that way.
449
449
  - **A third press while transcribing is refused**: a Pop, and "Still
450
450
  transcribing the last dictation." Wait for the clipboard notification, or
451
451
  `--cancel`, before dictating again.
452
+ - Can't tell Tink from Pop from Glass yet? Set `[stt] cues = "words"` and
453
+ vocalize says "Start.", "Stopped.", "Ready." instead.
452
454
 
453
455
  ### `vocalize listen`
454
456
 
@@ -483,6 +485,7 @@ input_device = "" # "" = system default; else an exact name from --list-dev
483
485
  cleanup = false # send the transcript (never audio) to Claude first
484
486
  max_seconds = 120 # 1-600; the recorder self-stops here, dictate backstops it
485
487
  sounds = true # the Tink/Pop/Glass feedback sounds
488
+ cues = "sounds" # "sounds" | "words" | "both" — speak "Start."/"Stopped."/"Ready." instead
486
489
  ```
487
490
 
488
491
  | Key | Allowed values | Default |
@@ -494,6 +497,7 @@ sounds = true # the Tink/Pop/Glass feedback sounds
494
497
  | `paste` | reserved — not implemented in 0.10.0 | `false` |
495
498
  | `max_seconds` | integer, 1–600 | `120` |
496
499
  | `sounds` | `true` / `false` | `true` |
500
+ | `cues` | `sounds`, `words`, `both` | `sounds` |
497
501
 
498
502
  An unknown key warns on stderr; a bad value is a `ConfigError` naming it —
499
503
  every one of these becomes a subprocess argument eventually, so nothing
@@ -147,6 +147,16 @@ never runs two transcriptions over the same recording, and it never
147
147
  silently drops one either. Wait for the clipboard notification, or use
148
148
  `--cancel`.
149
149
 
150
+ New to the sounds and can't tell Tink from Pop from Glass? Set
151
+ `[stt] cues = "words"` and vocalize says "Start.", "Stopped." and "Ready."
152
+ instead, or `"both"` to hear the word and then its sound — see the `[stt]`
153
+ table below. Timing matters: "Start." is spoken *before* the microphone
154
+ opens (so it is never in your recording), and the microphone is open
155
+ about a second later — the Tink marks that moment. So in `"both"` mode
156
+ talk after the Tink; in `"words"` mode give it a beat after "Start.".
157
+ Closing that gap is
158
+ [#2](https://github.com/matthager12-collab/vocalize/issues/2).
159
+
150
160
  ## `vocalize listen`
151
161
 
152
162
  The terminal-facing primitive behind the hotkey:
@@ -216,6 +226,7 @@ cleanup = false
216
226
  paste = false
217
227
  max_seconds = 120
218
228
  sounds = true
229
+ cues = "sounds" # "sounds" | "words" | "both"
219
230
  ```
220
231
 
221
232
  | Key | Type / allowlist | Default | Notes |
@@ -226,7 +237,8 @@ sounds = true
226
237
  | `cleanup` | `true` / `false` | `false` | see `--cleanup` above |
227
238
  | `paste` | reserved | `false` | not implemented in 0.10.0 — setting it does nothing |
228
239
  | `max_seconds` | integer, 1–600 | `120` | the recorder self-stops here; `dictate` backstops it a few seconds later in case the recorder doesn't |
229
- | `sounds` | `true` / `false` | `true` | the Tink/Pop/Glass feedback; `false` silences all three |
240
+ | `sounds` | `true` / `false` | `true` | the Tink/Pop/Glass feedback; `false` silences all three (words included) |
241
+ | `cues` | `sounds`, `words`, `both` | `sounds` | `"words"` speaks "Start.", "Stopped.", "Ready." instead of the system sounds; `"both"` speaks the word and then plays the sound — for the start cue, the word before the microphone opens and the Tink once it has. Has no effect while `sounds = false`. |
230
242
 
231
243
  Every value here eventually becomes a subprocess argument — the recorder's
232
244
  `--device`, or the whisper worker's `--model`/`--language` — so each one is
@@ -243,6 +255,7 @@ stt.model=small.en
243
255
  stt.language=en
244
256
  stt.cleanup=false
245
257
  stt.max_seconds=120
258
+ stt.cues=sounds
246
259
  ```
247
260
 
248
261
  ### The input-device gotcha
@@ -6,7 +6,8 @@ Future work, with the research behind each item. Status legend: **planned** = an
6
6
  |---|---|---|---|
7
7
  | Hotkey-triggered local dictation (voice → text via a local Whisper model, Quick Action toggle, clipboard output, optional cleanup) | **shipped** in 0.10.0, usable from 0.10.1 | [#1](https://github.com/matthager12-collab/vocalize/issues/1) · [plan](plans/2026-09-next-features/plan.md) | [analysis](next-features-analysis.md) · [full design](research/2026-09-01-dictation-design.md) · [voicebox findings](research/2026-09-01-voicebox-findings.md) |
8
8
  | `vocalize status` one-screen readiness check | **shipped** in 0.10.0 | [#1](https://github.com/matthager12-collab/vocalize/issues/1) | [analysis](next-features-analysis.md) |
9
- | Spoken cues for dictation (`[stt] cues = "words"`: "start" / "stopped" / "ready" instead of, or as well as, the sounds) | building | — | owner request, 2026-09-02 |
9
+ | Spoken cues for dictation (`[stt] cues = "words"`: "start" / "stopped" / "ready" instead of, or as well as, the sounds) | **shipped** in 0.10.2 | — | owner request, 2026-09-02 |
10
+ | Cue timing: the "talk now" cue must follow the *open* microphone and never be recorded (recorder warms the input, reports ready, waits for a `go` marker; `dictate` plays the cue, then touches `go`) | planned | [#2](https://github.com/matthager12-collab/vocalize/issues/2) | first live use of 0.10.2: people talk after "Start." and lose the first word |
10
11
  | Config portal (`vocalize portal`, stdlib local web page on top of the same readiness rows) | planned — runs 7–10 of the [plan](plans/2026-09-next-features/plan.md) | [#1](https://github.com/matthager12-collab/vocalize/issues/1) | [analysis](next-features-analysis.md) · [full design](research/2026-09-01-config-portal-design.md) |
11
12
 
12
13
  ## Decisions (2026-09-01, closed 2026-09-02)
@@ -624,8 +624,10 @@ def test_settings_prints_the_stt_lines(monkeypatch, tmp_path):
624
624
  _isolate_overflow_env(monkeypatch, tmp_path)
625
625
  cfg = tmp_path / "vocalize" / "config.toml"
626
626
  cfg.parent.mkdir(parents=True, exist_ok=True)
627
- cfg.write_text('[stt]\nmodel = "base.en"\ncleanup = true\nmax_seconds = 30\n',
628
- encoding="utf-8")
627
+ cfg.write_text(
628
+ '[stt]\nmodel = "base.en"\ncleanup = true\nmax_seconds = 30\ncues = "sounds"\n',
629
+ encoding="utf-8",
630
+ )
629
631
 
630
632
  result = CliRunner().invoke(main, ["settings"])
631
633
 
@@ -634,6 +636,7 @@ def test_settings_prints_the_stt_lines(monkeypatch, tmp_path):
634
636
  assert "stt.language=en" in result.output
635
637
  assert "stt.cleanup=true" in result.output
636
638
  assert "stt.max_seconds=30" in result.output
639
+ assert "stt.cues=sounds" in result.output
637
640
 
638
641
 
639
642
  def test_settings_prints_defaults_when_nothing_is_configured(monkeypatch, tmp_path):
@@ -697,6 +697,30 @@ def test_a_non_boolean_stt_flag_is_refused(monkeypatch, tmp_path, key):
697
697
  assert f"stt.{key}" in str(excinfo.value)
698
698
 
699
699
 
700
+ @pytest.mark.parametrize("value", ['"chime"', "1"])
701
+ def test_an_invalid_stt_cues_value_is_refused(monkeypatch, tmp_path, value):
702
+ with pytest.raises(ConfigError) as excinfo:
703
+ _load_stt(monkeypatch, tmp_path, f"[stt]\ncues = {value}\n")
704
+
705
+ message = str(excinfo.value)
706
+ assert "stt.cues" in message
707
+ assert "sounds" in message and "words" in message and "both" in message
708
+
709
+
710
+ @pytest.mark.parametrize("value", ["sounds", "words", "both"])
711
+ def test_each_stt_cues_mode_is_accepted(monkeypatch, tmp_path, value):
712
+ from vocalize.config import resolve_stt
713
+
714
+ data = _load_stt(monkeypatch, tmp_path, f'[stt]\ncues = "{value}"\n')
715
+ assert resolve_stt(data)["cues"] == value
716
+
717
+
718
+ def test_stt_cues_defaults_to_sounds(monkeypatch, tmp_path):
719
+ from vocalize.config import resolve_stt
720
+
721
+ assert resolve_stt(_load_stt(monkeypatch, tmp_path, ""))["cues"] == "sounds"
722
+
723
+
700
724
  def test_an_stt_value_that_is_not_a_table_is_refused(monkeypatch, tmp_path):
701
725
  with pytest.raises(ConfigError) as excinfo:
702
726
  _load_stt(monkeypatch, tmp_path, 'stt = "small.en"\n')
@@ -0,0 +1,28 @@
1
+ """The shipped `[stt] cues` word files: present, small, and well-formed.
2
+
3
+ Not a test of speech quality — just the shape that `dictate._play` and the
4
+ packaged wheel both depend on: a real mono 16-bit WAV, short and light
5
+ enough to ship and to speak without lagging behind the sound it replaces.
6
+ """
7
+
8
+ import wave
9
+ from pathlib import Path
10
+
11
+ import pytest
12
+
13
+ from vocalize.dictate import _CUE_WORDS
14
+
15
+ MAX_SECONDS = 1.5
16
+ MAX_BYTES = 80_000
17
+
18
+
19
+ @pytest.mark.parametrize("path", list(_CUE_WORDS.values()), ids=lambda p: p.name)
20
+ def test_a_cue_word_file_is_a_small_mono_16_bit_wav(path: Path):
21
+ assert path.is_file(), f"missing cue asset: {path}"
22
+ assert path.stat().st_size <= MAX_BYTES
23
+
24
+ with wave.open(str(path), "rb") as reader:
25
+ assert reader.getnchannels() == 1
26
+ assert reader.getsampwidth() == 2
27
+ duration = reader.getnframes() / reader.getframerate()
28
+ assert duration <= MAX_SECONDS
@@ -399,6 +399,91 @@ def test_the_second_press_transcribes_and_copies_to_the_clipboard(
399
399
  assert any(dictate._NOTIFY_COPIED in line for line in harness.notifications())
400
400
 
401
401
 
402
+ # --- spoken cues (`[stt] cues`) ----------------------------------------
403
+
404
+
405
+ def test_words_mode_speaks_start_stop_and_done(
406
+ recorder, transcriber, harness, monkeypatch
407
+ ):
408
+ recorder()
409
+ transcriber()
410
+
411
+ original_launch = dictate._launch_recorder
412
+
413
+ def launch_after_start_cue(workdir, settings):
414
+ # The spoken "Start." must finish before the microphone opens, or
415
+ # it would be recorded and transcribed along with the dictation.
416
+ assert "start.wav" in harness.played
417
+ return original_launch(workdir, settings)
418
+
419
+ monkeypatch.setattr(dictate, "_launch_recorder", launch_after_start_cue)
420
+
421
+ assert start(cues="words") == 0
422
+ assert press_again(monkeypatch, cues="words") == 0
423
+
424
+ assert harness.played == ["start.wav", "stopped.wav", "ready.wav"]
425
+
426
+
427
+ def test_both_mode_speaks_the_word_then_plays_the_sound(
428
+ recorder, transcriber, harness, monkeypatch
429
+ ):
430
+ recorder()
431
+ transcriber()
432
+
433
+ original_launch = dictate._launch_recorder
434
+
435
+ def launch_between_word_and_sound(workdir, settings):
436
+ # "Start." is "get ready" and plays before the microphone opens;
437
+ # the Tink is "talk now" and must wait until the recorder reports
438
+ # it is recording — otherwise the sound promises a microphone that
439
+ # is still a second away.
440
+ assert harness.played == ["start.wav"]
441
+ return original_launch(workdir, settings)
442
+
443
+ monkeypatch.setattr(dictate, "_launch_recorder", launch_between_word_and_sound)
444
+
445
+ assert start(cues="both") == 0
446
+ assert press_again(monkeypatch, cues="both") == 0
447
+
448
+ assert harness.played == [
449
+ "start.wav", "Tink.aiff",
450
+ "stopped.wav", "Pop.aiff",
451
+ "ready.wav", "Glass.aiff",
452
+ ]
453
+
454
+
455
+ def test_sounds_false_silences_words_too(recorder, transcriber, harness, monkeypatch):
456
+ recorder()
457
+ transcriber()
458
+
459
+ assert start(cues="words", sounds=False) == 0
460
+ assert press_again(monkeypatch, cues="words", sounds=False) == 0
461
+
462
+ assert harness.played == []
463
+
464
+
465
+ def test_a_missing_cue_word_file_falls_back_to_the_sound(
466
+ recorder, transcriber, harness, monkeypatch, tmp_path
467
+ ):
468
+ empty = tmp_path / "no-cues"
469
+ empty.mkdir()
470
+ monkeypatch.setattr(
471
+ dictate, "_CUE_WORDS",
472
+ {
473
+ dictate._SOUND_START: empty / "start.wav",
474
+ dictate._SOUND_STOP: empty / "stopped.wav",
475
+ dictate._SOUND_DONE: empty / "ready.wav",
476
+ },
477
+ )
478
+ recorder()
479
+ transcriber()
480
+
481
+ assert start(cues="words") == 0
482
+ assert press_again(monkeypatch, cues="words") == 0
483
+
484
+ assert harness.played == ["Tink.aiff", "Pop.aiff", "Glass.aiff"]
485
+
486
+
402
487
  def test_the_working_directory_and_session_are_gone_after_a_stop(
403
488
  recorder, transcriber, monkeypatch
404
489
  ):
@@ -7,4 +7,4 @@ formatting into something that actually sounds good spoken aloud
7
7
  which is close to useless).
8
8
  """
9
9
 
10
- __version__ = "0.10.1"
10
+ __version__ = "0.10.2"
@@ -0,0 +1,11 @@
1
+ # Dictation cue words
2
+
3
+ Spoken alternatives to the Tink/Pop/Glass system sounds (`[stt] cues`).
4
+
5
+ Generated with the local Kokoro voice `af_heart`, one word per file:
6
+
7
+ vocalize speak "Start." --provider kokoro --voice af_heart --no-play -o vocalize/assets/cues/start.wav
8
+ vocalize speak "Stopped." --provider kokoro --voice af_heart --no-play -o vocalize/assets/cues/stopped.wav
9
+ vocalize speak "Ready." --provider kokoro --voice af_heart --no-play -o vocalize/assets/cues/ready.wav
10
+
11
+ Kokoro-82M is Apache-2.0 licensed.
@@ -490,6 +490,7 @@ def settings() -> None:
490
490
  click.echo(f"stt.language={stt['language']}")
491
491
  click.echo(f"stt.cleanup={'true' if stt['cleanup'] else 'false'}")
492
492
  click.echo(f"stt.max_seconds={stt['max_seconds']}")
493
+ click.echo(f"stt.cues={stt['cues']}")
493
494
 
494
495
 
495
496
  @main.command()
@@ -57,8 +57,12 @@ KNOWN_STT_KEYS = (
57
57
  "paste",
58
58
  "max_seconds",
59
59
  "sounds",
60
+ "cues",
60
61
  )
61
62
 
63
+ # What `cues` may be: the fixed system sounds, spoken words instead, or both.
64
+ STT_CUE_MODES = ("sounds", "words", "both")
65
+
62
66
  # `paste` is reserved by DEC-006 and deliberately does nothing in 0.10.0.
63
67
  STT_DEFAULTS = {
64
68
  "model": "small.en",
@@ -68,6 +72,7 @@ STT_DEFAULTS = {
68
72
  "paste": False,
69
73
  "max_seconds": 120,
70
74
  "sounds": True,
75
+ "cues": "sounds",
71
76
  }
72
77
 
73
78
  # The recorder self-stops at max_seconds and `dictate` backstops it, so this
@@ -291,6 +296,13 @@ def _validate_stt_table(value, path: Path) -> None:
291
296
  if flag is not None and not isinstance(flag, bool):
292
297
  raise ConfigError(f"Invalid stt.{key} {flag!r} in {path}: expected true or false.")
293
298
 
299
+ cues = value.get("cues")
300
+ if cues is not None and cues not in STT_CUE_MODES:
301
+ raise ConfigError(
302
+ f"Invalid stt.cues {cues!r} in {path}. Expected one of: "
303
+ f"{', '.join(STT_CUE_MODES)}."
304
+ )
305
+
294
306
 
295
307
  def _validate_input_device(device, path: Path) -> None:
296
308
  if not isinstance(device, str):
@@ -136,6 +136,16 @@ _SOUND_START = _SOUND_DIR / "Tink.aiff"
136
136
  _SOUND_STOP = _SOUND_DIR / "Pop.aiff"
137
137
  _SOUND_DONE = _SOUND_DIR / "Glass.aiff"
138
138
 
139
+ # Spoken alternatives to the three system sounds (`[stt] cues`), shipped
140
+ # with the package — same package-relative pattern as
141
+ # `local/install.py`'s `_RECORDER_DIR`.
142
+ _CUES_DIR = Path(__file__).resolve().parent / "assets" / "cues"
143
+ _CUE_WORDS = {
144
+ _SOUND_START: _CUES_DIR / "start.wav",
145
+ _SOUND_STOP: _CUES_DIR / "stopped.wav",
146
+ _SOUND_DONE: _CUES_DIR / "ready.wav",
147
+ }
148
+
139
149
  # Every notification this module can ever show. Fixed strings, no
140
150
  # interpolation: a transcript must never reach Notification Center, and
141
151
  # `_notify` enforces that by refusing anything not in this set.
@@ -309,17 +319,34 @@ def _recorder_pid(workdir: Path) -> int | None:
309
319
  # --- feedback: sounds and notifications -------------------------------
310
320
 
311
321
 
312
- def _play(sound: Path, stt: dict) -> None:
313
- """One feedback sound, through the machine-wide playback lock.
322
+ def _play(sound: Path, stt: dict, *, only: str | None = None) -> None:
323
+ """One feedback cue, through the machine-wide playback lock.
314
324
 
315
325
  Through `audio.play` and not a raw `afplay` so a sound queues behind
316
326
  (and can be stopped with) any read in progress — the overlap 0.9.1
317
327
  fixed. Never fatal: a missing system sound must not lose a dictation.
328
+
329
+ `[stt] cues` picks what plays instead of (or alongside) the sound:
330
+ "sounds" (default) is unchanged; "words" speaks the cue word in place
331
+ of the sound; "both" speaks it and then plays the sound. A missing
332
+ word file — a package installed without `vocalize/assets/cues/`, say
333
+ — falls back to the sound rather than saying nothing at all.
334
+
335
+ `only` plays just the "word" half or just the "sound" half of the cue,
336
+ for the one caller that needs them at different moments: `_start`
337
+ speaks before the microphone opens and plays the sound after.
318
338
  """
319
339
  if not stt.get("sounds", True):
320
340
  return
341
+ cues = stt.get("cues", "sounds")
342
+ word = _CUE_WORDS.get(sound) if cues in ("words", "both") else None
343
+ if word is not None and not word.is_file():
344
+ word = None # fall back to the sound
321
345
  try:
322
- audio.play(sound)
346
+ if word is not None and only != "sound":
347
+ audio.play(word)
348
+ if only != "word" and (word is None or cues == "both"):
349
+ audio.play(sound)
323
350
  except Exception: # noqa: BLE001, S110 — feedback is never worth failing a dictation for
324
351
  pass
325
352
 
@@ -1039,6 +1066,14 @@ def _start(workdir: Path, stt: dict) -> int:
1039
1066
  # that read saves its place, and this dictation offers it back once the
1040
1067
  # transcript has landed (DEC-003).
1041
1068
  audio.stop_playback(remember=True)
1069
+ # A spoken "Start." has to finish *before* the microphone opens, or it
1070
+ # is recorded and transcribed along with the dictation (`audio.play`
1071
+ # blocks until playback ends). The Tink plays *after* the recorder
1072
+ # reports it is recording, as it always has — so in "both" mode the
1073
+ # two cues keep distinct meanings: the word is "get ready", the sound
1074
+ # is "the microphone is open, talk now". The gap between them is the
1075
+ # recorder's start-up; closing it is issue #2.
1076
+ _play(_SOUND_START, stt, only="word")
1042
1077
  try:
1043
1078
  _launch_recorder(workdir, stt)
1044
1079
  except DictationError:
@@ -1056,7 +1091,7 @@ def _start(workdir: Path, stt: dict) -> int:
1056
1091
  _play(_SOUND_STOP, stt)
1057
1092
  _notify(_NOTIFY_RECORDER_FAILED)
1058
1093
  return 1
1059
- _play(_SOUND_START, stt)
1094
+ _play(_SOUND_START, stt, only="sound")
1060
1095
  return 0
1061
1096
 
1062
1097
 
File without changes