vocalize-cli 0.10.0__tar.gz → 0.10.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/.gitignore +1 -0
  2. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/CHANGELOG.md +62 -0
  3. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/PKG-INFO +7 -2
  4. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/README.md +6 -1
  5. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/dictation.md +18 -2
  6. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-6-release-0-10-0/report.md +1 -0
  7. vocalize_cli-0.10.2/docs/roadmap.md +25 -0
  8. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/claude_stop_hook.py +38 -3
  9. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_claude_stop_hook.py +61 -0
  10. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_cli.py +5 -2
  11. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_config.py +24 -0
  12. vocalize_cli-0.10.2/tests/test_cue_assets.py +28 -0
  13. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_dictate.py +156 -0
  14. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_kokoro_provider.py +2 -0
  15. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_recorder_build.py +36 -0
  16. vocalize_cli-0.10.2/tests/test_uv_path.py +62 -0
  17. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/__init__.py +1 -1
  18. vocalize_cli-0.10.2/vocalize/assets/cues/README.md +11 -0
  19. vocalize_cli-0.10.2/vocalize/assets/cues/ready.wav +0 -0
  20. vocalize_cli-0.10.2/vocalize/assets/cues/start.wav +0 -0
  21. vocalize_cli-0.10.2/vocalize/assets/cues/stopped.wav +0 -0
  22. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/cli.py +1 -0
  23. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/config.py +12 -0
  24. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/dictate.py +87 -9
  25. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/local/__init__.py +10 -2
  26. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/local/install.py +14 -4
  27. vocalize_cli-0.10.2/vocalize/recorder/Recorder.entitlements +8 -0
  28. vocalize_cli-0.10.0/docs/roadmap.md +0 -22
  29. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/.env.example +0 -0
  30. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/.github/workflows/ci.yml +0 -0
  31. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/LICENSE +0 -0
  32. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/installation.md +0 -0
  33. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/next-features-analysis.md +0 -0
  34. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/choreography.md +0 -0
  35. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/decisions.md +0 -0
  36. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/design.md +0 -0
  37. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/plan.md +0 -0
  38. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/review-0.10.0.md +0 -0
  39. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-1-status/project-plan.md +0 -0
  40. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-1-status/report.md +0 -0
  41. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-1-status/validate-exit.sh +0 -0
  42. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-10-release-0-11-0/project-plan.md +0 -0
  43. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-10-release-0-11-0/validate-exit.sh +0 -0
  44. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-2-stt-runtime/project-plan.md +0 -0
  45. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-2-stt-runtime/report.md +0 -0
  46. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-2-stt-runtime/validate-exit.sh +0 -0
  47. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-3-recorder/project-plan.md +0 -0
  48. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-3-recorder/report.md +0 -0
  49. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-3-recorder/validate-exit.sh +0 -0
  50. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-4-dictation/project-plan.md +0 -0
  51. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-4-dictation/report.md +0 -0
  52. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-4-dictation/validate-exit.sh +0 -0
  53. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-5-resume/project-plan.md +0 -0
  54. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-5-resume/report.md +0 -0
  55. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-5-resume/validate-exit.sh +0 -0
  56. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-6-release-0-10-0/project-plan.md +0 -0
  57. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-6-release-0-10-0/validate-exit.sh +0 -0
  58. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-7-portal-read/project-plan.md +0 -0
  59. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-7-portal-read/validate-exit.sh +0 -0
  60. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-8-portal-write/project-plan.md +0 -0
  61. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-8-portal-write/validate-exit.sh +0 -0
  62. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-9-portal-page/project-plan.md +0 -0
  63. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/run-9-portal-page/validate-exit.sh +0 -0
  64. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/spike-2026-09-01.md +0 -0
  65. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/split-assessment.md +0 -0
  66. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/plans/2026-09-next-features/verification.md +0 -0
  67. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/provider-credentials.md +0 -0
  68. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/research/2026-09-01-config-portal-design.md +0 -0
  69. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/research/2026-09-01-dictation-design.md +0 -0
  70. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/docs/research/2026-09-01-voicebox-findings.md +0 -0
  71. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/install_hook.py +0 -0
  72. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/install_quick_action.py +0 -0
  73. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/quick_actions/Dictate with Vocalize.workflow/Contents/Info.plist +0 -0
  74. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/quick_actions/Dictate with Vocalize.workflow/Contents/Resources/document.wflow +0 -0
  75. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/quick_actions/Speak Latest Plan.workflow/Contents/Info.plist +0 -0
  76. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/quick_actions/Speak Latest Plan.workflow/Contents/Resources/document.wflow +0 -0
  77. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/quick_actions/Speak with Vocalize.workflow/Contents/Info.plist +0 -0
  78. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/quick_actions/Speak with Vocalize.workflow/Contents/Resources/document.wflow +0 -0
  79. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/quick_actions/Stop Vocalize.workflow/Contents/Info.plist +0 -0
  80. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/quick_actions/Stop Vocalize.workflow/Contents/Resources/document.wflow +0 -0
  81. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/speak_options.py +0 -0
  82. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/hooks/speak_url_gate.py +0 -0
  83. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/pyproject.toml +0 -0
  84. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/conftest.py +0 -0
  85. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_audio.py +0 -0
  86. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_auth.py +0 -0
  87. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_cache.py +0 -0
  88. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_chain.py +0 -0
  89. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_clipboard.py +0 -0
  90. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_elevenlabs_provider.py +0 -0
  91. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_exceptions.py +0 -0
  92. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_google_provider.py +0 -0
  93. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_http.py +0 -0
  94. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_install_hook.py +0 -0
  95. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_install_quick_action.py +0 -0
  96. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_kokoro_manifest.py +0 -0
  97. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_kokoro_worker.py +0 -0
  98. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_ledger.py +0 -0
  99. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_listen_check.py +0 -0
  100. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_local_install.py +0 -0
  101. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_openai_provider.py +0 -0
  102. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_polly_provider.py +0 -0
  103. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_preprocess.py +0 -0
  104. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_providers_registry.py +0 -0
  105. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_readiness.py +0 -0
  106. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_say_provider.py +0 -0
  107. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_speak_options.py +0 -0
  108. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_speak_url_gate.py +0 -0
  109. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_tts.py +0 -0
  110. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_whisper_manifest.py +0 -0
  111. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_whisper_worker.py +0 -0
  112. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/tests/test_wizard.py +0 -0
  113. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/__main__.py +0 -0
  114. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/audio.py +0 -0
  115. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/auth.py +0 -0
  116. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/cache.py +0 -0
  117. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/chain.py +0 -0
  118. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/clipboard.py +0 -0
  119. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/exceptions.py +0 -0
  120. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/interrupted.py +0 -0
  121. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/ledger.py +0 -0
  122. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/local/kokoro_manifest.py +0 -0
  123. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/local/kokoro_worker.py +0 -0
  124. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/local/whisper_manifest.py +0 -0
  125. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/local/whisper_worker.py +0 -0
  126. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/preprocess.py +0 -0
  127. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/providers/__init__.py +0 -0
  128. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/providers/_http.py +0 -0
  129. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/providers/elevenlabs.py +0 -0
  130. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/providers/google.py +0 -0
  131. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/providers/kokoro.py +0 -0
  132. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/providers/openai.py +0 -0
  133. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/providers/polly.py +0 -0
  134. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/providers/say.py +0 -0
  135. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/readiness.py +0 -0
  136. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/recorder/Info.plist.in +0 -0
  137. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/recorder/VocalizeRecorder.swift +0 -0
  138. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/tts.py +0 -0
  139. {vocalize_cli-0.10.0 → vocalize_cli-0.10.2}/vocalize/wizard.py +0 -0
@@ -11,5 +11,6 @@ dist/
11
11
  venv/
12
12
  *.mp3
13
13
  *.wav
14
+ !vocalize/assets/cues/*.wav
14
15
  .claude/
15
16
  uv.lock
@@ -3,6 +3,68 @@
3
3
  All notable changes to this project are documented here. Format follows
4
4
  [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
5
5
 
6
+ ## 0.10.2 - 2026-09-02
7
+
8
+ ### Added
9
+
10
+ - **`[stt] cues`** picks what a dictation's feedback sounds — say instead
11
+ of, or alongside, the Tink/Pop/Glass system sounds. `"sounds"` (default)
12
+ is unchanged; `"words"` speaks "Start.", "Stopped.", "Ready." in their
13
+ place; `"both"` speaks the word and then plays the sound. The word files
14
+ ship in `vocalize/assets/cues/`, generated with the local Kokoro voice.
15
+ A spoken "Start." plays *before* the recorder launches — played once the
16
+ microphone was open it would be recorded and transcribed along with the
17
+ dictation. In `"both"` mode the Tink still plays *after* the microphone
18
+ opens, so the two cues keep distinct meanings: the word is "get ready",
19
+ the sound is "talk now". The plain Tink is unaffected.
20
+
21
+ ### Known issue
22
+
23
+ - In `"words"` mode there is no cue for the moment the microphone actually
24
+ opens, which is a second or so after "Start." finishes (LaunchServices
25
+ start-up plus the input device switching on). People start talking too
26
+ soon and lose their first word. The fix — open and warm the microphone
27
+ first, play the cue, and only then capture — is tracked in
28
+ [#2](https://github.com/matthager12-collab/vocalize/issues/2). Until
29
+ then: in `"words"` mode, wait a beat after "Start."; in `"both"` mode,
30
+ talk after the Tink.
31
+
32
+ ## 0.10.1 - 2026-09-02
33
+
34
+ Three fixes found in the first owner-present run of 0.10.0's dictation.
35
+ Together they meant no hotkey dictation could succeed on 0.10.0; upgrade.
36
+
37
+ ### Fixed
38
+
39
+ - **No dictation could ever start on a fresh install.** The recorder was
40
+ signed with the hardened runtime but without the
41
+ `com.apple.security.device.audio-input` entitlement, so macOS refused the
42
+ microphone on the spot — no permission dialog, status stuck at
43
+ `notDetermined` — and every first press ended in "The recorder did not
44
+ start". The bundle is now signed with
45
+ `vocalize/recorder/Recorder.entitlements`, and the entitlements are part
46
+ of the recorder's fingerprint, so `vocalize local install --stt` rebuilds
47
+ the bundle once (and, as with any rebuild, macOS asks for the microphone
48
+ again — it never actually asked before).
49
+ - **Every hotkey dictation ended in "Dictation failed" on a machine whose
50
+ `uv` came from Homebrew.** A Services environment has a bare PATH, and
51
+ `uv_path()` looked only there and in `~/.local/bin`; the same dictation
52
+ worked from a terminal. `/opt/homebrew/bin/uv` and `/usr/local/bin/uv`
53
+ are now tried too (this also covers Kokoro from a Quick Action).
54
+ - **Holding the dictation hotkey down turned into a cancel-and-restart
55
+ loop.** macOS re-fires a Service shortcut at the key-repeat rate, and
56
+ every repeat landed as a second press. Presses within half a second of
57
+ the previous one are now ignored as the same press; a deliberate cancel
58
+ is "press, a beat, press" inside the two-second window, as before.
59
+ - `hooks/claude_stop_hook.py --latest`, run from inside a Claude Code turn
60
+ (which is how `/speak` runs it), spoke the agent's own status line —
61
+ "Checking settings." — instead of the response the user asked to hear.
62
+ It now skips the turn in progress, back past the `/speak` message
63
+ itself, and speaks the response before it. From a plain terminal, where
64
+ no turn is in progress, `--latest` still speaks the newest response; the
65
+ hook tells the two apart by the `CLAUDECODE` variable Claude Code sets
66
+ in its shell. The Stop-hook path is unchanged.
67
+
6
68
  ## 0.10.0 - 2026-09-02
7
69
 
8
70
  ### Added
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: vocalize-cli
3
- Version: 0.10.0
3
+ Version: 0.10.2
4
4
  Summary: A CLI that turns text, markdown, or piped stdin into speech via the ElevenLabs API, with markdown-table-aware preprocessing.
5
5
  Project-URL: Homepage, https://github.com/matthager12-collab/vocalize
6
6
  Project-URL: Repository, https://github.com/matthager12-collab/vocalize
@@ -478,12 +478,15 @@ terminal, if you'd rather trigger it that way.
478
478
  transcript is on your clipboard; nothing is typed for you automatically.
479
479
  - **Nothing heard** (silence, or a microphone that isn't actually picking
480
480
  anything up) ends the dictation quietly — no clipboard write.
481
- - **Cancel** with a second press *within two seconds* of the first, or at
481
+ - **Cancel** with a second press *within two seconds* of the first (but
482
+ not within half a second — that's a held key, and it's ignored), or at
482
483
  any point with `vocalize listen --cancel` — the audio is discarded, never
483
484
  transcribed.
484
485
  - **A third press while transcribing is refused**: a Pop, and "Still
485
486
  transcribing the last dictation." Wait for the clipboard notification, or
486
487
  `--cancel`, before dictating again.
488
+ - Can't tell Tink from Pop from Glass yet? Set `[stt] cues = "words"` and
489
+ vocalize says "Start.", "Stopped.", "Ready." instead.
487
490
 
488
491
  ### `vocalize listen`
489
492
 
@@ -518,6 +521,7 @@ input_device = "" # "" = system default; else an exact name from --list-dev
518
521
  cleanup = false # send the transcript (never audio) to Claude first
519
522
  max_seconds = 120 # 1-600; the recorder self-stops here, dictate backstops it
520
523
  sounds = true # the Tink/Pop/Glass feedback sounds
524
+ cues = "sounds" # "sounds" | "words" | "both" — speak "Start."/"Stopped."/"Ready." instead
521
525
  ```
522
526
 
523
527
  | Key | Allowed values | Default |
@@ -529,6 +533,7 @@ sounds = true # the Tink/Pop/Glass feedback sounds
529
533
  | `paste` | reserved — not implemented in 0.10.0 | `false` |
530
534
  | `max_seconds` | integer, 1–600 | `120` |
531
535
  | `sounds` | `true` / `false` | `true` |
536
+ | `cues` | `sounds`, `words`, `both` | `sounds` |
532
537
 
533
538
  An unknown key warns on stderr; a bad value is a `ConfigError` naming it —
534
539
  every one of these becomes a subprocess argument eventually, so nothing
@@ -442,12 +442,15 @@ terminal, if you'd rather trigger it that way.
442
442
  transcript is on your clipboard; nothing is typed for you automatically.
443
443
  - **Nothing heard** (silence, or a microphone that isn't actually picking
444
444
  anything up) ends the dictation quietly — no clipboard write.
445
- - **Cancel** with a second press *within two seconds* of the first, or at
445
+ - **Cancel** with a second press *within two seconds* of the first (but
446
+ not within half a second — that's a held key, and it's ignored), or at
446
447
  any point with `vocalize listen --cancel` — the audio is discarded, never
447
448
  transcribed.
448
449
  - **A third press while transcribing is refused**: a Pop, and "Still
449
450
  transcribing the last dictation." Wait for the clipboard notification, or
450
451
  `--cancel`, before dictating again.
452
+ - Can't tell Tink from Pop from Glass yet? Set `[stt] cues = "words"` and
453
+ vocalize says "Start.", "Stopped.", "Ready." instead.
451
454
 
452
455
  ### `vocalize listen`
453
456
 
@@ -482,6 +485,7 @@ input_device = "" # "" = system default; else an exact name from --list-dev
482
485
  cleanup = false # send the transcript (never audio) to Claude first
483
486
  max_seconds = 120 # 1-600; the recorder self-stops here, dictate backstops it
484
487
  sounds = true # the Tink/Pop/Glass feedback sounds
488
+ cues = "sounds" # "sounds" | "words" | "both" — speak "Start."/"Stopped."/"Ready." instead
485
489
  ```
486
490
 
487
491
  | Key | Allowed values | Default |
@@ -493,6 +497,7 @@ sounds = true # the Tink/Pop/Glass feedback sounds
493
497
  | `paste` | reserved — not implemented in 0.10.0 | `false` |
494
498
  | `max_seconds` | integer, 1–600 | `120` |
495
499
  | `sounds` | `true` / `false` | `true` |
500
+ | `cues` | `sounds`, `words`, `both` | `sounds` |
496
501
 
497
502
  An unknown key warns on stderr; a bad value is a `ConfigError` naming it —
498
503
  every one of these becomes a subprocess argument eventually, so nothing
@@ -132,7 +132,10 @@ touching System Settings at all.
132
132
 
133
133
  **A second press within two seconds of the first is a cancel**, not a
134
134
  stop — treated as "I changed my mind" rather than the end of a very short
135
- sentence. So is `vocalize listen --cancel`, from a terminal, at any point
135
+ sentence. (Presses closer than half a second apart are the key being
136
+ *held* — macOS repeats a Service shortcut at the key-repeat rate — and are
137
+ ignored, so a cancel is "press, a beat, press".) So is
138
+ `vocalize listen --cancel`, from a terminal, at any point
136
139
  — including while a take is being transcribed. (The transcription that was
137
140
  already running keeps its own copy of the recording and finishes normally;
138
141
  `--cancel` only releases the hotkey so your next press starts a fresh
@@ -144,6 +147,16 @@ never runs two transcriptions over the same recording, and it never
144
147
  silently drops one either. Wait for the clipboard notification, or use
145
148
  `--cancel`.
146
149
 
150
+ New to the sounds and can't tell Tink from Pop from Glass? Set
151
+ `[stt] cues = "words"` and vocalize says "Start.", "Stopped." and "Ready."
152
+ instead, or `"both"` to hear the word and then its sound — see the `[stt]`
153
+ table below. Timing matters: "Start." is spoken *before* the microphone
154
+ opens (so it is never in your recording), and the microphone is open
155
+ about a second later — the Tink marks that moment. So in `"both"` mode
156
+ talk after the Tink; in `"words"` mode give it a beat after "Start.".
157
+ Closing that gap is
158
+ [#2](https://github.com/matthager12-collab/vocalize/issues/2).
159
+
147
160
  ## `vocalize listen`
148
161
 
149
162
  The terminal-facing primitive behind the hotkey:
@@ -213,6 +226,7 @@ cleanup = false
213
226
  paste = false
214
227
  max_seconds = 120
215
228
  sounds = true
229
+ cues = "sounds" # "sounds" | "words" | "both"
216
230
  ```
217
231
 
218
232
  | Key | Type / allowlist | Default | Notes |
@@ -223,7 +237,8 @@ sounds = true
223
237
  | `cleanup` | `true` / `false` | `false` | see `--cleanup` above |
224
238
  | `paste` | reserved | `false` | not implemented in 0.10.0 — setting it does nothing |
225
239
  | `max_seconds` | integer, 1–600 | `120` | the recorder self-stops here; `dictate` backstops it a few seconds later in case the recorder doesn't |
226
- | `sounds` | `true` / `false` | `true` | the Tink/Pop/Glass feedback; `false` silences all three |
240
+ | `sounds` | `true` / `false` | `true` | the Tink/Pop/Glass feedback; `false` silences all three (words included) |
241
+ | `cues` | `sounds`, `words`, `both` | `sounds` | `"words"` speaks "Start.", "Stopped.", "Ready." instead of the system sounds; `"both"` speaks the word and then plays the sound — for the start cue, the word before the microphone opens and the Tink once it has. Has no effect while `sounds = false`. |
227
242
 
228
243
  Every value here eventually becomes a subprocess argument — the recorder's
229
244
  `--device`, or the whisper worker's `--model`/`--language` — so each one is
@@ -240,6 +255,7 @@ stt.model=small.en
240
255
  stt.language=en
241
256
  stt.cleanup=false
242
257
  stt.max_seconds=120
258
+ stt.cues=sounds
243
259
  ```
244
260
 
245
261
  ### The input-device gotcha
@@ -11,6 +11,7 @@ Branch `next-features`, shared checkout. Source: [project-plan.md](./project-pla
11
11
  - `status` (no mic): [run-1-status/report.md](../run-1-status/report.md) (the module and its exit-code contract) plus [run-4-dictation/report.md](../run-4-dictation/report.md) T-45 (the four STT readiness rows), further hardened this run (mic-verdict age, locked probe registry — see review-0.10.0.md).
12
12
  - Resume drill (no mic): [run-5-resume/report.md](../run-5-resume/report.md) T-46/T-47, with this run's review closing the gaps it left open (DEC-013's queued-read case, DEC-014f's voice/speed/chunking carry-over).
13
13
  - Note: rebuilding the recorder for the Swift permission-dialog fix (`94e5a46`) reset the microphone grant on the reference Mac to `notDetermined`, so Manual check 1 starts from a genuinely fresh state — not a regression, the state the fix exists for.
14
+ - **Owner-present run, 2026-09-02 evening (after 0.10.0 published):** Manual check 1 (mic grant) and 2 (hotkey path, real voice, transcript on the clipboard) done — but only after three defects were fixed and released as 0.10.1: the recorder was signed with the hardened runtime and no `audio-input` entitlement (no prompt ever appeared); `uv_path()` could not find Homebrew's uv from a Service's bare PATH (every stop ended in "Dictation failed"); and a held hotkey re-fired as serial cancel/restart presses (now debounced). Manual 3 is partly covered (held-key refusal); 4, 4b, 4c and 5 remain.
14
15
  - **T-53: pending** — merge and publish. Not started. This run does not merge to main or publish; per project-plan.md that happens on the owner's word, after T-52. The orchestrator handles this after reading this report.
15
16
 
16
17
  ## Security-gate summary (every run's verdict)
@@ -0,0 +1,25 @@
1
+ # Roadmap
2
+
3
+ Future work, with the research behind each item. Status legend: **planned** = analysed, decisions pending, nothing built; **spiked** = feasibility measured; **building**; **shipped**.
4
+
5
+ | Item | Status | Tracking | Research |
6
+ |---|---|---|---|
7
+ | Hotkey-triggered local dictation (voice → text via a local Whisper model, Quick Action toggle, clipboard output, optional cleanup) | **shipped** in 0.10.0, usable from 0.10.1 | [#1](https://github.com/matthager12-collab/vocalize/issues/1) · [plan](plans/2026-09-next-features/plan.md) | [analysis](next-features-analysis.md) · [full design](research/2026-09-01-dictation-design.md) · [voicebox findings](research/2026-09-01-voicebox-findings.md) |
8
+ | `vocalize status` one-screen readiness check | **shipped** in 0.10.0 | [#1](https://github.com/matthager12-collab/vocalize/issues/1) | [analysis](next-features-analysis.md) |
9
+ | Spoken cues for dictation (`[stt] cues = "words"`: "start" / "stopped" / "ready" instead of, or as well as, the sounds) | **shipped** in 0.10.2 | — | owner request, 2026-09-02 |
10
+ | Cue timing: the "talk now" cue must follow the *open* microphone and never be recorded (recorder warms the input, reports ready, waits for a `go` marker; `dictate` plays the cue, then touches `go`) | planned | [#2](https://github.com/matthager12-collab/vocalize/issues/2) | first live use of 0.10.2: people talk after "Start." and lose the first word |
11
+ | Config portal (`vocalize portal`, stdlib local web page on top of the same readiness rows) | planned — runs 7–10 of the [plan](plans/2026-09-next-features/plan.md) | [#1](https://github.com/matthager12-collab/vocalize/issues/1) | [analysis](next-features-analysis.md) · [full design](research/2026-09-01-config-portal-design.md) |
12
+
13
+ ## Decisions (2026-09-01, closed 2026-09-02)
14
+
15
+ 1. The dictation spike ran ([results](plans/2026-09-next-features/spike-2026-09-01.md)); Whisper `small.en` is the default.
16
+ 2. `vocalize status` first (0.10.0); the portal follows as 0.11.0.
17
+ 3. Hotkey chord ⌃⌥⌘D, set by the user in System Settings.
18
+
19
+ ## Deferred on purpose
20
+
21
+ Hold-to-talk (needs an event tap + Input Monitoring, ~6 h), auto-paste (Accessibility, ~2 h, belongs in the recorder bundle), Apple STT mode (~1.5 h), menu-bar agent (~12 h), local-LLM cleanup, resident/pre-warmed STT worker (only if the spike shows >2.5 s overhead), a native SwiftUI/Tauri app.
22
+
23
+ ## Sequencing suggestion
24
+
25
+ Dictation first (the spike decides the engine, then ~20 h), portal or `status` second.
@@ -7,7 +7,10 @@
7
7
  2. On demand, with `--latest`. Nothing is read from stdin; the script finds
8
8
  the most recently written transcript under ~/.claude/projects and speaks
9
9
  that response. Run it when you want speech instead of installing the
10
- hook and getting it after every turn.
10
+ hook and getting it after every turn. Run from inside a Claude Code turn
11
+ (the /speak command does this; Claude Code sets CLAUDECODE=1 in its
12
+ shell) it skips that turn — the newest text there is the agent's own
13
+ status line — and speaks the response before it.
11
14
 
12
15
  Add `--print-length` to either mode to print the response's character
13
16
  count instead of speaking — for wrappers deciding whether to ask about
@@ -59,7 +62,27 @@ def _speech_timeout(text: str) -> int:
59
62
  return min(TIMEOUT_CEILING_SECONDS, TIMEOUT_BASE_SECONDS + len(text) // CHARS_PER_SECOND)
60
63
 
61
64
 
62
- def _extract_last_assistant_text(transcript_path: str) -> str:
65
+ def _is_human_message(entry: dict) -> bool:
66
+ """A user entry the person typed, not one Claude Code wrote.
67
+
68
+ Claude Code logs tool results and its own injected text — a skill's
69
+ body, say — as user entries too; the latter carry ``isMeta``.
70
+ """
71
+ if entry.get("type") != "user" or entry.get("isMeta"):
72
+ return False
73
+ content = (entry.get("message") or {}).get("content") or []
74
+ if isinstance(content, str):
75
+ return True
76
+ return any(isinstance(b, dict) and b.get("type") != "tool_result" for b in content)
77
+
78
+
79
+ def _extract_last_assistant_text(transcript_path: str, skip_current_turn: bool = False) -> str:
80
+ """Newest assistant text in the transcript.
81
+
82
+ With skip_current_turn, everything after the newest human message is
83
+ the turn still in progress (the one running this script); skip it and
84
+ return the response before that message instead.
85
+ """
63
86
  last_text_parts: list[str] = []
64
87
  try:
65
88
  with open(transcript_path, "r", encoding="utf-8") as f:
@@ -76,6 +99,11 @@ def _extract_last_assistant_text(transcript_path: str) -> str:
76
99
  except json.JSONDecodeError:
77
100
  continue
78
101
 
102
+ if skip_current_turn:
103
+ if _is_human_message(entry):
104
+ skip_current_turn = False # past the /speak message; older is fair game
105
+ continue
106
+
79
107
  if entry.get("type") != "assistant":
80
108
  continue
81
109
 
@@ -117,17 +145,24 @@ def main() -> int:
117
145
  if "--latest" in sys.argv[1:]:
118
146
  # On-demand mode: no hook payload on stdin, so don't read it at all.
119
147
  transcript_path = _find_latest_transcript()
148
+ # Run from inside a Claude Code turn (the /speak command does this,
149
+ # and Claude Code sets CLAUDECODE=1 in its shell), the newest
150
+ # assistant text is that turn's own status line — "Checking
151
+ # settings." — not the response the user asked to hear. From a plain
152
+ # terminal there is no turn in progress and the newest text is right.
153
+ skip_current_turn = bool(os.environ.get("CLAUDECODE"))
120
154
  else:
121
155
  try:
122
156
  payload = json.load(sys.stdin)
123
157
  except json.JSONDecodeError:
124
158
  payload = {}
125
159
  transcript_path = payload.get("transcript_path")
160
+ skip_current_turn = False # at Stop the turn is over: newest text is the target
126
161
 
127
162
  if not transcript_path:
128
163
  return 0 # nothing to do — don't block Claude Code on a hook error
129
164
 
130
- text = _extract_last_assistant_text(transcript_path)
165
+ text = _extract_last_assistant_text(transcript_path, skip_current_turn)
131
166
  if not text.strip():
132
167
  if "--print-length" in sys.argv[1:]:
133
168
  print(0)
@@ -14,6 +14,10 @@ def _assistant(content) -> str:
14
14
  return json.dumps({"type": "assistant", "message": {"content": content}})
15
15
 
16
16
 
17
+ def _user(content, **extra) -> str:
18
+ return json.dumps({"type": "user", "message": {"content": content}, **extra})
19
+
20
+
17
21
  def _text(text: str) -> dict:
18
22
  return {"type": "text", "text": text}
19
23
 
@@ -28,6 +32,9 @@ def _patch_main(monkeypatch, payload, which="/usr/local/bin/vocalize", run=None,
28
32
  """Wire up stdin, PATH lookup and subprocess so main() can't shell out."""
29
33
  monkeypatch.delenv("VOCALIZE_MAX_CHARS", raising=False)
30
34
  monkeypatch.delenv("VOCALIZE_BIN", raising=False)
35
+ # The suite itself often runs inside a Claude Code turn, where this is
36
+ # set; a --latest test must not change meaning depending on who ran it.
37
+ monkeypatch.delenv("CLAUDECODE", raising=False)
31
38
  monkeypatch.setattr(hook.sys, "argv", ["claude_stop_hook.py"])
32
39
  monkeypatch.setattr(
33
40
  hook.sys, "stdin", io.StringIO(json.dumps(payload)) if stdin is None else stdin
@@ -102,6 +109,31 @@ def test_falls_back_past_tool_use_only_entry(tmp_path):
102
109
  assert hook._extract_last_assistant_text(path) == "the spoken answer"
103
110
 
104
111
 
112
+ def test_skip_current_turn_returns_the_response_before_the_newest_human_message(tmp_path):
113
+ # /speak runs the hook mid-turn. By then the agent has written a status
114
+ # line and Claude Code has logged the Bash call that runs the hook (and
115
+ # its result). The response the user wants is the one BEFORE /speak.
116
+ # Only the typed /speak line counts as the person speaking: tool results
117
+ # and Claude Code's own injected text (isMeta) are user entries too.
118
+ path = _write_transcript(
119
+ tmp_path,
120
+ [
121
+ _assistant([_text("the long response the user asked to hear")]),
122
+ _user("<command-name>/speak</command-name>"),
123
+ _user([_text("Base directory for this skill: ...")], isMeta=True),
124
+ _assistant([_text("Checking settings.")]),
125
+ _assistant([{"type": "tool_use", "name": "Bash", "input": {}}]),
126
+ _user([{"type": "tool_result", "tool_use_id": "t1", "content": "ok"}]),
127
+ ],
128
+ )
129
+
130
+ assert hook._extract_last_assistant_text(path, skip_current_turn=True) == (
131
+ "the long response the user asked to hear"
132
+ )
133
+ # The Stop-hook path still wants the newest text: at Stop, the turn is over.
134
+ assert hook._extract_last_assistant_text(path) == "Checking settings."
135
+
136
+
105
137
  def test_ignores_malformed_json_lines(tmp_path):
106
138
  # The garbage line must sit AFTER the newest assistant entry. The scan
107
139
  # walks reversed(lines) and breaks on the first text it finds, so a
@@ -326,6 +358,35 @@ def test_latest_mode_speaks_newest_transcript_without_reading_stdin(monkeypatch,
326
358
  assert calls[0][-1] == "newest session"
327
359
 
328
360
 
361
+ def test_latest_mode_skips_the_current_turn_only_inside_claude_code(monkeypatch, tmp_path):
362
+ transcript = tmp_path / "session.jsonl"
363
+ transcript.write_text(
364
+ "\n".join(
365
+ [
366
+ _assistant([_text("the long response")]),
367
+ _user("/speak"),
368
+ _assistant([_text("Checking settings.")]),
369
+ ]
370
+ )
371
+ + "\n",
372
+ encoding="utf-8",
373
+ )
374
+ calls = _patch_main(monkeypatch, {})
375
+ monkeypatch.setattr(hook.sys, "argv", ["claude_stop_hook.py", "--latest"])
376
+ monkeypatch.setattr(hook, "_find_latest_transcript", lambda: str(transcript))
377
+
378
+ # From a plain terminal there is no turn in progress: newest text wins,
379
+ # exactly as the README's on-demand mode promises.
380
+ assert hook.main() == 0
381
+ assert calls[-1][-1] == "Checking settings."
382
+
383
+ # Inside a Claude Code turn, that newest text is the agent's own status
384
+ # line; the user wants the response before their /speak.
385
+ monkeypatch.setenv("CLAUDECODE", "1")
386
+ assert hook.main() == 0
387
+ assert calls[-1][-1] == "the long response"
388
+
389
+
329
390
  def test_subprocess_failure_is_logged_not_raised(monkeypatch, tmp_path, capsys):
330
391
  def boom(argv):
331
392
  raise RuntimeError("no audio device")
@@ -624,8 +624,10 @@ def test_settings_prints_the_stt_lines(monkeypatch, tmp_path):
624
624
  _isolate_overflow_env(monkeypatch, tmp_path)
625
625
  cfg = tmp_path / "vocalize" / "config.toml"
626
626
  cfg.parent.mkdir(parents=True, exist_ok=True)
627
- cfg.write_text('[stt]\nmodel = "base.en"\ncleanup = true\nmax_seconds = 30\n',
628
- encoding="utf-8")
627
+ cfg.write_text(
628
+ '[stt]\nmodel = "base.en"\ncleanup = true\nmax_seconds = 30\ncues = "sounds"\n',
629
+ encoding="utf-8",
630
+ )
629
631
 
630
632
  result = CliRunner().invoke(main, ["settings"])
631
633
 
@@ -634,6 +636,7 @@ def test_settings_prints_the_stt_lines(monkeypatch, tmp_path):
634
636
  assert "stt.language=en" in result.output
635
637
  assert "stt.cleanup=true" in result.output
636
638
  assert "stt.max_seconds=30" in result.output
639
+ assert "stt.cues=sounds" in result.output
637
640
 
638
641
 
639
642
  def test_settings_prints_defaults_when_nothing_is_configured(monkeypatch, tmp_path):
@@ -697,6 +697,30 @@ def test_a_non_boolean_stt_flag_is_refused(monkeypatch, tmp_path, key):
697
697
  assert f"stt.{key}" in str(excinfo.value)
698
698
 
699
699
 
700
+ @pytest.mark.parametrize("value", ['"chime"', "1"])
701
+ def test_an_invalid_stt_cues_value_is_refused(monkeypatch, tmp_path, value):
702
+ with pytest.raises(ConfigError) as excinfo:
703
+ _load_stt(monkeypatch, tmp_path, f"[stt]\ncues = {value}\n")
704
+
705
+ message = str(excinfo.value)
706
+ assert "stt.cues" in message
707
+ assert "sounds" in message and "words" in message and "both" in message
708
+
709
+
710
+ @pytest.mark.parametrize("value", ["sounds", "words", "both"])
711
+ def test_each_stt_cues_mode_is_accepted(monkeypatch, tmp_path, value):
712
+ from vocalize.config import resolve_stt
713
+
714
+ data = _load_stt(monkeypatch, tmp_path, f'[stt]\ncues = "{value}"\n')
715
+ assert resolve_stt(data)["cues"] == value
716
+
717
+
718
+ def test_stt_cues_defaults_to_sounds(monkeypatch, tmp_path):
719
+ from vocalize.config import resolve_stt
720
+
721
+ assert resolve_stt(_load_stt(monkeypatch, tmp_path, ""))["cues"] == "sounds"
722
+
723
+
700
724
  def test_an_stt_value_that_is_not_a_table_is_refused(monkeypatch, tmp_path):
701
725
  with pytest.raises(ConfigError) as excinfo:
702
726
  _load_stt(monkeypatch, tmp_path, 'stt = "small.en"\n')
@@ -0,0 +1,28 @@
1
+ """The shipped `[stt] cues` word files: present, small, and well-formed.
2
+
3
+ Not a test of speech quality — just the shape that `dictate._play` and the
4
+ packaged wheel both depend on: a real mono 16-bit WAV, short and light
5
+ enough to ship and to speak without lagging behind the sound it replaces.
6
+ """
7
+
8
+ import wave
9
+ from pathlib import Path
10
+
11
+ import pytest
12
+
13
+ from vocalize.dictate import _CUE_WORDS
14
+
15
+ MAX_SECONDS = 1.5
16
+ MAX_BYTES = 80_000
17
+
18
+
19
+ @pytest.mark.parametrize("path", list(_CUE_WORDS.values()), ids=lambda p: p.name)
20
+ def test_a_cue_word_file_is_a_small_mono_16_bit_wav(path: Path):
21
+ assert path.is_file(), f"missing cue asset: {path}"
22
+ assert path.stat().st_size <= MAX_BYTES
23
+
24
+ with wave.open(str(path), "rb") as reader:
25
+ assert reader.getnchannels() == 1
26
+ assert reader.getsampwidth() == 2
27
+ duration = reader.getnframes() / reader.getframerate()
28
+ assert duration <= MAX_SECONDS