echoai-helper 1.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. echoai_helper-1.2.0.dist-info/METADATA +306 -0
  2. echoai_helper-1.2.0.dist-info/RECORD +67 -0
  3. echoai_helper-1.2.0.dist-info/WHEEL +5 -0
  4. echoai_helper-1.2.0.dist-info/entry_points.txt +2 -0
  5. echoai_helper-1.2.0.dist-info/licenses/LICENSE +21 -0
  6. echoai_helper-1.2.0.dist-info/top_level.txt +2 -0
  7. scripts/__init__.py +0 -0
  8. scripts/calibrate_vad.py +183 -0
  9. scripts/check_audio.py +317 -0
  10. scripts/install_launcher.py +54 -0
  11. scripts/setup_audio.py +65 -0
  12. src/AudioRecorder.py +62 -0
  13. src/AudioTranscriber.py +532 -0
  14. src/GPTResponder.py +293 -0
  15. src/ResponseManager.py +374 -0
  16. src/SettingsManager.py +147 -0
  17. src/TemplateManager.py +186 -0
  18. src/TranscriberModels.py +158 -0
  19. src/TranscriptUI.py +499 -0
  20. src/app.py +1053 -0
  21. src/asr/asr_factory.py +44 -0
  22. src/asr/asr_interface.py +33 -0
  23. src/asr/asr_with_vad.py +266 -0
  24. src/asr/azure_asr.py +76 -0
  25. src/asr/diarization.py +255 -0
  26. src/asr/faster_whisper_asr.py +44 -0
  27. src/asr/fun_asr.py +119 -0
  28. src/asr/hypothesis.py +164 -0
  29. src/asr/models/silero_vad.onnx +0 -0
  30. src/asr/openai_whisper_asr.py +35 -0
  31. src/asr/segmenter.py +301 -0
  32. src/asr/vad.py +55 -0
  33. src/asr/whisper_cpp_asr.py +42 -0
  34. src/audio/__init__.py +43 -0
  35. src/audio/backend.py +109 -0
  36. src/audio/coreaudio.py +286 -0
  37. src/audio/macos.py +168 -0
  38. src/audio/setup_macos.py +402 -0
  39. src/audio/windows.py +111 -0
  40. src/cli.py +178 -0
  41. src/conf.yaml +120 -0
  42. src/config.py +312 -0
  43. src/custom_speech_recognition/__init__.py +1565 -0
  44. src/custom_speech_recognition/__main__.py +24 -0
  45. src/custom_speech_recognition/audio.py +318 -0
  46. src/custom_speech_recognition/exceptions.py +22 -0
  47. src/export_dialog.py +357 -0
  48. src/export_markdown.py +175 -0
  49. src/launcher_macos.py +171 -0
  50. src/llm/__init__.py +4 -0
  51. src/llm/cli_provider.py +189 -0
  52. src/llm/litellm_provider.py +146 -0
  53. src/llm/llm_provider.py +42 -0
  54. src/llm/openai_provider.py +85 -0
  55. src/llm/provider_factory.py +96 -0
  56. src/polish.py +242 -0
  57. src/profiles.py +106 -0
  58. src/prompts.py +95 -0
  59. src/resources/config/settings.json +13 -0
  60. src/resources/prompt/Interview Analysis.md +96 -0
  61. src/resources/prompt/Meeting Analysis.md +235 -0
  62. src/resources/prompt/case_detail/inbound_cs.txt +26 -0
  63. src/resources/prompt/case_detail/none.txt +3 -0
  64. src/resources/prompt/knowledge/none.txt +2 -0
  65. src/resources/prompt/system_role/inbound_cs.py +31 -0
  66. src/resources/prompt/system_role/phone_interview.py +32 -0
  67. src/session.py +303 -0
@@ -0,0 +1,306 @@
1
+ Metadata-Version: 2.4
2
+ Name: echoai-helper
3
+ Version: 1.2.0
4
+ Summary: Real-time meeting transcription and interview assistance, on-device
5
+ Author: colakang
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/colakang/echoai_helper
8
+ Project-URL: Issues, https://github.com/colakang/echoai_helper/issues
9
+ Keywords: transcription,asr,meeting-notes,diarization,funasr
10
+ Requires-Python: <3.13,>=3.12
11
+ Description-Content-Type: text/markdown
12
+ License-File: LICENSE
13
+ Requires-Dist: sounddevice; sys_platform != "win32"
14
+ Requires-Dist: PyAudioWPatch; sys_platform == "win32"
15
+ Requires-Dist: numpy<2
16
+ Requires-Dist: scipy
17
+ Requires-Dist: torch
18
+ Requires-Dist: torchaudio
19
+ Requires-Dist: funasr
20
+ Requires-Dist: modelscope
21
+ Requires-Dist: huggingface_hub
22
+ Requires-Dist: loguru
23
+ Requires-Dist: onnxruntime
24
+ Requires-Dist: soundfile
25
+ Requires-Dist: openai
26
+ Requires-Dist: customtkinter==5.1.3
27
+ Requires-Dist: python-dotenv
28
+ Requires-Dist: pyyaml
29
+ Requires-Dist: pytz
30
+ Provides-Extra: litellm
31
+ Requires-Dist: litellm; extra == "litellm"
32
+ Provides-Extra: dev
33
+ Requires-Dist: pytest; extra == "dev"
34
+ Dynamic: license-file
35
+
36
+ # 🎙️ EchoAI Helper
37
+
38
+ [![GitHub Stars](https://img.shields.io/github/stars/colakang/echoai_helper?style=social)](https://github.com/colakang/echoai_helper/stargazers)
39
+ [![License](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
40
+ [![Python](https://img.shields.io/badge/python-3.12-blue.svg)](https://python.org)
41
+ [![macOS](https://img.shields.io/badge/macOS-13%2B-lightgrey.svg)](#-install)
42
+ [![Windows](https://img.shields.io/badge/Windows-10%2B-lightgrey.svg)](#windows)
43
+
44
+ Real-time meeting transcription and interview assistance, running on your own
45
+ machine. It records both sides of a conversation — your microphone and whatever
46
+ the meeting app is playing — transcribes them as they happen, labels who is
47
+ speaking, and exports readable notes.
48
+
49
+ Speech recognition is local. Audio never leaves the machine unless you ask a
50
+ language model to clean up the finished transcript.
51
+
52
+ <p>
53
+ <a href="https://www.producthunt.com/posts/echoai-interview-copilot?embed=true&utm_source=badge-featured&utm_medium=badge&utm_souce=badge-echoai&#0045;interview&#0045;copilot" target="_blank"><img src="https://api.producthunt.com/widgets/embed-image/v1/featured.svg?post_id=601490&theme=light" alt="EchoAI&#0032;Interview&#0032;Copilot&#0032; - Real&#0045;time&#0032;conversation&#0032;with&#0032;LLM&#0032;responses | Product Hunt" style="width: 250px; height: 54px;" width="250" height="54" /></a>
54
+ </p>
55
+
56
+ <p align="center">
57
+ <img width="800" alt="EchoAI Helper Interface" src="https://github.com/colakang/echoai_helper/raw/main/docs/images/ui.png">
58
+ </p>
59
+
60
+ ---
61
+
62
+ ## ✨ What it does
63
+
64
+ - **Local speech recognition** — FunASR / SenseVoice, on-device, GPU-accelerated
65
+ where one is available (Metal on Apple Silicon, CUDA on Windows)
66
+ - **Automatic language detection** — Mandarin, Cantonese, English, Japanese and
67
+ Korean, switching per utterance, including mid-sentence code-switching
68
+ - **Both sides of the call** — your microphone and the far end, on separate tracks
69
+ - **Speaker labelling** — voices on the far-end track are told apart, with the
70
+ number of people configurable when you know it
71
+ - **Pause-based segmentation** — sentences are cut where people actually pause,
72
+ not on a fixed timer, so the model sees whole utterances
73
+ - **Crash-safe recording** — every settled sentence is written to disk as it is
74
+ produced; a crash costs the last line, not the meeting
75
+ - **LLM cleanup on export** — an optional pass that fixes mis-heard words using
76
+ the surrounding conversation, through an API key or through a coding-agent CLI
77
+ on a subscription you already pay for
78
+ - **Markdown and JSON export** — Markdown to read, JSON as a complete record
79
+ - **Live reply suggestions** — for interviews, where a prompt is wanted while the
80
+ other person is still talking
81
+
82
+ ## 💡 Two modes
83
+
84
+ | | Meeting notes | Live interview |
85
+ |---|---|---|
86
+ | Text appears | at each pause | as you speak (~0.6s) |
87
+ | Model calls | one per sentence | roughly three times as many |
88
+ | Best for | an accurate record | a prompt you can act on |
89
+
90
+ Switch in the app; the several settings that differ move together.
91
+
92
+ ## 🎬 Demo
93
+
94
+ https://github.com/user-attachments/assets/0d627e4a-960b-4628-8bbc-8d892f02cfd1
95
+
96
+ ---
97
+
98
+ ## ⚡ Install
99
+
100
+ ### macOS
101
+
102
+ ```bash
103
+ uv tool install echoai-helper # uv brings its own Python 3.12
104
+ echoai-helper setup # audio routing — one password prompt
105
+ echoai-helper install-launcher # adds an icon to Launchpad
106
+ ```
107
+
108
+ Launch from Launchpad, or run `echoai-helper`.
109
+
110
+ `setup` installs a virtual audio device and builds the Multi-Output that lets
111
+ you hear a meeting while it is being recorded. macOS asks for a password once,
112
+ because that installs an audio driver — nothing else needs a privilege, and
113
+ Audio MIDI Setup is not involved.
114
+
115
+ The app takes the audio output while it runs and gives it back silently when it
116
+ quits, including after a crash. `echoai-helper setup --restore` does it by hand;
117
+ `--status` shows what routing is in place.
118
+
119
+ <details>
120
+ <summary>Prefer Homebrew?</summary>
121
+
122
+ A formula is in [`packaging/`](packaging/echoai-helper.rb) for a tap. Homebrew
123
+ can declare the virtual audio device as a dependency, which removes the one step
124
+ that needs a password.
125
+ </details>
126
+
127
+ ### Windows
128
+
129
+ ```powershell
130
+ uv tool install echoai-helper
131
+ echoai-helper
132
+ ```
133
+
134
+ No audio setup step: Windows exposes WASAPI loopback directly, so the far end of
135
+ a call is capturable without a virtual device. The macOS-only commands (`setup`,
136
+ `install-launcher`) report that there is nothing to do. See
137
+ [known limitations](#-known-limitations) — this path has not been re-tested since
138
+ the segmentation rework.
139
+
140
+ ### First run
141
+
142
+ Speech models (~1.5GB) download on first launch. Nothing else is needed.
143
+
144
+ <details>
145
+ <summary>Why Python 3.12 specifically</summary>
146
+
147
+ Not caution. The vendored `src/custom_speech_recognition` imports `aifc` and
148
+ `audioop` at module level, and **both were removed from the standard library in
149
+ Python 3.13**. `uv` installs a suitable interpreter itself, and its 3.12 build
150
+ ships tkinter, so there is no separate Tk step.
151
+ </details>
152
+
153
+ ---
154
+
155
+ ## 🎯 Using it
156
+
157
+ 1. Open the app. If audio routing is not in place, it offers to finish it.
158
+ 2. Pick a mode, and set the number of people if you know it.
159
+ 3. Hold your meeting. Nothing needs touching.
160
+ 4. **Export** — one dialog covers format, cleanup, which model to clean with,
161
+ and merging over-split speakers.
162
+
163
+ Cleanup runs in the background with a progress bar and an estimate, and can be
164
+ stopped: whatever finished is kept.
165
+
166
+ ### Choosing a cleanup backend
167
+
168
+ | | Cost | Speed (measured) |
169
+ |---|---|---|
170
+ | **API** (`conf.yaml`) | per token | 5–8s per batch of lines |
171
+ | **Claude CLI** | included in a subscription | 20–50s per batch |
172
+
173
+ Both are offered at export, and the dialog turns the per-batch figure into an
174
+ estimate for the transcript in hand. The live reply suggestions always use the
175
+ configured API — a CLI takes seconds per answer, which is too late to be useful
176
+ while someone is still talking.
177
+
178
+ ### Past recordings
179
+
180
+ ```bash
181
+ echoai-helper sessions # list them
182
+ echoai-helper sessions --export 0 # export one again
183
+ echoai-helper sessions --delete 0
184
+ ```
185
+
186
+ Re-exporting is the point: a different format, another pass of cleanup, a
187
+ different number of speakers, without re-recording anything. An unfinished
188
+ session from the last 12 hours is offered on the next launch.
189
+
190
+ ---
191
+
192
+ ## ⚙️ Configuration
193
+
194
+ Model settings live in `conf.yaml`; everything else is in the app. Installed
195
+ from a wheel the shipped copy sits inside the package, so make yourself an
196
+ editable one:
197
+
198
+ ```bash
199
+ echoai-helper config # writes conf.yaml where you can reach it, and prints the path
200
+ ```
201
+
202
+ That copy overrides the defaults. You will not usually need it — both values
203
+ below already ship as `auto`.
204
+
205
+ ```yaml
206
+ FunASR:
207
+ model_name: "iic/SenseVoiceSmall"
208
+ device: "auto" # cuda, then mps, then cpu
209
+ language: "auto" # zh, en, yue, ja, ko
210
+
211
+ LLM:
212
+ provider: "openai" # openai | litellm | cli
213
+ ```
214
+
215
+ Both `auto` values are load-bearing rather than lazy defaults:
216
+
217
+ - **`device: "auto"`** — measured on an M4, dual-track real-time factor is 2.04
218
+ on cpu (falling behind twice over) against 0.35 on Metal. Landing on cpu by
219
+ accident means transcription that cannot keep up. Naming a device explicitly is
220
+ also wrong on every machine that does not have it, and this file travels.
221
+ - **`language: "auto"`** — pinning a language does not bias the model, it forces
222
+ the syllables onto words of that language. A Cantonese call transcribed with
223
+ `language: "en"` comes back as fluent nonsense.
224
+
225
+ An OpenAI key goes in `.env` (see `.env.example`), or in a file called `.llm`
226
+ holding nothing else. Both are gitignored.
227
+
228
+ ---
229
+
230
+ ## 🔍 Troubleshooting
231
+
232
+ ```bash
233
+ echoai-helper check-audio # what is being captured, and from where
234
+ echoai-helper setup --status # what routing is in place
235
+ ```
236
+
237
+ **Nothing from the far end (macOS).** The meeting app has its own audio settings
238
+ and remembers them. Set its speaker to `EchoAI Meeting`.
239
+
240
+ **Nothing from the microphone over Bluetooth.** A Bluetooth headset can only send
241
+ its microphone to one device. If it is on a phone call, the Mac gets silence —
242
+ and the stream does not recover on its own; restart the app. A USB microphone
243
+ avoids this entirely.
244
+
245
+ **More speakers than people.** Voice prints drift with volume and connection
246
+ quality, so one person can end up split across several labels. Set the number of
247
+ people before the meeting, or merge them at export — the export carries the voice
248
+ prints, so this works after the fact.
249
+
250
+ ---
251
+
252
+ ## 🛠️ Development
253
+
254
+ ```bash
255
+ git clone https://github.com/colakang/echoai_helper.git
256
+ cd echoai_helper
257
+ uv venv --python 3.12 .venv
258
+ uv pip install --python .venv/bin/python -r requirements-macos.txt # or requirements.txt
259
+ .venv/bin/python -m pytest tests/ -q
260
+ ```
261
+
262
+ [`docs/macos-audio-setup.md`](docs/macos-audio-setup.md) covers the audio
263
+ routing, what has been measured, and where the sharp edges are.
264
+
265
+ ---
266
+
267
+ ## 📝 Known limitations
268
+
269
+ - **A dead microphone stream does not recover.** Seen on a real call: capture
270
+ stops silently and the app looks like it is still working. Restarting fixes it.
271
+ This is the most consequential item on the list.
272
+ - **Speaker labelling is tuned against a clean two-party recording** and
273
+ over-splits on group calls over a lossy connection. Merging at export is the
274
+ workaround, not the cure.
275
+ - **The Windows capture path is untested** since the segmentation rework. Its
276
+ detector threshold was lowered to pass silence through, because segmentation
277
+ now happens on pauses and a recogniser that only reports speech never delivers
278
+ them — reasoned, not verified. Reports welcome.
279
+ - **macOS audio routing is a shared setting.** Selecting the Multi-Output changes
280
+ the output for every app, and macOS sometimes moves it back after sleep.
281
+ Checked at every launch.
282
+
283
+ ## 🤝 Contributing
284
+
285
+ Pull requests welcome; for anything substantial, open an issue first. See
286
+ [CONTRIBUTING.md](CONTRIBUTING.md).
287
+
288
+ ## 🙌 Credits
289
+
290
+ - [FunASR](https://github.com/modelscope/FunASR) — speech recognition and speaker
291
+ embeddings
292
+ - [silero-vad](https://github.com/snakers4/silero-vad) — voice activity detection
293
+ - [WhisperLiveKit](https://github.com/QuentinFuxa/WhisperLiveKit) — the
294
+ LocalAgreement idea behind stable partial transcripts
295
+ - [BlackHole](https://existential.audio/blackhole/) — virtual audio device on macOS
296
+ - [CustomTkinter](https://github.com/TomSchimansky/CustomTkinter) — interface
297
+ - [Ecoute](https://github.com/SevaSk/ecoute) — the original inspiration
298
+ - [@zixing0131](https://github.com/zixing0131) — core audio processing
299
+
300
+ ## 📞 Contact
301
+
302
+ [echo365.ai](https://www.echo365.ai) · [Issues](https://github.com/colakang/echoai_helper/issues)
303
+
304
+ ## 📄 License
305
+
306
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,67 @@
1
+ echoai_helper-1.2.0.dist-info/licenses/LICENSE,sha256=GENvlKmKL1xWmXu4yDRd4C0gU-lxMCA9Z1L_iNqeMSw,1066
2
+ scripts/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
3
+ scripts/calibrate_vad.py,sha256=_erRGc6aqCUSI-_BzdeYIko18svt-E2KxVtF2Q3Fauc,6622
4
+ scripts/check_audio.py,sha256=rwD6U0aNnDIa40_7w83v9PgCzjn7MT2Q613KptbolUY,10858
5
+ scripts/install_launcher.py,sha256=yJaI1ycXjTjhTK6sCS7gDkruia1nKkmCtP89KeClxBQ,1788
6
+ scripts/setup_audio.py,sha256=AKhN56YimIYpXTi5KyqrxSvIkR87Zb_PT_L6_oGcULY,2284
7
+ src/AudioRecorder.py,sha256=3g4NEehNZLCYL2WdP_OUfyhl5KHTB3ibQAVgQez_sy4,2705
8
+ src/AudioTranscriber.py,sha256=dPf6CewhJbZJ8qQwPFhNOLcQQhHMYcKRGejUcmhhNmo,23774
9
+ src/GPTResponder.py,sha256=0VYEw9YU7THfXxSd7zI88zjyiKJjlpEPsVi5TAYel4o,11872
10
+ src/ResponseManager.py,sha256=Db7O261PDbvCmElw_UUC0QWjPMRFBtjzMj2yNOWF3_E,15729
11
+ src/SettingsManager.py,sha256=VEtbIlvDVAFFMF_qZSU89RAMlz4og-WOEVRIKvT82lM,5356
12
+ src/TemplateManager.py,sha256=4cquOIaZvb5WoH8fiwOisYa9U_q7k4LApPg3wHoGMR8,7981
13
+ src/TranscriberModels.py,sha256=qY9D-16BINGM0Ln-O6SInrX4En_KWohJfQ0WjgIZNUA,5873
14
+ src/TranscriptUI.py,sha256=mU7SJw_8upHYoVrp2jvVN4lNrkIaspB_U4PCD_rRUCI,22921
15
+ src/app.py,sha256=Ca9lwjQYFJZ8cUrvHaISFbHke08vYsGGmsyNkdPqrI0,41005
16
+ src/cli.py,sha256=VcRg6jKO6Y3HEFyHwKK_aYD1YM2BUfCDRd5sP4OKKNc,5482
17
+ src/conf.yaml,sha256=iHTFARU1Mk6vxZD7NnGSUD_zAo2bY-I9JGPYYkr4f1Y,5081
18
+ src/config.py,sha256=mxw4v-i7MSwEcqLTL9Qwv78ypVK9nF9LXIXt6H6szV0,10445
19
+ src/export_dialog.py,sha256=AuFqln9iWgGbomXLcvATMiiqzVzBX1RizBpG12U6yBw,14037
20
+ src/export_markdown.py,sha256=-nbylYUu-S3bVLdM8xD1SKkmMVn78IUdpNQfchqkhms,6277
21
+ src/launcher_macos.py,sha256=99gH39JIxup8_t2AYVeCSnKLon5djhKb4e_Avbg4DDA,6020
22
+ src/polish.py,sha256=iuheB_i0MwEjpjz5a_qRQCu1tNllkDwJL4Cdg143-nM,9275
23
+ src/profiles.py,sha256=qSZ32T9Pop3vDHUgTavLimCpyj9uqDWQFIQoeQslw2g,3357
24
+ src/prompts.py,sha256=tVF7zH5PIGfuTzOPllHUC_5NwgJhKiq3ANtdiv_SJkQ,3657
25
+ src/session.py,sha256=kspPbn14Cu3D-ZYR6iVf4-xmKMGFeoodGvVyRtIihlg,10717
26
+ src/asr/asr_factory.py,sha256=zygmh9G_QVOdbBU3owzKTXQgqnkYcUbcVL5wpGKn_ZM,1877
27
+ src/asr/asr_interface.py,sha256=pRv_ZSbv1oYMpnwoH9_6-anl8g0-tQIJ5YPmSEpBIDU,1022
28
+ src/asr/asr_with_vad.py,sha256=5vbnrKpefTtgayhT87vw5oI6bQ93MI8c92CadW9IiZ8,10251
29
+ src/asr/azure_asr.py,sha256=te6CoB9iHwDzLqnPhYOP6xpyL9XZcHhWF5NU7z4hupA,3290
30
+ src/asr/diarization.py,sha256=_VJzWoFm5HdzYSgbrR9r6RhAlDOGSvgcEBMuv0Y-Wgo,9636
31
+ src/asr/faster_whisper_asr.py,sha256=a3myeI6dguNIcCxw-SPrVCp7Jj-NX8jh_lmHR4e7ZcU,1288
32
+ src/asr/fun_asr.py,sha256=kh566fUHC-VtqLOafGHan2072NV_7jHBhklG9yhNJgE,3678
33
+ src/asr/hypothesis.py,sha256=GcNiho_-t5M5UUJXnZoi8yABex0nrePUP_466Y0PR7g,6532
34
+ src/asr/openai_whisper_asr.py,sha256=6sf-xaM38Zpxy4hP1D0JlEQ292zClGfQRgd7ICA2KMY,903
35
+ src/asr/segmenter.py,sha256=4B_yga7a8SNbGxTRdp8JGYY6A8KiT6JZyCvnZ06_qtU,11779
36
+ src/asr/vad.py,sha256=hq6gR0CYTgir8e824hkgI955VOVXVzkSO5hTbwT7p8M,1921
37
+ src/asr/whisper_cpp_asr.py,sha256=CI7pwQ6Pqfb1PLo7WIeT0lt0C1cZSzBIaltT-2DR4fk,1150
38
+ src/asr/models/silero_vad.onnx,sha256=o16_Uv085fFGmyo2FY26dhvEe5c-ozgrMYbKFbH1ryg,1807522
39
+ src/audio/__init__.py,sha256=CWzsEu7UBUhom60q3zY_scFxgXuLXD_BNZmAgwqy9dw,1175
40
+ src/audio/backend.py,sha256=PxApwHfyvxRAJx62VwWi3KN1uFPlpRyYH6EBUa-tn7c,4106
41
+ src/audio/coreaudio.py,sha256=pSG30wWDCqP7UrmyUdFQ3hxxKIrKP1ueJB_aP1tN01k,10951
42
+ src/audio/macos.py,sha256=G8GtUPvTsxtkGgtA4beYHnU_aDbZdZ4gfofLaVeh8UQ,6707
43
+ src/audio/setup_macos.py,sha256=pjwK6dXQ4r-GeDl2xS2PW4bFbQ83ve3nQ4DM2DQpb_E,13993
44
+ src/audio/windows.py,sha256=s4MsX1I8XRkmWT7_NnY2aRB7WofgWlCSyuN9WfzGWPQ,4194
45
+ src/custom_speech_recognition/__init__.py,sha256=k6-RB8BivP6i3FJA0ZTKSzMYKtlI6tRt5wMDHgKAKV4,92343
46
+ src/custom_speech_recognition/__main__.py,sha256=Afbp3l1yq2AW2Bsm7YBz-O0wm6gHNudgPZh69Ijjubc,833
47
+ src/custom_speech_recognition/audio.py,sha256=THoN6uaHcbbwcvhW6vzkBpAZkpP8s-ouClba48KbOkw,14824
48
+ src/custom_speech_recognition/exceptions.py,sha256=LnKmutOVQ6RSMcSO1Ji5Af-UZN-Aqrq4petMgBFXkKs,273
49
+ src/llm/__init__.py,sha256=NxnYypZQbgQGTISvHVPnY6epbiwDoCbwtpKoEjuOtvI,107
50
+ src/llm/cli_provider.py,sha256=3L1tPbtjoyXv4QsgRZVdtRUDE2rI2_P7FQeU_3Jrb6k,7193
51
+ src/llm/litellm_provider.py,sha256=ynhRyyGoJ7YSYnQ6wa3ZmGI1SeV_wN1iRSc58Vlss0c,4815
52
+ src/llm/llm_provider.py,sha256=WtDqYeTymv-qB5M0Ebz7WonQKqrPnA4BfwuhAMtBiPg,1122
53
+ src/llm/openai_provider.py,sha256=9TMKZpK2BDNoJp8GcyXgmgfDEMIe8n7hrp377q4aRQo,2540
54
+ src/llm/provider_factory.py,sha256=n1weyKvWF9JnZ5sHOnh1p-6_aq5cyKtoZpM2bQ_xsQ0,2941
55
+ src/resources/config/settings.json,sha256=QKvOifkogZpb_owPEMAve7r7SLbkXiih89nazuCWauQ,306
56
+ src/resources/prompt/Interview Analysis.md,sha256=stVlPDLrXYPbO9Wg3ex73rnIYQrfSCSLN2uE_zS-oLs,2885
57
+ src/resources/prompt/Meeting Analysis.md,sha256=_YdMLJ_jCclxrDOHHsl5EzYa4XqiK5QBhKMut7SngvQ,7976
58
+ src/resources/prompt/case_detail/inbound_cs.txt,sha256=JJhU8OCCXOTOVAjmJjrU8s5oNb5zsE7AVX6Nebh0he0,1180
59
+ src/resources/prompt/case_detail/none.txt,sha256=AO0frMCzJXR3KHDQWPu-WnUFUYnuKiQKQbHvfPRfgmc,8
60
+ src/resources/prompt/knowledge/none.txt,sha256=6RcVRI85-y-CqHADDxMdwYn67HD6JxyfVme0mDKPdQU,7
61
+ src/resources/prompt/system_role/inbound_cs.py,sha256=5kMRdhhuTDIbJ2awrVhqVHoRzlIHSQMP2u49dOHDQDM,1235
62
+ src/resources/prompt/system_role/phone_interview.py,sha256=vf-OB-NjSRtxYEc_eH08DyBQwtxSEY8Xha-n7FQnWng,1194
63
+ echoai_helper-1.2.0.dist-info/METADATA,sha256=v590S7f0e0wO0VGTRUVy7olaoqsFVVZwtFxzlD8y9q8,11890
64
+ echoai_helper-1.2.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
65
+ echoai_helper-1.2.0.dist-info/entry_points.txt,sha256=HNWAt73pLEsWFen4F5bVbGyOHOoJJQ-yL_OD0cwCLp8,47
66
+ echoai_helper-1.2.0.dist-info/top_level.txt,sha256=2u1B708VEAisYx1t_BY-Ya_-RVnGzeibiEH429s02AM,12
67
+ echoai_helper-1.2.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ echoai-helper = src.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 Cola Kang
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,2 @@
1
+ scripts
2
+ src
scripts/__init__.py ADDED
File without changes
@@ -0,0 +1,183 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ scripts/calibrate_vad.py — pick VAD thresholds from a real recording.
4
+
5
+ min_silence_ms is the dominant accuracy/latency knob in the pipeline: too
6
+ short splits sentences and denies the model the context it needs to
7
+ punctuate; too long merges separate turns and delays the "sentence finished"
8
+ signal that triggers LLM cleanup.
9
+
10
+ It cannot be set from synthesised speech. `say` produces near-continuous
11
+ audio whose longest pause is ~0.3s, where a real speaker pauses 0.5-2s
12
+ between sentences. Record 30-60s of ordinary talking -- ideally the actual
13
+ meeting or interview setting, with its real background noise -- and run:
14
+
15
+ .venv/bin/python scripts/calibrate_vad.py recording.wav
16
+
17
+ # also transcribe each segment, to see what the model receives
18
+ .venv/bin/python scripts/calibrate_vad.py recording.wav --transcribe
19
+
20
+ Any format ffmpeg reads is fine; it is converted to mono 16kHz internally.
21
+ """
22
+
23
+ import argparse
24
+ import os
25
+ import subprocess
26
+ import sys
27
+ import tempfile
28
+
29
+ import numpy as np
30
+
31
+ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
32
+
33
+ from src.asr.segmenter import ( # noqa: E402
34
+ Event, SegmenterConfig, SpeechSegmenter, SAMPLE_RATE, WINDOW_SAMPLES,
35
+ )
36
+ from src.asr.vad import VAD # noqa: E402
37
+
38
+ VAD_MODEL = os.path.join(
39
+ os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
40
+ "src", "asr", "models", "silero_vad.onnx",
41
+ )
42
+
43
+ CANDIDATE_SILENCE_MS = [200, 300, 400, 500, 700, 1000, 1500]
44
+
45
+
46
+ def load_audio(path: str) -> np.ndarray:
47
+ """Decode anything ffmpeg understands into mono float32 @16kHz."""
48
+ import soundfile as sf
49
+
50
+ with tempfile.TemporaryDirectory() as tmp:
51
+ wav = os.path.join(tmp, "audio.wav")
52
+ subprocess.run(
53
+ ["ffmpeg", "-y", "-loglevel", "error", "-i", path,
54
+ "-ar", str(SAMPLE_RATE), "-ac", "1", "-c:a", "pcm_s16le", wav],
55
+ check=True,
56
+ )
57
+ data, rate = sf.read(wav, dtype="float32")
58
+ assert rate == SAMPLE_RATE
59
+ return data
60
+
61
+
62
+ def probability_profile(audio: np.ndarray) -> np.ndarray:
63
+ vad = VAD(VAD_MODEL)
64
+ return np.array([
65
+ float(vad.process_chunk(audio[i:i + WINDOW_SAMPLES]))
66
+ for i in range(0, len(audio) - WINDOW_SAMPLES, WINDOW_SAMPLES)
67
+ ])
68
+
69
+
70
+ def describe_pauses(probabilities: np.ndarray, threshold: float) -> None:
71
+ """Where the natural boundaries actually are in this recording."""
72
+ below = probabilities < threshold
73
+ runs, current = [], 0
74
+ for quiet in below:
75
+ if quiet:
76
+ current += 1
77
+ elif current:
78
+ runs.append(current * 100)
79
+ current = 0
80
+ if current:
81
+ runs.append(current * 100)
82
+
83
+ print(f" speech probability: median {np.median(probabilities):.2f} "
84
+ f"above {threshold}: {(~below).mean() * 100:.0f}% of windows")
85
+ if not runs:
86
+ print(" no pauses at all — nothing to segment on")
87
+ return
88
+
89
+ runs.sort()
90
+ print(f" {len(runs)} pauses, longest {max(runs)}ms, median {int(np.median(runs))}ms")
91
+ buckets = [(200, 300), (300, 500), (500, 800), (800, 1500), (1500, 10 ** 9)]
92
+ for low, high in buckets:
93
+ count = sum(1 for r in runs if low <= r < high)
94
+ if count:
95
+ label = f"{low}-{high}ms" if high < 10 ** 9 else f">{low}ms"
96
+ print(f" {label:>12}: {count}")
97
+
98
+
99
+ def segment_with(audio: np.ndarray, config: SegmenterConfig):
100
+ vad = VAD(VAD_MODEL)
101
+ segmenter = SpeechSegmenter(vad, config)
102
+ segments = []
103
+ step = int(0.6 * SAMPLE_RATE) # the recorder's real chunk size
104
+ for i in range(0, len(audio), step):
105
+ for event, segment in segmenter.process(audio[i:i + step]):
106
+ if event is Event.SPEECH_END:
107
+ segments.append(segment)
108
+ tail = segmenter.flush()
109
+ if tail is not None:
110
+ segments.append(tail)
111
+ return segments
112
+
113
+
114
+ def main() -> int:
115
+ parser = argparse.ArgumentParser(description=__doc__,
116
+ formatter_class=argparse.RawDescriptionHelpFormatter)
117
+ parser.add_argument("audio", help="a real recording of ordinary speech")
118
+ parser.add_argument("--transcribe", action="store_true",
119
+ help="transcribe each segment (loads FunASR)")
120
+ parser.add_argument("--speech-threshold", type=float, default=0.6)
121
+ parser.add_argument("--silence-threshold", type=float, default=0.35)
122
+ args = parser.parse_args()
123
+
124
+ if not os.path.exists(args.audio):
125
+ print(f"[FATAL] no such file: {args.audio}")
126
+ return 1
127
+
128
+ audio = load_audio(args.audio)
129
+ duration = len(audio) / SAMPLE_RATE
130
+ print("=" * 70)
131
+ print(f"{os.path.basename(args.audio)} — {duration:.1f}s")
132
+ print("=" * 70)
133
+
134
+ describe_pauses(probability_profile(audio), args.silence_threshold)
135
+ print()
136
+
137
+ print("=" * 70)
138
+ print("SEGMENTATION vs min_silence_ms")
139
+ print("=" * 70)
140
+ print(f" {'min_silence':>12} {'segments':>9} {'median':>8} {'longest':>8} capped")
141
+ results = {}
142
+ for ms in CANDIDATE_SILENCE_MS:
143
+ config = SegmenterConfig(
144
+ speech_threshold=args.speech_threshold,
145
+ silence_threshold=args.silence_threshold,
146
+ min_silence_ms=ms,
147
+ )
148
+ segments = segment_with(audio, config)
149
+ results[ms] = segments
150
+ if not segments:
151
+ print(f" {ms:>10}ms {0:>9}")
152
+ continue
153
+ durations = [s.duration_s for s in segments]
154
+ # Segments at the ceiling were cut by the cap, not by a real pause:
155
+ # a high count means this threshold is not finding boundaries.
156
+ capped = sum(1 for d in durations if d >= config.max_segment_s - 0.5)
157
+ print(f" {ms:>10}ms {len(segments):>9} {np.median(durations):>7.1f}s "
158
+ f"{max(durations):>7.1f}s {capped}")
159
+
160
+ print()
161
+ print(" A good setting gives segments of roughly one sentence (2-10s)")
162
+ print(" with no capped ones. Many capped segments means the threshold is")
163
+ print(" too long and no natural pause is ever reaching it.")
164
+
165
+ if args.transcribe:
166
+ print()
167
+ print("=" * 70)
168
+ print("TRANSCRIPTS")
169
+ print("=" * 70)
170
+ from src.TranscriberModels import get_model
171
+ model = get_model(use_api=False)
172
+ for ms in (300, 500, 700):
173
+ segments = results.get(ms, [])
174
+ print(f"\n--- min_silence={ms}ms, {len(segments)} segments ---")
175
+ for index, segment in enumerate(segments[:8]):
176
+ text = model.get_transcription_np(segment.audio).text
177
+ print(f" [{index}] {segment.duration_s:4.1f}s {text!r}")
178
+
179
+ return 0
180
+
181
+
182
+ if __name__ == "__main__":
183
+ sys.exit(main())