entune 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. entune-0.1.0/LICENSE +21 -0
  2. entune-0.1.0/PKG-INFO +230 -0
  3. entune-0.1.0/README.md +202 -0
  4. entune-0.1.0/pyproject.toml +96 -0
  5. entune-0.1.0/pyproject.toml.orig +68 -0
  6. entune-0.1.0/src/entune/__init__.py +7 -0
  7. entune-0.1.0/src/entune/__main__.py +3 -0
  8. entune-0.1.0/src/entune/api/__init__.py +0 -0
  9. entune-0.1.0/src/entune/api/common.py +34 -0
  10. entune-0.1.0/src/entune/api/data.py +125 -0
  11. entune-0.1.0/src/entune/api/desktop.py +65 -0
  12. entune-0.1.0/src/entune/api/dictionary.py +93 -0
  13. entune-0.1.0/src/entune/api/learning.py +179 -0
  14. entune-0.1.0/src/entune/api/models.py +47 -0
  15. entune-0.1.0/src/entune/api/recordings.py +130 -0
  16. entune-0.1.0/src/entune/api/settings.py +252 -0
  17. entune-0.1.0/src/entune/app/__init__.py +0 -0
  18. entune-0.1.0/src/entune/app/audio_import.py +183 -0
  19. entune-0.1.0/src/entune/app/capture.py +61 -0
  20. entune-0.1.0/src/entune/app/decision_models.py +35 -0
  21. entune-0.1.0/src/entune/app/desktop_bridge.py +40 -0
  22. entune-0.1.0/src/entune/app/dictation.py +402 -0
  23. entune-0.1.0/src/entune/app/dictionary_file.py +101 -0
  24. entune-0.1.0/src/entune/app/entune.py +123 -0
  25. entune-0.1.0/src/entune/app/learning.py +212 -0
  26. entune-0.1.0/src/entune/app/local_data.py +109 -0
  27. entune-0.1.0/src/entune/app/metrics.py +170 -0
  28. entune-0.1.0/src/entune/app/models.py +136 -0
  29. entune-0.1.0/src/entune/app/operations.py +125 -0
  30. entune-0.1.0/src/entune/app/settings.py +282 -0
  31. entune-0.1.0/src/entune/app/shortcuts.py +112 -0
  32. entune-0.1.0/src/entune/app/suggestion_runs.py +421 -0
  33. entune-0.1.0/src/entune/assets/Entune.icns +0 -0
  34. entune-0.1.0/src/entune/assets/icon-512.png +0 -0
  35. entune-0.1.0/src/entune/assets/icon.png +0 -0
  36. entune-0.1.0/src/entune/assets/menubar-template.png +0 -0
  37. entune-0.1.0/src/entune/audio/__init__.py +0 -0
  38. entune-0.1.0/src/entune/audio/convert.py +24 -0
  39. entune-0.1.0/src/entune/audio/formats.py +166 -0
  40. entune-0.1.0/src/entune/audio/recorder.py +132 -0
  41. entune-0.1.0/src/entune/cli.py +233 -0
  42. entune-0.1.0/src/entune/desktop/__init__.py +0 -0
  43. entune-0.1.0/src/entune/desktop/app.py +545 -0
  44. entune-0.1.0/src/entune/desktop/engine.py +105 -0
  45. entune-0.1.0/src/entune/desktop/macos/__init__.py +0 -0
  46. entune-0.1.0/src/entune/desktop/macos/actions.py +174 -0
  47. entune-0.1.0/src/entune/desktop/macos/adapters.py +49 -0
  48. entune-0.1.0/src/entune/desktop/macos/bundle.py +205 -0
  49. entune-0.1.0/src/entune/desktop/macos/hotkeys.py +244 -0
  50. entune-0.1.0/src/entune/desktop/macos/indicator.py +119 -0
  51. entune-0.1.0/src/entune/desktop/macos/permissions.py +61 -0
  52. entune-0.1.0/src/entune/desktop/macos/webview.py +172 -0
  53. entune-0.1.0/src/entune/desktop/platform.py +117 -0
  54. entune-0.1.0/src/entune/desktop/webview/__init__.py +0 -0
  55. entune-0.1.0/src/entune/desktop/webview/shell.py +348 -0
  56. entune-0.1.0/src/entune/dictionary/__init__.py +0 -0
  57. entune-0.1.0/src/entune/dictionary/changes.py +167 -0
  58. entune-0.1.0/src/entune/dictionary/corrections.py +110 -0
  59. entune-0.1.0/src/entune/dictionary/document.py +207 -0
  60. entune-0.1.0/src/entune/dictionary/entries.py +161 -0
  61. entune-0.1.0/src/entune/dictionary/matching.py +201 -0
  62. entune-0.1.0/src/entune/learning/__init__.py +0 -0
  63. entune-0.1.0/src/entune/learning/batches.py +185 -0
  64. entune-0.1.0/src/entune/learning/generate.py +175 -0
  65. entune-0.1.0/src/entune/learning/inputs.py +33 -0
  66. entune-0.1.0/src/entune/learning/replies.py +331 -0
  67. entune-0.1.0/src/entune/learning/suggestion_model/__init__.py +30 -0
  68. entune-0.1.0/src/entune/learning/suggestion_model/call.py +179 -0
  69. entune-0.1.0/src/entune/learning/suggestion_model/catalog.py +75 -0
  70. entune-0.1.0/src/entune/learning/suggestion_model/chatgpt.py +175 -0
  71. entune-0.1.0/src/entune/learning/suggestion_model/providers.py +148 -0
  72. entune-0.1.0/src/entune/learning/suggestion_model/request.py +60 -0
  73. entune-0.1.0/src/entune/processing/__init__.py +0 -0
  74. entune-0.1.0/src/entune/processing/cleanup.py +44 -0
  75. entune-0.1.0/src/entune/processing/formatting.py +83 -0
  76. entune-0.1.0/src/entune/processing/jev.py +254 -0
  77. entune-0.1.0/src/entune/processing/jev_client.py +354 -0
  78. entune-0.1.0/src/entune/processing/laya.py +238 -0
  79. entune-0.1.0/src/entune/processing/pipeline.py +164 -0
  80. entune-0.1.0/src/entune/processing/results.py +120 -0
  81. entune-0.1.0/src/entune/processing/text_edits.py +47 -0
  82. entune-0.1.0/src/entune/prompts/__init__.py +50 -0
  83. entune-0.1.0/src/entune/prompts/dictionary-foundation.txt +87 -0
  84. entune-0.1.0/src/entune/prompts/dictionary-generate-system.txt +70 -0
  85. entune-0.1.0/src/entune/prompts/dictionary-generate-user.txt +7 -0
  86. entune-0.1.0/src/entune/prompts/dictionary-refine-system.txt +152 -0
  87. entune-0.1.0/src/entune/prompts/dictionary-refine-user.txt +7 -0
  88. entune-0.1.0/src/entune/prompts/jev-cleanup.json +18 -0
  89. entune-0.1.0/src/entune/prompts/jev-formatting.json +30 -0
  90. entune-0.1.0/src/entune/prompts/jev-meaning-examples.json +8 -0
  91. entune-0.1.0/src/entune/prompts/jev-meaning-option.txt +1 -0
  92. entune-0.1.0/src/entune/prompts/jev-meaning-reading.txt +1 -0
  93. entune-0.1.0/src/entune/prompts/jev-meaning.json +8 -0
  94. entune-0.1.0/src/entune/providers/__init__.py +1 -0
  95. entune-0.1.0/src/entune/providers/cloud/__init__.py +1 -0
  96. entune-0.1.0/src/entune/providers/cloud/assemblyai.py +245 -0
  97. entune-0.1.0/src/entune/providers/cloud/contracts.py +34 -0
  98. entune-0.1.0/src/entune/providers/cloud/elevenlabs.py +45 -0
  99. entune-0.1.0/src/entune/providers/cloud/groq.py +45 -0
  100. entune-0.1.0/src/entune/providers/cloud/http.py +58 -0
  101. entune-0.1.0/src/entune/providers/cloud/soniox.py +118 -0
  102. entune-0.1.0/src/entune/providers/cloud/xai.py +45 -0
  103. entune-0.1.0/src/entune/providers/contracts.py +61 -0
  104. entune-0.1.0/src/entune/providers/local/__init__.py +1 -0
  105. entune-0.1.0/src/entune/providers/local/contracts.py +36 -0
  106. entune-0.1.0/src/entune/providers/local/downloads.py +107 -0
  107. entune-0.1.0/src/entune/providers/local/parakeet.py +256 -0
  108. entune-0.1.0/src/entune/providers/local/parakeet_helper.py +119 -0
  109. entune-0.1.0/src/entune/providers/local/whisper.py +262 -0
  110. entune-0.1.0/src/entune/providers/registry.py +57 -0
  111. entune-0.1.0/src/entune/providers/resources.py +169 -0
  112. entune-0.1.0/src/entune/py.typed +0 -0
  113. entune-0.1.0/src/entune/server.py +114 -0
  114. entune-0.1.0/src/entune/storage/__init__.py +0 -0
  115. entune-0.1.0/src/entune/storage/data_folder.py +52 -0
  116. entune-0.1.0/src/entune/storage/paths.py +34 -0
  117. entune-0.1.0/src/entune/storage/records.py +93 -0
  118. entune-0.1.0/src/entune/storage/schema.py +66 -0
  119. entune-0.1.0/src/entune/storage/store.py +529 -0
  120. entune-0.1.0/src/entune/web/app.js +367 -0
  121. entune-0.1.0/src/entune/web/audio-onboarding.js +265 -0
  122. entune-0.1.0/src/entune/web/dictionary-build.js +141 -0
  123. entune-0.1.0/src/entune/web/dictionary-view.js +1215 -0
  124. entune-0.1.0/src/entune/web/favicon.svg +1 -0
  125. entune-0.1.0/src/entune/web/history-card.js +316 -0
  126. entune-0.1.0/src/entune/web/history.js +96 -0
  127. entune-0.1.0/src/entune/web/index.html +650 -0
  128. entune-0.1.0/src/entune/web/permissions-view.js +79 -0
  129. entune-0.1.0/src/entune/web/recording.js +161 -0
  130. entune-0.1.0/src/entune/web/settings-view.js +686 -0
  131. entune-0.1.0/src/entune/web/style.css +638 -0
  132. entune-0.1.0/src/entune/web/tokens.css +165 -0
  133. entune-0.1.0/src/entune/web/ui.js +83 -0
entune-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Elias Andualem
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
entune-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,230 @@
1
+ Metadata-Version: 2.4
2
+ Name: entune
3
+ Version: 0.1.0
4
+ Summary: Dictation with your choice of speech model and a personal dictionary that selects corrections in context.
5
+ License-Expression: MIT
6
+ License-File: LICENSE
7
+ Requires-Dist: httpx>=0.27
8
+ Requires-Dist: httpx2
9
+ Requires-Dist: numpy
10
+ Requires-Dist: pillow
11
+ Requires-Dist: pydantic>=2
12
+ Requires-Dist: pydantic-ai-slim[anthropic,google,groq,mistral,openai]>=2.48
13
+ Requires-Dist: pynput>=1.7,<2 ; sys_platform == 'darwin'
14
+ Requires-Dist: pyobjc-framework-applicationservices>=10 ; sys_platform == 'darwin'
15
+ Requires-Dist: pyobjc-framework-avfoundation>=10 ; sys_platform == 'darwin'
16
+ Requires-Dist: pystray>=0.19
17
+ Requires-Dist: python-multipart>=0.0.12
18
+ Requires-Dist: pywebview>=5
19
+ Requires-Dist: pywhispercpp>=1.5
20
+ Requires-Dist: sounddevice>=0.5 ; sys_platform == 'darwin'
21
+ Requires-Dist: starlette>=0.41
22
+ Requires-Dist: uvicorn>=0.30
23
+ Requires-Python: >=3.12
24
+ Project-URL: Homepage, https://github.com/eandualem/entune
25
+ Project-URL: Documentation, https://github.com/eandualem/entune#readme
26
+ Project-URL: Issues, https://github.com/eandualem/entune/issues
27
+ Description-Content-Type: text/markdown
28
+
29
+ <h1>
30
+ <picture>
31
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/eandualem/entune/main/docs/brand/entune-logo-dark.svg" />
32
+ <img src="https://raw.githubusercontent.com/eandualem/entune/main/docs/brand/entune-logo-light.svg" alt="Entune" height="64" />
33
+ </picture>
34
+ </h1>
35
+
36
+ [![CI](https://github.com/eandualem/entune/actions/workflows/ci.yml/badge.svg?branch=develop)](https://github.com/eandualem/entune/actions/workflows/ci.yml)
37
+
38
+ **Your speech model. Your vocabulary. Corrections that consider the context.**
39
+
40
+ Entune is an open-source, cross-platform dictation app. Record in your browser,
41
+ choose a cloud or local speech model, and keep your recordings on your computer.
42
+ On macOS, you can also dictate into other apps with a global shortcut.
43
+ Its personal dictionary learns from your dictation; a decision model chooses
44
+ when a dictionary replacement actually fits the sentence.
45
+
46
+ ## Install
47
+
48
+ **Before the first PyPI release is published**, use the source-install option
49
+ below. The standard commands install the published version:
50
+
51
+ With [uv](https://docs.astral.sh/uv/getting-started/installation/):
52
+
53
+ ```sh
54
+ uv tool install entune
55
+ entune
56
+ ```
57
+
58
+ Or use pip in a Python 3.12+ environment:
59
+
60
+ ```sh
61
+ python -m pip install entune
62
+ entune
63
+ ```
64
+
65
+ <details>
66
+ <summary>Install from source before the first release, or try development changes</summary>
67
+
68
+ This option requires [Git](https://git-scm.com/downloads):
69
+
70
+ ```sh
71
+ uv tool install "git+https://github.com/eandualem/entune.git@develop"
72
+ entune
73
+ ```
74
+
75
+ With pip, use `python -m pip install "git+https://github.com/eandualem/entune.git@develop"`.
76
+
77
+ </details>
78
+
79
+ **Start dictating:**
80
+
81
+ 1. In **Models**, add a speech provider's API key, or install a local model.
82
+ Choose it as your default. AssemblyAI is a straightforward cloud starting
83
+ point; Parakeet is our local recommendation on Apple Silicon.
84
+ 2. Click **Record** and allow microphone access. Speak, then stop recording
85
+ to transcribe. Your audio and transcript are saved in **History**.
86
+ 3. Copy the transcript into any app. On **macOS**, you can also enable
87
+ Microphone, Input Monitoring and Accessibility in **Settings**, set a
88
+ shortcut, and dictate directly into the focused text field.
89
+
90
+ You can start without a dictionary or decision model and add them later.
91
+
92
+ **macOS permissions:** when launched from a terminal, permission entries may
93
+ belong to the terminal or Python. For a named `Entune.app`, use the
94
+ [standalone app installation](https://github.com/eandualem/entune/blob/develop/docs/guide.md#install).
95
+ It requires a source build; a notarized app download is not available.
96
+ The [permission guide](https://github.com/eandualem/entune/blob/develop/docs/guide.md#permissions-macos)
97
+ covers missing shortcuts, “1 of 3 allowed,” and keeping permissions across updates.
98
+
99
+ Entune uses browser mode on Windows and Linux; on macOS, use
100
+ `entune --no-menu` to open it in your browser. **Windows and Linux have not yet
101
+ been tested end to end.** Global shortcuts and automatic paste currently
102
+ require macOS.
103
+
104
+ <p align="center">
105
+ <img src="https://raw.githubusercontent.com/eandualem/entune/main/docs/demo/history-light.jpg" width="49%" alt="Entune history: recordings, transcripts, audio playback and retry" />
106
+ <img src="https://raw.githubusercontent.com/eandualem/entune/main/docs/demo/models-dark.jpg" width="49%" alt="Choose your speech models and configure their keys" />
107
+ </p>
108
+
109
+ ## A dictionary match should not always become a replacement
110
+
111
+ A recognizer might write “cloud” when you meant “Claude.” But replacing every
112
+ “cloud” would also damage a sentence about cloud storage. Entune stores both
113
+ meanings and asks a decision model which fits the surrounding words.
114
+
115
+ The decision model **selects from your dictionary**. It does not generate or
116
+ rewrite your dictation. Entune applies the stored spelling you can inspect
117
+ and edit.
118
+
119
+ ### 95% fewer incorrect replacements with Jev in our test
120
+
121
+ We tested a fixed learned dictionary on **56 new Parakeet recordings**, in
122
+ three consecutive batches. These recordings were not used to build the
123
+ dictionary. Here are the combined results:
124
+
125
+ | Method | Replacements made | Correct | Incorrect | Uncertain |
126
+ |---|---:|---:|---:|---:|
127
+ | Replace every dictionary match | 93 | 31 | 61 | 1 |
128
+ | Choose with **Jev** | 34 | 30 | **3** | 1 |
129
+ | Choose with **Laya**, locally | 42 | 22 | **20** | 0 |
130
+
131
+ - **Jev prevented 58 of 61 wrong replacements (95%)**, while keeping
132
+ 30 of the 31 correct replacements.
133
+ - **Laya prevented 41 of 61 wrong replacements (67%)**, while keeping
134
+ 22 of the 31 correct replacements.
135
+
136
+ Both reduced wrong replacements in every batch. Jev retained more valid
137
+ corrections; Laya keeps decision processing on your computer.
138
+
139
+ This is a small, single-user test, judged from text context before the models
140
+ ran—not an overall transcription-accuracy claim. Repeated contractions
141
+ contributed substantially to the result. The comparison applies the first
142
+ available replacement unconditionally; it is not the app's decision-model-off
143
+ setting. Read the [per-batch results and method](https://github.com/eandualem/entune/blob/develop/docs/decision-model-results.md)
144
+ for the denominators, limitations and current Laya input constraints.
145
+
146
+ ## Choose the models that suit you
147
+
148
+ Entune separates three jobs, so you can choose each independently:
149
+
150
+ | Job | Our starting recommendation | When it runs |
151
+ |---|---|---|
152
+ | Turn audio into text | **Parakeet** on Apple Silicon, or **AssemblyAI** in the cloud | After each recording |
153
+ | Build your dictionary | **GPT-6.1 Sol**, medium effort, 24,000-character batches | When you request suggestions |
154
+ | Choose dictionary replacements | **Jev** for the stronger result in our test; **Laya** for local processing | After transcription, when enabled |
155
+
156
+ Speech options also include Groq, Soniox, ElevenLabs, xAI and local Whisper.cpp.
157
+ Cloud services use your own provider accounts and keys; Entune does not sell
158
+ inference credits.
159
+
160
+ **Use your ChatGPT subscription to build the dictionary.** Choose **OpenAI →
161
+ ChatGPT subscription → Sign in with ChatGPT** in dictionary setup. This access
162
+ option does not require an OpenAI API key; your plan's model access and usage
163
+ limits apply. An OpenAI API key is also available as a separate access option.
164
+ ChatGPT sign-in is currently experimental; see the
165
+ [access details](https://github.com/eandualem/entune/blob/develop/docs/models.md#dictionary-generation).
166
+ It covers dictionary generation, not cloud speech recognition or Jev.
167
+
168
+ The [model guide](https://github.com/eandualem/entune/blob/develop/docs/models.md)
169
+ covers exact model IDs, local-engine installation, account access, and the
170
+ limits of our recommendations.
171
+
172
+ ## Teach Entune your vocabulary
173
+
174
+ Use **Dictionary → Suggest new entries** to learn from the selected speech
175
+ model's history. Or choose **Learn from audio** to import recordings from
176
+ another dictation app or an audio folder. Entune transcribes imported audio
177
+ with your chosen speech model, then proposes entries for review.
178
+
179
+ **Dictionary generation takes time.** It runs sequentially in batches, with
180
+ the growing dictionary included in each request. Large histories can take
181
+ minutes to hours; our 540-transcript Sol build took about **2 hours 15 minutes**,
182
+ including recovery from a failed request. Importing audio adds transcription
183
+ time. The audio selector shows a rough estimate as you choose recordings:
184
+ allow about **10–20 minutes per audio hour** with Sol, plus transcription.
185
+ This is separate from the fast decision step on each new dictation.
186
+
187
+ Review the proposed entries before applying them. You can stop a build and
188
+ review completed batches, or retry from its checkpoint. Entune pauses dictation
189
+ and separate dictionary editing while learning or proposal review is active.
190
+
191
+ Learned entries belong to their speech model: a Parakeet dictionary is not
192
+ automatically an AssemblyAI dictionary. Pin entries you deliberately want to
193
+ share. **Suggest improvements** can revise learned entries later; more
194
+ refinement does not guarantee a better dictionary.
195
+
196
+ ## Keep control of your recordings
197
+
198
+ - **History:** replay audio, copy text, inspect processing changes, and retry
199
+ a recording with another speech model. Provider failures remain visible.
200
+ - **macOS shortcuts:** hold to talk or toggle hands-free recording. Cancel without
201
+ pasting; usable captured audio stays available for retry.
202
+ - **Local data:** audio, transcripts, settings and keys stay in Entune's data
203
+ folder. Export or delete them in **Settings → Data & Privacy**.
204
+ - **Optional processing:** dictionary correction, repeated-filler reduction,
205
+ and paragraph/bullet formatting have separate controls.
206
+
207
+ There is no Entune account, telemetry or hosted history. Cloud speech sends
208
+ audio to your chosen provider; dictionary generation sends its selected
209
+ transcripts and dictionary to the chosen language-model provider. **Jev sends
210
+ matched context to TypeSafe even when speech recognition is local.** Laya
211
+ keeps that step local. See [data and privacy](https://github.com/eandualem/entune/blob/develop/docs/guide.md#data-and-privacy).
212
+
213
+ Entune is for a trusted, single-user machine. Its loopback API has no
214
+ authentication: other local processes can read or change data through it.
215
+ Do not expose its port to a network or tunnel.
216
+
217
+ ## Guides and contributing
218
+
219
+ - [Using Entune](https://github.com/eandualem/entune/blob/develop/docs/guide.md): permissions, shortcuts, imports, updates and troubleshooting.
220
+ - [Choosing models](https://github.com/eandualem/entune/blob/develop/docs/models.md): cloud and local setup, recommendations and generation time.
221
+ - [Decision-model results](https://github.com/eandualem/entune/blob/develop/docs/decision-model-results.md): what changed, how it was measured, and limitations.
222
+ - [Dictionary reference](https://github.com/eandualem/entune/blob/develop/docs/dictionary.md): meanings, matching, learning and the file format.
223
+ - [Building Entune.app](https://github.com/eandualem/entune/blob/develop/docs/packaging.md), [local API](https://github.com/eandualem/entune/blob/develop/docs/agents-api.md), and [architecture](https://github.com/eandualem/entune/blob/develop/docs/architecture.md).
224
+
225
+ To work on Entune, clone the repository and run `uv sync`, then `uv run entune`.
226
+ Run `uv run ruff check .`, `uv run ruff format --check .`, `uv run mypy` and
227
+ `uv run pytest` before contributing. Pull requests target `develop`;
228
+ `main` receives reviewed releases. See [CONTRIBUTING.md](https://github.com/eandualem/entune/blob/develop/CONTRIBUTING.md).
229
+
230
+ [MIT licensed](https://github.com/eandualem/entune/blob/develop/LICENSE).
entune-0.1.0/README.md ADDED
@@ -0,0 +1,202 @@
1
+ <h1>
2
+ <picture>
3
+ <source media="(prefers-color-scheme: dark)" srcset="https://raw.githubusercontent.com/eandualem/entune/main/docs/brand/entune-logo-dark.svg" />
4
+ <img src="https://raw.githubusercontent.com/eandualem/entune/main/docs/brand/entune-logo-light.svg" alt="Entune" height="64" />
5
+ </picture>
6
+ </h1>
7
+
8
+ [![CI](https://github.com/eandualem/entune/actions/workflows/ci.yml/badge.svg?branch=develop)](https://github.com/eandualem/entune/actions/workflows/ci.yml)
9
+
10
+ **Your speech model. Your vocabulary. Corrections that consider the context.**
11
+
12
+ Entune is an open-source, cross-platform dictation app. Record in your browser,
13
+ choose a cloud or local speech model, and keep your recordings on your computer.
14
+ On macOS, you can also dictate into other apps with a global shortcut.
15
+ Its personal dictionary learns from your dictation; a decision model chooses
16
+ when a dictionary replacement actually fits the sentence.
17
+
18
+ ## Install
19
+
20
+ **Before the first PyPI release is published**, use the source-install option
21
+ below. The standard commands install the published version:
22
+
23
+ With [uv](https://docs.astral.sh/uv/getting-started/installation/):
24
+
25
+ ```sh
26
+ uv tool install entune
27
+ entune
28
+ ```
29
+
30
+ Or use pip in a Python 3.12+ environment:
31
+
32
+ ```sh
33
+ python -m pip install entune
34
+ entune
35
+ ```
36
+
37
+ <details>
38
+ <summary>Install from source before the first release, or try development changes</summary>
39
+
40
+ This option requires [Git](https://git-scm.com/downloads):
41
+
42
+ ```sh
43
+ uv tool install "git+https://github.com/eandualem/entune.git@develop"
44
+ entune
45
+ ```
46
+
47
+ With pip, use `python -m pip install "git+https://github.com/eandualem/entune.git@develop"`.
48
+
49
+ </details>
50
+
51
+ **Start dictating:**
52
+
53
+ 1. In **Models**, add a speech provider's API key, or install a local model.
54
+ Choose it as your default. AssemblyAI is a straightforward cloud starting
55
+ point; Parakeet is our local recommendation on Apple Silicon.
56
+ 2. Click **Record** and allow microphone access. Speak, then stop recording
57
+ to transcribe. Your audio and transcript are saved in **History**.
58
+ 3. Copy the transcript into any app. On **macOS**, you can also enable
59
+ Microphone, Input Monitoring and Accessibility in **Settings**, set a
60
+ shortcut, and dictate directly into the focused text field.
61
+
62
+ You can start without a dictionary or decision model and add them later.
63
+
64
+ **macOS permissions:** when launched from a terminal, permission entries may
65
+ belong to the terminal or Python. For a named `Entune.app`, use the
66
+ [standalone app installation](https://github.com/eandualem/entune/blob/develop/docs/guide.md#install).
67
+ It requires a source build; a notarized app download is not available.
68
+ The [permission guide](https://github.com/eandualem/entune/blob/develop/docs/guide.md#permissions-macos)
69
+ covers missing shortcuts, “1 of 3 allowed,” and keeping permissions across updates.
70
+
71
+ Entune uses browser mode on Windows and Linux; on macOS, use
72
+ `entune --no-menu` to open it in your browser. **Windows and Linux have not yet
73
+ been tested end to end.** Global shortcuts and automatic paste currently
74
+ require macOS.
75
+
76
+ <p align="center">
77
+ <img src="https://raw.githubusercontent.com/eandualem/entune/main/docs/demo/history-light.jpg" width="49%" alt="Entune history: recordings, transcripts, audio playback and retry" />
78
+ <img src="https://raw.githubusercontent.com/eandualem/entune/main/docs/demo/models-dark.jpg" width="49%" alt="Choose your speech models and configure their keys" />
79
+ </p>
80
+
81
+ ## A dictionary match should not always become a replacement
82
+
83
+ A recognizer might write “cloud” when you meant “Claude.” But replacing every
84
+ “cloud” would also damage a sentence about cloud storage. Entune stores both
85
+ meanings and asks a decision model which fits the surrounding words.
86
+
87
+ The decision model **selects from your dictionary**. It does not generate or
88
+ rewrite your dictation. Entune applies the stored spelling you can inspect
89
+ and edit.
90
+
91
+ ### 95% fewer incorrect replacements with Jev in our test
92
+
93
+ We tested a fixed learned dictionary on **56 new Parakeet recordings**, in
94
+ three consecutive batches. These recordings were not used to build the
95
+ dictionary. Here are the combined results:
96
+
97
+ | Method | Replacements made | Correct | Incorrect | Uncertain |
98
+ |---|---:|---:|---:|---:|
99
+ | Replace every dictionary match | 93 | 31 | 61 | 1 |
100
+ | Choose with **Jev** | 34 | 30 | **3** | 1 |
101
+ | Choose with **Laya**, locally | 42 | 22 | **20** | 0 |
102
+
103
+ - **Jev prevented 58 of 61 wrong replacements (95%)**, while keeping
104
+ 30 of the 31 correct replacements.
105
+ - **Laya prevented 41 of 61 wrong replacements (67%)**, while keeping
106
+ 22 of the 31 correct replacements.
107
+
108
+ Both reduced wrong replacements in every batch. Jev retained more valid
109
+ corrections; Laya keeps decision processing on your computer.
110
+
111
+ This is a small, single-user test, judged from text context before the models
112
+ ran—not an overall transcription-accuracy claim. Repeated contractions
113
+ contributed substantially to the result. The comparison applies the first
114
+ available replacement unconditionally; it is not the app's decision-model-off
115
+ setting. Read the [per-batch results and method](https://github.com/eandualem/entune/blob/develop/docs/decision-model-results.md)
116
+ for the denominators, limitations and current Laya input constraints.
117
+
118
+ ## Choose the models that suit you
119
+
120
+ Entune separates three jobs, so you can choose each independently:
121
+
122
+ | Job | Our starting recommendation | When it runs |
123
+ |---|---|---|
124
+ | Turn audio into text | **Parakeet** on Apple Silicon, or **AssemblyAI** in the cloud | After each recording |
125
+ | Build your dictionary | **GPT-6.1 Sol**, medium effort, 24,000-character batches | When you request suggestions |
126
+ | Choose dictionary replacements | **Jev** for the stronger result in our test; **Laya** for local processing | After transcription, when enabled |
127
+
128
+ Speech options also include Groq, Soniox, ElevenLabs, xAI and local Whisper.cpp.
129
+ Cloud services use your own provider accounts and keys; Entune does not sell
130
+ inference credits.
131
+
132
+ **Use your ChatGPT subscription to build the dictionary.** Choose **OpenAI →
133
+ ChatGPT subscription → Sign in with ChatGPT** in dictionary setup. This access
134
+ option does not require an OpenAI API key; your plan's model access and usage
135
+ limits apply. An OpenAI API key is also available as a separate access option.
136
+ ChatGPT sign-in is currently experimental; see the
137
+ [access details](https://github.com/eandualem/entune/blob/develop/docs/models.md#dictionary-generation).
138
+ It covers dictionary generation, not cloud speech recognition or Jev.
139
+
140
+ The [model guide](https://github.com/eandualem/entune/blob/develop/docs/models.md)
141
+ covers exact model IDs, local-engine installation, account access, and the
142
+ limits of our recommendations.
143
+
144
+ ## Teach Entune your vocabulary
145
+
146
+ Use **Dictionary → Suggest new entries** to learn from the selected speech
147
+ model's history. Or choose **Learn from audio** to import recordings from
148
+ another dictation app or an audio folder. Entune transcribes imported audio
149
+ with your chosen speech model, then proposes entries for review.
150
+
151
+ **Dictionary generation takes time.** It runs sequentially in batches, with
152
+ the growing dictionary included in each request. Large histories can take
153
+ minutes to hours; our 540-transcript Sol build took about **2 hours 15 minutes**,
154
+ including recovery from a failed request. Importing audio adds transcription
155
+ time. The audio selector shows a rough estimate as you choose recordings:
156
+ allow about **10–20 minutes per audio hour** with Sol, plus transcription.
157
+ This is separate from the fast decision step on each new dictation.
158
+
159
+ Review the proposed entries before applying them. You can stop a build and
160
+ review completed batches, or retry from its checkpoint. Entune pauses dictation
161
+ and separate dictionary editing while learning or proposal review is active.
162
+
163
+ Learned entries belong to their speech model: a Parakeet dictionary is not
164
+ automatically an AssemblyAI dictionary. Pin entries you deliberately want to
165
+ share. **Suggest improvements** can revise learned entries later; more
166
+ refinement does not guarantee a better dictionary.
167
+
168
+ ## Keep control of your recordings
169
+
170
+ - **History:** replay audio, copy text, inspect processing changes, and retry
171
+ a recording with another speech model. Provider failures remain visible.
172
+ - **macOS shortcuts:** hold to talk or toggle hands-free recording. Cancel without
173
+ pasting; usable captured audio stays available for retry.
174
+ - **Local data:** audio, transcripts, settings and keys stay in Entune's data
175
+ folder. Export or delete them in **Settings → Data & Privacy**.
176
+ - **Optional processing:** dictionary correction, repeated-filler reduction,
177
+ and paragraph/bullet formatting have separate controls.
178
+
179
+ There is no Entune account, telemetry or hosted history. Cloud speech sends
180
+ audio to your chosen provider; dictionary generation sends its selected
181
+ transcripts and dictionary to the chosen language-model provider. **Jev sends
182
+ matched context to TypeSafe even when speech recognition is local.** Laya
183
+ keeps that step local. See [data and privacy](https://github.com/eandualem/entune/blob/develop/docs/guide.md#data-and-privacy).
184
+
185
+ Entune is for a trusted, single-user machine. Its loopback API has no
186
+ authentication: other local processes can read or change data through it.
187
+ Do not expose its port to a network or tunnel.
188
+
189
+ ## Guides and contributing
190
+
191
+ - [Using Entune](https://github.com/eandualem/entune/blob/develop/docs/guide.md): permissions, shortcuts, imports, updates and troubleshooting.
192
+ - [Choosing models](https://github.com/eandualem/entune/blob/develop/docs/models.md): cloud and local setup, recommendations and generation time.
193
+ - [Decision-model results](https://github.com/eandualem/entune/blob/develop/docs/decision-model-results.md): what changed, how it was measured, and limitations.
194
+ - [Dictionary reference](https://github.com/eandualem/entune/blob/develop/docs/dictionary.md): meanings, matching, learning and the file format.
195
+ - [Building Entune.app](https://github.com/eandualem/entune/blob/develop/docs/packaging.md), [local API](https://github.com/eandualem/entune/blob/develop/docs/agents-api.md), and [architecture](https://github.com/eandualem/entune/blob/develop/docs/architecture.md).
196
+
197
+ To work on Entune, clone the repository and run `uv sync`, then `uv run entune`.
198
+ Run `uv run ruff check .`, `uv run ruff format --check .`, `uv run mypy` and
199
+ `uv run pytest` before contributing. Pull requests target `develop`;
200
+ `main` receives reviewed releases. See [CONTRIBUTING.md](https://github.com/eandualem/entune/blob/develop/CONTRIBUTING.md).
201
+
202
+ [MIT licensed](https://github.com/eandualem/entune/blob/develop/LICENSE).
@@ -0,0 +1,96 @@
1
+ [project]
2
+ name = "entune"
3
+ version = "0.1.0"
4
+ description = "Dictation with your choice of speech model and a personal dictionary that selects corrections in context."
5
+ readme = "README.md"
6
+ license = "MIT"
7
+ license-files = ["LICENSE"]
8
+ requires-python = ">=3.12"
9
+ dependencies = [
10
+ "httpx>=0.27",
11
+ "httpx2",
12
+ "numpy",
13
+ "pillow",
14
+ "pydantic>=2",
15
+ "pydantic-ai-slim[anthropic,google,groq,mistral,openai]>=2.48",
16
+ "pynput>=1.7,<2 ; sys_platform == 'darwin'",
17
+ "pyobjc-framework-applicationservices>=10 ; sys_platform == 'darwin'",
18
+ "pyobjc-framework-AVFoundation>=10 ; sys_platform == 'darwin'",
19
+ "pystray>=0.19",
20
+ "python-multipart>=0.0.12",
21
+ "pywebview>=5",
22
+ "pywhispercpp>=1.5",
23
+ "sounddevice>=0.5 ; sys_platform == 'darwin'",
24
+ "starlette>=0.41",
25
+ "uvicorn>=0.30",
26
+ ]
27
+
28
+ [project.urls]
29
+ Homepage = "https://github.com/eandualem/entune"
30
+ Documentation = "https://github.com/eandualem/entune#readme"
31
+ Issues = "https://github.com/eandualem/entune/issues"
32
+
33
+ [project.scripts]
34
+ entune = "entune.cli:main"
35
+
36
+ [build-system]
37
+ requires = ["uv_build>=0.8"]
38
+ build-backend = "uv_build"
39
+
40
+ [dependency-groups]
41
+ build = ["pyinstaller>=6"]
42
+ dev = [
43
+ "mypy>=1.11",
44
+ "pytest>=8",
45
+ "ruff>=0.6",
46
+ "types-pynput>=1.8.1.20260712",
47
+ ]
48
+
49
+ [tool.ruff]
50
+ line-length = 100
51
+ target-version = "py312"
52
+ extend-exclude = ["packaging/Entune.spec"]
53
+
54
+ [tool.ruff.lint]
55
+ select = [
56
+ "E",
57
+ "F",
58
+ "I",
59
+ "UP",
60
+ "B",
61
+ "SIM",
62
+ "RUF",
63
+ ]
64
+
65
+ [tool.mypy]
66
+ strict = true
67
+ mypy_path = "src"
68
+ files = [
69
+ "src/entune",
70
+ "tests",
71
+ ]
72
+
73
+ [[tool.mypy.overrides]]
74
+ module = [
75
+ "sounddevice",
76
+ "objc",
77
+ "Quartz",
78
+ "ApplicationServices",
79
+ "AppKit",
80
+ "AVFoundation",
81
+ "Foundation",
82
+ "WebKit",
83
+ "PyObjCTools",
84
+ "pystray.*",
85
+ "webview.*",
86
+ "PIL.*",
87
+ "parakeet_mlx",
88
+ "parakeet_mlx.*",
89
+ "mlx.*",
90
+ "librosa",
91
+ ]
92
+ ignore_missing_imports = true
93
+
94
+ [tool.pytest.ini_options]
95
+ testpaths = ["tests"]
96
+ filterwarnings = ["ignore:The anyio.abc.BlockingPortal alias is deprecated:DeprecationWarning"]
@@ -0,0 +1,68 @@
1
+ [project]
2
+ name = "entune"
3
+ version = "0.1.0"
4
+ description = "Dictation with your choice of speech model and a personal dictionary that selects corrections in context."
5
+ readme = "README.md"
6
+ license = "MIT"
7
+ license-files = ["LICENSE"]
8
+ requires-python = ">=3.12"
9
+ dependencies = [
10
+ "httpx>=0.27",
11
+ "httpx2",
12
+ "numpy",
13
+ "pillow",
14
+ "pydantic>=2",
15
+ "pydantic-ai-slim[anthropic,google,groq,mistral,openai]>=2.48",
16
+ "pynput>=1.7,<2 ; sys_platform == 'darwin'",
17
+ "pyobjc-framework-applicationservices>=10 ; sys_platform == 'darwin'",
18
+ "pyobjc-framework-AVFoundation>=10 ; sys_platform == 'darwin'",
19
+ "pystray>=0.19",
20
+ "python-multipart>=0.0.12",
21
+ "pywebview>=5",
22
+ "pywhispercpp>=1.5",
23
+ "sounddevice>=0.5 ; sys_platform == 'darwin'",
24
+ "starlette>=0.41",
25
+ "uvicorn>=0.30",
26
+ ]
27
+
28
+ [project.urls]
29
+ Homepage = "https://github.com/eandualem/entune"
30
+ Documentation = "https://github.com/eandualem/entune#readme"
31
+ Issues = "https://github.com/eandualem/entune/issues"
32
+
33
+ [project.scripts]
34
+ entune = "entune.cli:main"
35
+
36
+ [build-system]
37
+ requires = ["uv_build>=0.8"]
38
+ build-backend = "uv_build"
39
+
40
+ [dependency-groups]
41
+ build = ["pyinstaller>=6"]
42
+ dev = [
43
+ "mypy>=1.11",
44
+ "pytest>=8",
45
+ "ruff>=0.6",
46
+ "types-pynput>=1.8.1.20260712",
47
+ ]
48
+
49
+ [tool.ruff]
50
+ line-length = 100
51
+ target-version = "py312"
52
+ extend-exclude = ["packaging/Entune.spec"]
53
+
54
+ [tool.ruff.lint]
55
+ select = ["E", "F", "I", "UP", "B", "SIM", "RUF"]
56
+
57
+ [tool.mypy]
58
+ strict = true
59
+ mypy_path = "src"
60
+ files = ["src/entune", "tests"]
61
+
62
+ [[tool.mypy.overrides]]
63
+ module = ["sounddevice", "objc", "Quartz", "ApplicationServices", "AppKit", "AVFoundation", "Foundation", "WebKit", "PyObjCTools", "pystray.*", "webview.*", "PIL.*", "parakeet_mlx", "parakeet_mlx.*", "mlx.*", "librosa"]
64
+ ignore_missing_imports = true
65
+
66
+ [tool.pytest.ini_options]
67
+ testpaths = ["tests"]
68
+ filterwarnings = ["ignore:The anyio.abc.BlockingPortal alias is deprecated:DeprecationWarning"]
@@ -0,0 +1,7 @@
1
+ """Entune: a personal dictation workbench.
2
+
3
+ Record a clip, transcribe it with the speech-to-text model you pick, keep the
4
+ history, retry with another model when one fails, copy the result.
5
+ """
6
+
7
+ __version__ = "0.1.0"
@@ -0,0 +1,3 @@
1
+ from entune.cli import main
2
+
3
+ main()
File without changes
@@ -0,0 +1,34 @@
1
+ """What the API modules share: error responses and the JSON forms of records."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import asdict
6
+ from typing import Any
7
+
8
+ from starlette.responses import PlainTextResponse, Response
9
+
10
+ from entune.app.entune import Entune
11
+ from entune.storage.records import Recording
12
+
13
+
14
+ def bad(message: str, status: int = 400) -> Response:
15
+ return PlainTextResponse(message, status_code=status)
16
+
17
+
18
+ def recording_json(recording: Recording) -> dict[str, Any]:
19
+ return asdict(recording)
20
+
21
+
22
+ def shortcuts_json(app: Entune) -> dict[str, str | None]:
23
+ shortcuts = app.settings.shortcuts()
24
+ return {
25
+ "hold": "+".join(shortcuts.hold) if shortcuts.hold else None,
26
+ "toggle": "+".join(shortcuts.toggle) if shortcuts.toggle else None,
27
+ "cancel": "+".join(shortcuts.cancel) if shortcuts.cancel else None,
28
+ }
29
+
30
+
31
+ def optional_text(value: object, name: str) -> str | None:
32
+ if value is not None and not isinstance(value, str):
33
+ raise ValueError(f"{name} must be a string or null")
34
+ return value