digue 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- digue/__init__.py +67 -0
- digue/__main__.py +6 -0
- digue/audio.py +471 -0
- digue/benchmark.py +316 -0
- digue/cli.py +402 -0
- digue/config.py +443 -0
- digue/container.py +885 -0
- digue/convert.py +273 -0
- digue/delivery.py +165 -0
- digue/dictate.py +424 -0
- digue/notify.py +106 -0
- digue/recording.py +1000 -0
- digue/transcribe.py +841 -0
- digue-0.1.0.dist-info/METADATA +460 -0
- digue-0.1.0.dist-info/RECORD +19 -0
- digue-0.1.0.dist-info/WHEEL +5 -0
- digue-0.1.0.dist-info/entry_points.txt +2 -0
- digue-0.1.0.dist-info/licenses/LICENSE +7 -0
- digue-0.1.0.dist-info/top_level.txt +1 -0
digue/__init__.py
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Local speech-to-text dictation and transcription using whisper.cpp."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__version__ = "0.1.0"
|
|
6
|
+
|
|
7
|
+
DEFAULT_PORT = 8178
|
|
8
|
+
DEFAULT_LANGUAGE = "auto"
|
|
9
|
+
DEFAULT_MODELS = {
|
|
10
|
+
"nvidia": "large-v3-turbo-q8_0",
|
|
11
|
+
"amd": "large-v3-turbo-q8_0",
|
|
12
|
+
"intel": "large-v3-turbo-q8_0",
|
|
13
|
+
"cpu": "small-q8_0",
|
|
14
|
+
}
|
|
15
|
+
# Every ggml model published in huggingface.co/ggerganov/whisper.cpp (the download source), in size order per family:
|
|
16
|
+
# f16, then q8_0, then q5_x. "-q8_0"/"-q5_0"/"-q5_1" are integer-quantized copies (smaller file and RAM, usually
|
|
17
|
+
# faster on CPU, slightly lower accuracy at q5); ".en" are English-only.
|
|
18
|
+
AVAILABLE_MODELS = (
|
|
19
|
+
"tiny",
|
|
20
|
+
"tiny-q8_0",
|
|
21
|
+
"tiny-q5_1",
|
|
22
|
+
"tiny.en",
|
|
23
|
+
"tiny.en-q8_0",
|
|
24
|
+
"tiny.en-q5_1",
|
|
25
|
+
"base",
|
|
26
|
+
"base-q8_0",
|
|
27
|
+
"base-q5_1",
|
|
28
|
+
"base.en",
|
|
29
|
+
"base.en-q8_0",
|
|
30
|
+
"base.en-q5_1",
|
|
31
|
+
"small",
|
|
32
|
+
"small-q8_0",
|
|
33
|
+
"small-q5_1",
|
|
34
|
+
"small.en",
|
|
35
|
+
"small.en-q8_0",
|
|
36
|
+
"small.en-q5_1",
|
|
37
|
+
"medium",
|
|
38
|
+
"medium-q8_0",
|
|
39
|
+
"medium-q5_0",
|
|
40
|
+
"medium.en",
|
|
41
|
+
"medium.en-q8_0",
|
|
42
|
+
"medium.en-q5_0",
|
|
43
|
+
"large-v1",
|
|
44
|
+
"large-v2",
|
|
45
|
+
"large-v2-q8_0",
|
|
46
|
+
"large-v2-q5_0",
|
|
47
|
+
"large-v3",
|
|
48
|
+
"large-v3-q5_0",
|
|
49
|
+
"large-v3-turbo",
|
|
50
|
+
"large-v3-turbo-q8_0",
|
|
51
|
+
"large-v3-turbo-q5_0",
|
|
52
|
+
)
|
|
53
|
+
DEFAULT_MAX_RECORD_SECONDS = 300
|
|
54
|
+
# whisper-server answers only after transcribing the whole file, so this bounds
|
|
55
|
+
# the file length a CPU can handle; [transcribe] timeout overrides it.
|
|
56
|
+
DEFAULT_TRANSCRIPTION_TIMEOUT = 600
|
|
57
|
+
|
|
58
|
+
from digue.config import load_config # noqa: E402
|
|
59
|
+
from digue.recording import record_to # noqa: E402
|
|
60
|
+
from digue.transcribe import transcribe_file # noqa: E402
|
|
61
|
+
|
|
62
|
+
__all__ = [
|
|
63
|
+
"__version__",
|
|
64
|
+
"load_config",
|
|
65
|
+
"record_to",
|
|
66
|
+
"transcribe_file",
|
|
67
|
+
]
|
digue/__main__.py
ADDED
digue/audio.py
ADDED
|
@@ -0,0 +1,471 @@
|
|
|
1
|
+
"""Dictation audio archive: compress, save, rescue, and clean."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import contextlib
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from digue.dictate import DeliveryResult
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def now_timestamp() -> str:
|
|
16
|
+
"""Shell-friendly timestamp for filenames: YYYYMMDD-HHMMSS (no ':' to escape)."""
|
|
17
|
+
import datetime
|
|
18
|
+
|
|
19
|
+
return datetime.datetime.now().strftime("%Y%m%d-%H%M%S")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def month_dir_for(timestamp: str) -> Path:
|
|
23
|
+
"""Returns the YYYY/MM relative path for a YYYYMMDD-HHMMSS timestamp.
|
|
24
|
+
|
|
25
|
+
Derived from the timestamp itself (not from now()), so the .txt always lands beside the audio saved with the same
|
|
26
|
+
timestamp even across midnight.
|
|
27
|
+
"""
|
|
28
|
+
return Path(timestamp[:4]) / timestamp[4:6]
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _saved_stem(timestamp: str, take_id: str | None) -> str:
|
|
32
|
+
"""Stem for saved files: the take id suffix makes two takes that end in the same second unique; the exclusive write
|
|
33
|
+
is the second line of defense."""
|
|
34
|
+
return f"{timestamp}-{take_id}" if take_id else timestamp
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _compress_audio(
|
|
38
|
+
rec_file: str | Path, audio_format: str, backend: str | None = None, container_name: str | None = None
|
|
39
|
+
) -> Path:
|
|
40
|
+
"""Compresses a WAV recording in place. Returns the new path (rec_file swapped).
|
|
41
|
+
|
|
42
|
+
audio_format: "wav" (no-op), "flac", or "opus".
|
|
43
|
+
- flac: lossless, ~35% of WAV for speech, decodable by whisper-server natively (verified). Safe choice: the archive
|
|
44
|
+
is bit-exact to what was transcribed.
|
|
45
|
+
- opus: ~7% of WAV at 24 kbit/s (lossy). Speech quality is excellent, but the archive is not identical to the
|
|
46
|
+
input; whisper-server rejects opus, so a retranscription goes through the ffmpeg fallback.
|
|
47
|
+
|
|
48
|
+
Tries host ffmpeg first, then falls back to running ffmpeg inside the local container (`container_name`, the
|
|
49
|
+
configured `server.container-name`) via stdin/stdout pipe when backend is not "remote" (any other value means
|
|
50
|
+
local; only remote-or-not is looked at). Without a container name there is no container fallback.
|
|
51
|
+
|
|
52
|
+
The final name is reserved up front (exclusive creation): two takes can never overwrite each other's compressed
|
|
53
|
+
file; a collision raises and the caller rescues the WAV. Whatever happens afterwards -- ffmpeg failure, timeout, a
|
|
54
|
+
docker error -- the reservation and the temp file are dropped unless the compressed file was published, so a
|
|
55
|
+
failure never leaves an empty .flac next to the WAV it kept.
|
|
56
|
+
"""
|
|
57
|
+
if audio_format == "wav":
|
|
58
|
+
return Path(rec_file)
|
|
59
|
+
if audio_format not in ("flac", "opus"):
|
|
60
|
+
raise KeyError(audio_format)
|
|
61
|
+
|
|
62
|
+
rec_file = Path(rec_file)
|
|
63
|
+
converted = rec_file.with_suffix(f".{audio_format}")
|
|
64
|
+
temp_converted = converted.with_name(f".{converted.name}.{os.getpid()}.tmp")
|
|
65
|
+
converted.touch(exist_ok=False)
|
|
66
|
+
published = False
|
|
67
|
+
try:
|
|
68
|
+
published = _run_compression(rec_file, converted, temp_converted, audio_format, backend, container_name)
|
|
69
|
+
finally:
|
|
70
|
+
if not published:
|
|
71
|
+
temp_converted.unlink(missing_ok=True)
|
|
72
|
+
converted.unlink(missing_ok=True) # drop the reservation, keep the WAV
|
|
73
|
+
if not published:
|
|
74
|
+
return rec_file
|
|
75
|
+
rec_file.unlink(missing_ok=True)
|
|
76
|
+
return converted
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _run_compression(
|
|
80
|
+
rec_file: Path,
|
|
81
|
+
converted: Path,
|
|
82
|
+
temp_converted: Path,
|
|
83
|
+
audio_format: str,
|
|
84
|
+
backend: str | None,
|
|
85
|
+
container_name: str | None,
|
|
86
|
+
) -> bool:
|
|
87
|
+
"""Writes the compressed audio into temp_converted and publishes it as converted. Returns True when published;
|
|
88
|
+
False (after a warning) when compression was not possible. Exceptions propagate to the caller."""
|
|
89
|
+
import shutil
|
|
90
|
+
import subprocess
|
|
91
|
+
|
|
92
|
+
from digue.container import container_status
|
|
93
|
+
|
|
94
|
+
codec_args = {
|
|
95
|
+
"flac": ["-c:a", "flac"],
|
|
96
|
+
"opus": ["-c:a", "libopus", "-b:a", "24k"],
|
|
97
|
+
}
|
|
98
|
+
format_args = {
|
|
99
|
+
"flac": ["-f", "flac"],
|
|
100
|
+
"opus": ["-f", "ogg"],
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
if shutil.which("ffmpeg"):
|
|
104
|
+
# The temp is ours alone (ffmpeg opens the path itself, so exclusivity is the temp name); no -y: the final file
|
|
105
|
+
# is never ffmpeg's to overwrite.
|
|
106
|
+
temp_converted.unlink(missing_ok=True)
|
|
107
|
+
result = subprocess.run(
|
|
108
|
+
[
|
|
109
|
+
"ffmpeg",
|
|
110
|
+
"-loglevel",
|
|
111
|
+
"error",
|
|
112
|
+
"-i",
|
|
113
|
+
str(rec_file),
|
|
114
|
+
*codec_args[audio_format],
|
|
115
|
+
*format_args[audio_format],
|
|
116
|
+
str(temp_converted),
|
|
117
|
+
],
|
|
118
|
+
capture_output=True,
|
|
119
|
+
timeout=600,
|
|
120
|
+
)
|
|
121
|
+
if result.returncode == 0 and temp_converted.exists():
|
|
122
|
+
os.replace(temp_converted, converted)
|
|
123
|
+
return True
|
|
124
|
+
print(
|
|
125
|
+
f"Warning: ffmpeg failed to compress recording ({result.stderr.decode(errors='replace').strip()[:150]}); keeping WAV",
|
|
126
|
+
file=sys.stderr,
|
|
127
|
+
)
|
|
128
|
+
return False
|
|
129
|
+
|
|
130
|
+
if backend != "remote" and container_name and container_status(container_name) == "running":
|
|
131
|
+
cmd = [
|
|
132
|
+
"docker",
|
|
133
|
+
"exec",
|
|
134
|
+
"-i",
|
|
135
|
+
container_name,
|
|
136
|
+
"ffmpeg",
|
|
137
|
+
"-loglevel",
|
|
138
|
+
"error",
|
|
139
|
+
"-i",
|
|
140
|
+
"pipe:0",
|
|
141
|
+
*codec_args[audio_format],
|
|
142
|
+
*format_args[audio_format],
|
|
143
|
+
"pipe:1",
|
|
144
|
+
]
|
|
145
|
+
result = subprocess.run(
|
|
146
|
+
cmd,
|
|
147
|
+
input=rec_file.read_bytes(),
|
|
148
|
+
capture_output=True,
|
|
149
|
+
timeout=600,
|
|
150
|
+
)
|
|
151
|
+
if result.returncode == 0 and result.stdout:
|
|
152
|
+
temp_converted.unlink(missing_ok=True)
|
|
153
|
+
with open(temp_converted, "xb") as temp_file:
|
|
154
|
+
temp_file.write(result.stdout)
|
|
155
|
+
os.replace(temp_converted, converted)
|
|
156
|
+
return True
|
|
157
|
+
print(
|
|
158
|
+
f"Warning: ffmpeg failed to compress recording ({result.stderr.decode(errors='replace').strip()[:150]}); keeping WAV",
|
|
159
|
+
file=sys.stderr,
|
|
160
|
+
)
|
|
161
|
+
return False
|
|
162
|
+
|
|
163
|
+
print(
|
|
164
|
+
f"Warning: ffmpeg not found, keeping the recording as WAV (install ffmpeg for {audio_format})",
|
|
165
|
+
file=sys.stderr,
|
|
166
|
+
)
|
|
167
|
+
return False
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def save_audio(
|
|
171
|
+
rec_file: str | Path,
|
|
172
|
+
audio_dir: str | Path,
|
|
173
|
+
audio_format: str = "wav",
|
|
174
|
+
timestamp: str | None = None,
|
|
175
|
+
backend: str | None = None,
|
|
176
|
+
take_id: str | None = None,
|
|
177
|
+
container_name: str | None = None,
|
|
178
|
+
) -> tuple[Path, str]:
|
|
179
|
+
"""Copies audio to <audio_dir>/YYYY/MM/<timestamp>-<take_id>.<ext>. Returns (saved_path, timestamp).
|
|
180
|
+
|
|
181
|
+
The timestamp comes from the caller (dictate_toggle generates it when the take stops, so the audio and its
|
|
182
|
+
transcript share the same name even when archiving runs later). Without one, the current time is used. The copy is
|
|
183
|
+
exclusive: a name collision raises instead of overwriting another take's file. audio_format "flac" or "opus"
|
|
184
|
+
compresses the copy; the live recording file is kept as WAV and removed after saving.
|
|
185
|
+
"""
|
|
186
|
+
audio_dir = Path(audio_dir)
|
|
187
|
+
timestamp = timestamp or now_timestamp()
|
|
188
|
+
month_dir = audio_dir / month_dir_for(timestamp)
|
|
189
|
+
month_dir.mkdir(parents=True, exist_ok=True)
|
|
190
|
+
source = Path(rec_file)
|
|
191
|
+
source_suffix = source.suffix.lower() if source.suffix.lower() in {".wav", ".flac", ".opus"} else ".wav"
|
|
192
|
+
saved = month_dir / f"{_saved_stem(timestamp, take_id)}{source_suffix}"
|
|
193
|
+
_copy_file_exclusive(source, saved)
|
|
194
|
+
if audio_format != "wav" and saved.suffix.lower() != f".{audio_format}":
|
|
195
|
+
saved = _compress_audio(saved, audio_format, backend=backend, container_name=container_name)
|
|
196
|
+
return saved, timestamp
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _copy_file_exclusive(source: Path, destination: Path) -> None:
|
|
200
|
+
"""Copies source to destination with exclusive creation ("xb"): a collision raises FileExistsError instead of
|
|
201
|
+
silently overwriting another take's file."""
|
|
202
|
+
import shutil
|
|
203
|
+
|
|
204
|
+
with open(destination, "xb") as destination_file, source.open("rb") as source_file:
|
|
205
|
+
shutil.copyfileobj(source_file, destination_file)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def rescue_recording(
|
|
209
|
+
rec_file: str | Path, audio_dir: str | Path, timestamp: str, take_id: str | None = None
|
|
210
|
+
) -> Path | None:
|
|
211
|
+
"""Keeps a recording that could not be fully delivered. Never raises.
|
|
212
|
+
|
|
213
|
+
Copies the recording to <audio_dir>/YYYY/MM/<timestamp>-<take_id>.<ext> (the live suffix is kept: a native FLAC
|
|
214
|
+
take stays .flac) via an exclusive temp sibling + flush + fsync + exclusive publish (runtime dir and audio-dir
|
|
215
|
+
usually live on different filesystems, the destination must never be readable in a partial state, and an existing
|
|
216
|
+
destination is never overwritten; see _publish_exclusive), then removes the origin -- only after the destination is
|
|
217
|
+
valid. Any failure before the publish removes the temp, preserves the origin and reports on stderr; once the
|
|
218
|
+
destination is linked the rescue is done, and a failure to remove the origin is only reported (the caller must not
|
|
219
|
+
retry: the published name is exclusive).
|
|
220
|
+
"""
|
|
221
|
+
import shutil
|
|
222
|
+
|
|
223
|
+
rec_file = Path(rec_file)
|
|
224
|
+
temp_archived: Path | None = None
|
|
225
|
+
try:
|
|
226
|
+
month_dir = Path(audio_dir) / month_dir_for(timestamp)
|
|
227
|
+
month_dir.mkdir(parents=True, exist_ok=True)
|
|
228
|
+
archived = month_dir / f"{_saved_stem(timestamp, take_id)}{rec_file.suffix.lower() or '.wav'}"
|
|
229
|
+
temp_archived = archived.with_name(f".{archived.name}.{os.getpid()}.tmp")
|
|
230
|
+
with open(temp_archived, "xb") as temp_file, rec_file.open("rb") as source_file:
|
|
231
|
+
shutil.copyfileobj(source_file, temp_file)
|
|
232
|
+
temp_file.flush()
|
|
233
|
+
os.fsync(temp_file.fileno())
|
|
234
|
+
_publish_exclusive(temp_archived, archived)
|
|
235
|
+
except Exception as rescue_exc:
|
|
236
|
+
if temp_archived is not None:
|
|
237
|
+
temp_archived.unlink(missing_ok=True)
|
|
238
|
+
print(f"Failed to keep recording: {rescue_exc}; audio still at {rec_file}", file=sys.stderr)
|
|
239
|
+
return None
|
|
240
|
+
temp_archived.unlink(missing_ok=True)
|
|
241
|
+
try:
|
|
242
|
+
rec_file.unlink()
|
|
243
|
+
except OSError as unlink_exc:
|
|
244
|
+
print(f"Recording kept at {archived}, but the origin could not be removed: {unlink_exc}", file=sys.stderr)
|
|
245
|
+
return archived
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _publish_exclusive(temp_path: Path, destination: Path) -> None:
|
|
249
|
+
"""Publishes temp_path as destination without ever overwriting it.
|
|
250
|
+
|
|
251
|
+
os.link fails with FileExistsError instead of replacing, so it is as exclusive as the temp file. Filesystems
|
|
252
|
+
without hard links (vfat/exFAT, some FUSE mounts such as rclone or sshfs) refuse the link with EPERM or EOPNOTSUPP;
|
|
253
|
+
there the fallback is an existence check followed by os.replace. That check-then-replace has a window another
|
|
254
|
+
rescue could slip into, accepted because the name already carries the take id: the alternative was every rescue
|
|
255
|
+
failing on such an audio-dir and the take staying in the runtime dir (tmpfs, gone at reboot).
|
|
256
|
+
"""
|
|
257
|
+
import errno
|
|
258
|
+
|
|
259
|
+
try:
|
|
260
|
+
os.link(temp_path, destination)
|
|
261
|
+
except OSError as exc:
|
|
262
|
+
if exc.errno not in (errno.EPERM, errno.EOPNOTSUPP, errno.ENOTSUP, errno.EXDEV):
|
|
263
|
+
raise
|
|
264
|
+
if destination.exists():
|
|
265
|
+
raise FileExistsError(errno.EEXIST, "destination already exists", str(destination)) from exc
|
|
266
|
+
os.replace(temp_path, destination)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _write_transcript(audio_dir: Path, timestamp: str, text: str, take_id: str | None = None) -> Path:
|
|
270
|
+
"""Writes the transcript next to the recording: <audio_dir>/YYYY/MM/<timestamp>-<take_id>.txt.
|
|
271
|
+
|
|
272
|
+
The month folder comes from the timestamp itself (not from now()), so the .txt always lands beside the audio saved
|
|
273
|
+
with the same timestamp. The write is exclusive: a collision raises instead of overwriting another take's text.
|
|
274
|
+
"""
|
|
275
|
+
text_path = audio_dir / month_dir_for(timestamp) / f"{_saved_stem(timestamp, take_id)}.txt"
|
|
276
|
+
text_path.parent.mkdir(parents=True, exist_ok=True)
|
|
277
|
+
with open(text_path, "xb") as text_file:
|
|
278
|
+
text_file.write((text + "\n").encode())
|
|
279
|
+
return text_path
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _archive_recording(
|
|
283
|
+
config: dict[str, dict[str, Any]], rec_file: Path, timestamp: str, take_id: str | None
|
|
284
|
+
) -> tuple[bool, Path | None]:
|
|
285
|
+
"""Archives a delivered recording: copy + compression when save-audio is on (the slow part), then removes the live
|
|
286
|
+
file. Returns (archived, rescued_path).
|
|
287
|
+
|
|
288
|
+
On failure the live recording is kept somewhere the user can find it: when the exclusive copy already completed and
|
|
289
|
+
only the compression raised (ffmpeg timeout, docker exec error), that copy is the rescue -- it has the exact name
|
|
290
|
+
`rescue_recording` would use, so rescuing again would collide and strand the live file in the runtime dir;
|
|
291
|
+
otherwise the live file is rescued (moved) as is. Either way the user is told.
|
|
292
|
+
"""
|
|
293
|
+
from digue.container import _is_remote, resolve_container_name
|
|
294
|
+
from digue.notify import send_notification
|
|
295
|
+
|
|
296
|
+
audio_dir = Path(config["dictate"]["audio_dir"])
|
|
297
|
+
try:
|
|
298
|
+
if config["dictate"]["save_audio"]:
|
|
299
|
+
# compression only needs remote-or-not: no hardware detection (nvidia-smi/lspci) on every delivery
|
|
300
|
+
save_audio(
|
|
301
|
+
rec_file,
|
|
302
|
+
audio_dir,
|
|
303
|
+
config["dictate"].get("audio_format", "wav"),
|
|
304
|
+
timestamp=timestamp,
|
|
305
|
+
backend="remote" if _is_remote(config) else "local",
|
|
306
|
+
take_id=take_id,
|
|
307
|
+
container_name=resolve_container_name(config),
|
|
308
|
+
)
|
|
309
|
+
rec_file.unlink(missing_ok=True)
|
|
310
|
+
return True, None
|
|
311
|
+
except Exception as save_exc:
|
|
312
|
+
uncompressed = _completed_copy(rec_file, audio_dir, timestamp, take_id)
|
|
313
|
+
if uncompressed is not None:
|
|
314
|
+
rec_file.unlink(missing_ok=True)
|
|
315
|
+
send_notification(
|
|
316
|
+
f"Failed to compress audio: {save_exc}; uncompressed copy kept at {uncompressed}", timeout_ms=10000
|
|
317
|
+
)
|
|
318
|
+
return False, uncompressed
|
|
319
|
+
rescued_path = rescue_recording(rec_file, audio_dir, timestamp, take_id)
|
|
320
|
+
message = f"Failed to save audio: {save_exc}"
|
|
321
|
+
if rescued_path:
|
|
322
|
+
message += f"; uncompressed copy kept at {rescued_path}"
|
|
323
|
+
send_notification(message, timeout_ms=10000)
|
|
324
|
+
return False, rescued_path
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def _completed_copy(rec_file: Path, audio_dir: Path, timestamp: str, take_id: str | None) -> Path | None:
|
|
328
|
+
"""The uncompressed copy `save_audio` makes before compressing, when it is complete (same size as the live file);
|
|
329
|
+
None when it does not exist or the copy itself is what failed (partial)."""
|
|
330
|
+
copy = audio_dir / month_dir_for(timestamp) / f"{_saved_stem(timestamp, take_id)}{rec_file.suffix.lower()}"
|
|
331
|
+
with contextlib.suppress(OSError):
|
|
332
|
+
if copy.stat().st_size == rec_file.stat().st_size:
|
|
333
|
+
return copy
|
|
334
|
+
return None
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _delivered_transcript(audio_dir: Path, take_id: str) -> Path | None:
|
|
338
|
+
"""Returns the transcript already saved for a take, if any.
|
|
339
|
+
|
|
340
|
+
finish_dictation writes the transcript right after pasting, so its presence proves the text reached the user: a
|
|
341
|
+
recovery of a take whose daemon died afterwards (during the archive) must not paste it again.
|
|
342
|
+
"""
|
|
343
|
+
return next(iter(sorted(audio_dir.glob(f"[0-9][0-9][0-9][0-9]/[0-9][0-9]/*-{take_id}.txt"))), None)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _archive_recovered_take(config: dict[str, dict[str, Any]], rec_file: Path, transcript: Path) -> DeliveryResult:
|
|
347
|
+
"""Finishes a take whose text was already pasted and saved: archive only.
|
|
348
|
+
|
|
349
|
+
The audio takes the transcript's timestamp and take id, so it lands next to the .txt. Every outcome is terminal
|
|
350
|
+
(the text was delivered).
|
|
351
|
+
"""
|
|
352
|
+
|
|
353
|
+
from digue.notify import send_notification
|
|
354
|
+
|
|
355
|
+
timestamp, _, take_id = transcript.stem.rpartition("-")
|
|
356
|
+
send_notification("Recovering the previous recording (text already delivered)")
|
|
357
|
+
# The daemon died mid-archive, so a partial .wav copy, an empty compressed
|
|
358
|
+
# reservation or a .tmp of this same take may already sit next to the .txt.
|
|
359
|
+
# Same stem means same take id: they are provably incomplete products of
|
|
360
|
+
# this take, and the live WAV is the source of truth. Without this, the
|
|
361
|
+
# exclusive archive collides and the good audio stays in the runtime dir.
|
|
362
|
+
for leftover in transcript.parent.glob(f"*{transcript.stem}*"):
|
|
363
|
+
if leftover.suffix not in (".txt", ".json"):
|
|
364
|
+
leftover.unlink(missing_ok=True)
|
|
365
|
+
archived, rescued_path = _archive_recording(config, rec_file, timestamp, take_id)
|
|
366
|
+
if archived:
|
|
367
|
+
return DeliveryResult(outcome="delivered", exit_code=0)
|
|
368
|
+
if rescued_path is not None:
|
|
369
|
+
return DeliveryResult(outcome="rescued", exit_code=1, rescued_path=rescued_path)
|
|
370
|
+
return DeliveryResult(outcome="delivered", exit_code=1)
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
DICTATION_RECORDING_SUFFIXES = frozenset((".wav", ".flac", ".opus"))
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def _dictation_files(audio_dir: Path, suffixes: frozenset[str]) -> list[Path]:
|
|
377
|
+
"""Lists <audio_dir>/YYYY/MM/<timestamp>[-<take_id>].<suffix> dictation files.
|
|
378
|
+
|
|
379
|
+
Only that exact layout qualifies: audio-dir is user-configurable, and a recursive *.wav/*.flac/*.txt glob pointed
|
|
380
|
+
at a music folder would remove the library. Both layouts are accepted: the pre-take-id stem (YYYYMMDD-HHMMSS, from
|
|
381
|
+
now_timestamp()) and the current YYYYMMDD-HHMMSS-<16 hex chars>. Symlinks are skipped: clean unlinks what it lists,
|
|
382
|
+
and deleting a symlink's target would destroy an outside file.
|
|
383
|
+
"""
|
|
384
|
+
import re
|
|
385
|
+
|
|
386
|
+
stem_pattern = re.compile(r"^\d{8}-\d{6}(-[0-9a-f]{16})?$")
|
|
387
|
+
found = []
|
|
388
|
+
for year_dir in audio_dir.glob("[0-9][0-9][0-9][0-9]"):
|
|
389
|
+
for month_dir in year_dir.glob("[0-9][0-9]"):
|
|
390
|
+
if not month_dir.is_dir():
|
|
391
|
+
continue
|
|
392
|
+
for path in month_dir.iterdir():
|
|
393
|
+
if (
|
|
394
|
+
path.is_file()
|
|
395
|
+
and not path.is_symlink()
|
|
396
|
+
and path.suffix.lower() in suffixes
|
|
397
|
+
and stem_pattern.match(path.stem)
|
|
398
|
+
):
|
|
399
|
+
found.append(path)
|
|
400
|
+
return sorted(found)
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def cmd_clean(args: argparse.Namespace, config: dict[str, dict[str, Any]]) -> int:
|
|
404
|
+
"""Removes dictation recordings and/or transcripts from the audio directory.
|
|
405
|
+
|
|
406
|
+
Lists what it found and asks for confirmation; --force removes right away. --what selects what is removed:
|
|
407
|
+
recordings, transcripts, or both (default).
|
|
408
|
+
"""
|
|
409
|
+
|
|
410
|
+
audio_dir = Path(config["dictate"]["audio_dir"])
|
|
411
|
+
if not audio_dir.exists():
|
|
412
|
+
print(f"Audio directory does not exist: {audio_dir}", file=sys.stderr)
|
|
413
|
+
return 0
|
|
414
|
+
|
|
415
|
+
what = args.what
|
|
416
|
+
recordings = _dictation_files(audio_dir, DICTATION_RECORDING_SUFFIXES) if what in ("recordings", "both") else []
|
|
417
|
+
transcripts = _dictation_files(audio_dir, frozenset((".txt",))) if what in ("transcripts", "both") else []
|
|
418
|
+
|
|
419
|
+
total_mb = sum(path.stat().st_size for path in recordings) / (1024 * 1024)
|
|
420
|
+
print(f"Audio directory: {audio_dir}", file=sys.stderr)
|
|
421
|
+
print(f" Recordings: {len(recordings)} file(s), {total_mb:.1f} MB", file=sys.stderr)
|
|
422
|
+
print(f" Transcripts: {len(transcripts)} file(s)", file=sys.stderr)
|
|
423
|
+
|
|
424
|
+
# A rescued take's .json is metadata of the recording, not its own category: it is removed together with the
|
|
425
|
+
# recording of the same stem (counted as one unit), never listed as a transcript, and a .json whose recording is
|
|
426
|
+
# gone is preserved.
|
|
427
|
+
recording_metadata = {path: path.with_suffix(".json") for path in recordings if path.with_suffix(".json").exists()}
|
|
428
|
+
|
|
429
|
+
if not recordings and not transcripts:
|
|
430
|
+
print("Nothing to remove.", file=sys.stderr)
|
|
431
|
+
return 0
|
|
432
|
+
|
|
433
|
+
total = len(recordings) + len(transcripts)
|
|
434
|
+
if not args.force:
|
|
435
|
+
for path in sorted(recordings):
|
|
436
|
+
print(f" {path.relative_to(audio_dir)}", file=sys.stderr)
|
|
437
|
+
if path in recording_metadata:
|
|
438
|
+
print(f" {recording_metadata[path].relative_to(audio_dir)} (metadata)", file=sys.stderr)
|
|
439
|
+
for path in sorted(transcripts):
|
|
440
|
+
print(f" {path.relative_to(audio_dir)}", file=sys.stderr)
|
|
441
|
+
try:
|
|
442
|
+
answer = input(f"Remove all {total} file(s)? [y/N] ")
|
|
443
|
+
except EOFError:
|
|
444
|
+
# no terminal to ask (cron, a pipe): the listing above says what
|
|
445
|
+
# would go; the user opts in explicitly
|
|
446
|
+
print("\nNo terminal to confirm on; run again with --force to remove.", file=sys.stderr)
|
|
447
|
+
return 1
|
|
448
|
+
if answer.strip().lower() not in ("y", "yes"):
|
|
449
|
+
print("Aborted.", file=sys.stderr)
|
|
450
|
+
return 1
|
|
451
|
+
|
|
452
|
+
count = 0
|
|
453
|
+
for path in recordings:
|
|
454
|
+
path.unlink()
|
|
455
|
+
count += 1
|
|
456
|
+
metadata_path = recording_metadata.get(path)
|
|
457
|
+
if metadata_path is not None:
|
|
458
|
+
metadata_path.unlink()
|
|
459
|
+
for path in transcripts:
|
|
460
|
+
path.unlink()
|
|
461
|
+
count += 1
|
|
462
|
+
|
|
463
|
+
# Remove now-empty month/year directories (deepest first)
|
|
464
|
+
for directory in sorted((parent for parent in audio_dir.rglob("*") if parent.is_dir()), reverse=True):
|
|
465
|
+
with contextlib.suppress(OSError):
|
|
466
|
+
directory.rmdir()
|
|
467
|
+
with contextlib.suppress(OSError):
|
|
468
|
+
audio_dir.rmdir()
|
|
469
|
+
|
|
470
|
+
print(f"Removed {count} file(s).", file=sys.stderr)
|
|
471
|
+
return 0
|