audio-transcode-watcher 0.4.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- audio_transcode_watcher/__init__.py +3 -0
- audio_transcode_watcher/config.py +190 -0
- audio_transcode_watcher/encoder.py +208 -0
- audio_transcode_watcher/lyrics.py +193 -0
- audio_transcode_watcher/main.py +97 -0
- audio_transcode_watcher/sync.py +471 -0
- audio_transcode_watcher/utils.py +150 -0
- audio_transcode_watcher/watcher.py +109 -0
- audio_transcode_watcher-0.4.2.dist-info/METADATA +305 -0
- audio_transcode_watcher-0.4.2.dist-info/RECORD +13 -0
- audio_transcode_watcher-0.4.2.dist-info/WHEEL +4 -0
- audio_transcode_watcher-0.4.2.dist-info/entry_points.txt +2 -0
- audio_transcode_watcher-0.4.2.dist-info/licenses/LICENSE +674 -0
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""Configuration loading for audio-transcode-watcher."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
import yaml
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
# Codec to file extension mapping
|
|
15
|
+
CODEC_EXTENSIONS = {
|
|
16
|
+
"alac": ".m4a",
|
|
17
|
+
"aac": ".m4a",
|
|
18
|
+
"mp3": ".mp3",
|
|
19
|
+
"opus": ".opus",
|
|
20
|
+
"flac": ".flac",
|
|
21
|
+
"wav": ".wav",
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
# Codecs that support embedded artwork
|
|
25
|
+
ARTWORK_SUPPORTED_CODECS = {"alac", "aac", "mp3", "flac"}
|
|
26
|
+
|
|
27
|
+
# Default bitrates for lossy codecs
|
|
28
|
+
DEFAULT_BITRATES = {
|
|
29
|
+
"aac": "256k",
|
|
30
|
+
"mp3": "256k",
|
|
31
|
+
"opus": "128k",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class OutputConfig:
|
|
37
|
+
"""Configuration for a single output destination."""
|
|
38
|
+
|
|
39
|
+
name: str
|
|
40
|
+
codec: str
|
|
41
|
+
path: str
|
|
42
|
+
bitrate: str = ""
|
|
43
|
+
include_artwork: bool = True
|
|
44
|
+
|
|
45
|
+
def __post_init__(self) -> None:
|
|
46
|
+
"""Validate and set defaults after initialization."""
|
|
47
|
+
self.codec = self.codec.lower()
|
|
48
|
+
|
|
49
|
+
if self.codec not in CODEC_EXTENSIONS:
|
|
50
|
+
raise ValueError(
|
|
51
|
+
f"Unknown codec '{self.codec}'. "
|
|
52
|
+
f"Supported: {', '.join(CODEC_EXTENSIONS.keys())}"
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
# Set default bitrate for lossy codecs
|
|
56
|
+
if not self.bitrate and self.codec in DEFAULT_BITRATES:
|
|
57
|
+
self.bitrate = DEFAULT_BITRATES[self.codec]
|
|
58
|
+
|
|
59
|
+
# Artwork not supported for some codecs
|
|
60
|
+
if self.include_artwork and self.codec not in ARTWORK_SUPPORTED_CODECS:
|
|
61
|
+
self.include_artwork = False
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def extension(self) -> str:
|
|
65
|
+
"""Get the file extension for this codec."""
|
|
66
|
+
return CODEC_EXTENSIONS[self.codec]
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def is_lossless(self) -> bool:
|
|
70
|
+
"""Check if this codec is lossless."""
|
|
71
|
+
return self.codec in {"alac", "flac", "wav"}
|
|
72
|
+
|
|
73
|
+
@classmethod
|
|
74
|
+
def from_dict(cls, data: dict[str, Any]) -> OutputConfig:
|
|
75
|
+
"""Create OutputConfig from a dictionary."""
|
|
76
|
+
return cls(
|
|
77
|
+
name=data["name"],
|
|
78
|
+
codec=data["codec"],
|
|
79
|
+
path=data["path"],
|
|
80
|
+
bitrate=data.get("bitrate", ""),
|
|
81
|
+
include_artwork=data.get("include_artwork", True),
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass
|
|
86
|
+
class Config:
|
|
87
|
+
"""Main configuration for audio-transcode-watcher."""
|
|
88
|
+
|
|
89
|
+
source_path: str
|
|
90
|
+
outputs: list[OutputConfig] = field(default_factory=list)
|
|
91
|
+
force_reencode: bool = False
|
|
92
|
+
allow_initial_bulk_encode: bool = True # Allow encoding when outputs are empty
|
|
93
|
+
parallel_workers: int = 4 # Number of parallel encoding workers
|
|
94
|
+
stability_timeout: float = 60.0
|
|
95
|
+
min_stable_seconds: float = 1.0
|
|
96
|
+
fetch_lyrics: bool = True # Auto-fetch .lrc lyrics via syncedlyrics
|
|
97
|
+
whisper_fallback: bool = True # Use Whisper local transcription as fallback
|
|
98
|
+
whisper_model: str = "base" # Whisper model size: tiny, base, small, medium, large
|
|
99
|
+
|
|
100
|
+
def __post_init__(self) -> None:
|
|
101
|
+
"""Validate configuration after initialization."""
|
|
102
|
+
if not self.source_path:
|
|
103
|
+
raise ValueError("source_path is required")
|
|
104
|
+
|
|
105
|
+
if not self.outputs:
|
|
106
|
+
raise ValueError("At least one output is required")
|
|
107
|
+
|
|
108
|
+
# Check for duplicate output names
|
|
109
|
+
names = [o.name for o in self.outputs]
|
|
110
|
+
if len(names) != len(set(names)):
|
|
111
|
+
raise ValueError("Duplicate output names detected")
|
|
112
|
+
|
|
113
|
+
# Check for duplicate output paths
|
|
114
|
+
paths = [o.path for o in self.outputs]
|
|
115
|
+
if len(paths) != len(set(paths)):
|
|
116
|
+
raise ValueError("Duplicate output paths detected")
|
|
117
|
+
|
|
118
|
+
@property
|
|
119
|
+
def output_paths(self) -> list[str]:
|
|
120
|
+
"""Get list of all output directory paths."""
|
|
121
|
+
return [o.path for o in self.outputs]
|
|
122
|
+
|
|
123
|
+
def get_output_by_name(self, name: str) -> OutputConfig | None:
|
|
124
|
+
"""Get an output configuration by name."""
|
|
125
|
+
for output in self.outputs:
|
|
126
|
+
if output.name == name:
|
|
127
|
+
return output
|
|
128
|
+
return None
|
|
129
|
+
|
|
130
|
+
@classmethod
|
|
131
|
+
def from_dict(cls, data: dict[str, Any]) -> Config:
|
|
132
|
+
"""Create Config from a dictionary."""
|
|
133
|
+
outputs = [OutputConfig.from_dict(o) for o in data.get("outputs", [])]
|
|
134
|
+
settings = data.get("settings", {})
|
|
135
|
+
|
|
136
|
+
return cls(
|
|
137
|
+
source_path=data.get("source", {}).get("path", ""),
|
|
138
|
+
outputs=outputs,
|
|
139
|
+
force_reencode=settings.get("force_reencode", False),
|
|
140
|
+
allow_initial_bulk_encode=settings.get("allow_initial_bulk_encode", True),
|
|
141
|
+
parallel_workers=settings.get("parallel_workers", 4),
|
|
142
|
+
stability_timeout=settings.get("stability_timeout", 60.0),
|
|
143
|
+
min_stable_seconds=settings.get("min_stable_seconds", 1.0),
|
|
144
|
+
fetch_lyrics=settings.get("fetch_lyrics", True),
|
|
145
|
+
whisper_fallback=settings.get("whisper_fallback", True),
|
|
146
|
+
whisper_model=settings.get("whisper_model", "base"),
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
@classmethod
|
|
150
|
+
def from_yaml_file(cls, path: str) -> Config:
|
|
151
|
+
"""Load configuration from a YAML file."""
|
|
152
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
153
|
+
data = yaml.safe_load(f)
|
|
154
|
+
return cls.from_dict(data)
|
|
155
|
+
|
|
156
|
+
@classmethod
|
|
157
|
+
def from_json_string(cls, json_str: str) -> Config:
|
|
158
|
+
"""Load configuration from a JSON string."""
|
|
159
|
+
data = json.loads(json_str)
|
|
160
|
+
return cls.from_dict(data)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def load_config() -> Config:
|
|
164
|
+
"""
|
|
165
|
+
Load configuration from environment variables.
|
|
166
|
+
|
|
167
|
+
Configuration is loaded from one of these sources (in priority order):
|
|
168
|
+
1. CONFIG_FILE env var - path to a YAML config file
|
|
169
|
+
2. CONFIG_JSON env var - JSON string with full configuration
|
|
170
|
+
|
|
171
|
+
Raises:
|
|
172
|
+
ValueError: If no valid configuration is found
|
|
173
|
+
"""
|
|
174
|
+
# Try CONFIG_FILE first
|
|
175
|
+
config_file = os.getenv("CONFIG_FILE")
|
|
176
|
+
if config_file:
|
|
177
|
+
if not Path(config_file).exists():
|
|
178
|
+
raise ValueError(f"CONFIG_FILE not found: {config_file}")
|
|
179
|
+
return Config.from_yaml_file(config_file)
|
|
180
|
+
|
|
181
|
+
# Try CONFIG_JSON
|
|
182
|
+
config_json = os.getenv("CONFIG_JSON")
|
|
183
|
+
if config_json:
|
|
184
|
+
return Config.from_json_string(config_json)
|
|
185
|
+
|
|
186
|
+
# No configuration provided
|
|
187
|
+
raise ValueError(
|
|
188
|
+
"No configuration found. Set CONFIG_FILE (path to YAML config) "
|
|
189
|
+
"or CONFIG_JSON (JSON configuration string)."
|
|
190
|
+
)
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"""FFmpeg encoding logic for audio-transcode-watcher."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import os
|
|
7
|
+
import subprocess
|
|
8
|
+
|
|
9
|
+
from .config import OutputConfig
|
|
10
|
+
from .utils import nfc_path
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def build_ffmpeg_command(
|
|
16
|
+
source: str,
|
|
17
|
+
dest: str,
|
|
18
|
+
output_config: OutputConfig,
|
|
19
|
+
) -> list[str]:
|
|
20
|
+
"""
|
|
21
|
+
Build an FFmpeg command for transcoding.
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
source: Path to source audio file
|
|
25
|
+
dest: Path to destination file
|
|
26
|
+
output_config: Output configuration
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
FFmpeg command as list of arguments
|
|
30
|
+
"""
|
|
31
|
+
source = nfc_path(source)
|
|
32
|
+
dest = nfc_path(dest)
|
|
33
|
+
|
|
34
|
+
# Common arguments
|
|
35
|
+
cmd = [
|
|
36
|
+
"ffmpeg", "-loglevel", "error", "-y",
|
|
37
|
+
"-i", source,
|
|
38
|
+
"-map", "0:a:0", # First audio stream
|
|
39
|
+
]
|
|
40
|
+
|
|
41
|
+
# Add video/artwork mapping if enabled
|
|
42
|
+
if output_config.include_artwork:
|
|
43
|
+
cmd.extend(["-map", "0:v:0?"]) # First video/image stream (optional)
|
|
44
|
+
|
|
45
|
+
# Copy metadata
|
|
46
|
+
cmd.extend(["-map_metadata", "0"])
|
|
47
|
+
|
|
48
|
+
# Codec-specific options
|
|
49
|
+
codec = output_config.codec
|
|
50
|
+
|
|
51
|
+
if codec == "alac":
|
|
52
|
+
cmd.extend(["-c:a", "alac"])
|
|
53
|
+
if output_config.include_artwork:
|
|
54
|
+
cmd.extend(["-c:v", "copy"])
|
|
55
|
+
# Ensure album_artist is mapped correctly for M4A (aART tag)
|
|
56
|
+
cmd.extend(["-movflags", "+faststart", "-f", "mp4"])
|
|
57
|
+
|
|
58
|
+
elif codec == "aac":
|
|
59
|
+
cmd.extend(["-c:a", "aac", "-b:a", output_config.bitrate])
|
|
60
|
+
if output_config.include_artwork:
|
|
61
|
+
cmd.extend(["-c:v", "copy"])
|
|
62
|
+
cmd.extend(["-movflags", "+faststart", "-f", "mp4"])
|
|
63
|
+
|
|
64
|
+
elif codec == "mp3":
|
|
65
|
+
cmd.extend(["-c:a", "libmp3lame", "-b:a", output_config.bitrate])
|
|
66
|
+
if output_config.include_artwork:
|
|
67
|
+
# MP3 needs mjpeg for ID3 APIC artwork
|
|
68
|
+
cmd.extend(["-c:v", "mjpeg"])
|
|
69
|
+
cmd.extend(["-id3v2_version", "3", "-write_id3v2", "1", "-f", "mp3"])
|
|
70
|
+
|
|
71
|
+
elif codec == "opus":
|
|
72
|
+
cmd.extend(["-c:a", "libopus", "-b:a", output_config.bitrate])
|
|
73
|
+
cmd.extend(["-f", "opus"])
|
|
74
|
+
|
|
75
|
+
elif codec == "flac":
|
|
76
|
+
cmd.extend(["-c:a", "flac"])
|
|
77
|
+
if output_config.include_artwork:
|
|
78
|
+
cmd.extend(["-c:v", "copy"])
|
|
79
|
+
cmd.extend(["-f", "flac"])
|
|
80
|
+
|
|
81
|
+
elif codec == "wav":
|
|
82
|
+
cmd.extend(["-c:a", "pcm_s16le", "-f", "wav"])
|
|
83
|
+
|
|
84
|
+
else:
|
|
85
|
+
raise ValueError(f"Unsupported codec: {codec}")
|
|
86
|
+
|
|
87
|
+
cmd.append(dest)
|
|
88
|
+
return cmd
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _remove_artwork_from_command(cmd: list[str]) -> list[str]:
|
|
92
|
+
"""
|
|
93
|
+
Remove artwork-related options from an FFmpeg command.
|
|
94
|
+
|
|
95
|
+
Used for retry when artwork encoding fails.
|
|
96
|
+
"""
|
|
97
|
+
filtered = []
|
|
98
|
+
i = 0
|
|
99
|
+
|
|
100
|
+
while i < len(cmd):
|
|
101
|
+
arg = cmd[i]
|
|
102
|
+
|
|
103
|
+
# Skip -map 0:v:0? pair
|
|
104
|
+
if arg == "-map" and i + 1 < len(cmd) and cmd[i + 1] == "0:v:0?":
|
|
105
|
+
i += 2
|
|
106
|
+
continue
|
|
107
|
+
|
|
108
|
+
# Skip -c:v and its value
|
|
109
|
+
if arg == "-c:v":
|
|
110
|
+
i += 2
|
|
111
|
+
continue
|
|
112
|
+
|
|
113
|
+
# Skip -vf and its value
|
|
114
|
+
if arg.startswith("-vf"):
|
|
115
|
+
i += 2
|
|
116
|
+
continue
|
|
117
|
+
|
|
118
|
+
filtered.append(arg)
|
|
119
|
+
i += 1
|
|
120
|
+
|
|
121
|
+
return filtered
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def atomic_ffmpeg_encode(
|
|
125
|
+
cmd: list[str],
|
|
126
|
+
final_dest: str,
|
|
127
|
+
retry_without_artwork: bool = True,
|
|
128
|
+
) -> int:
|
|
129
|
+
"""
|
|
130
|
+
Run FFmpeg with atomic output (write to temp, then rename).
|
|
131
|
+
|
|
132
|
+
Args:
|
|
133
|
+
cmd: FFmpeg command (last element is destination)
|
|
134
|
+
final_dest: Final destination path
|
|
135
|
+
retry_without_artwork: If True, retry without artwork on failure
|
|
136
|
+
|
|
137
|
+
Returns:
|
|
138
|
+
Return code (0 for success)
|
|
139
|
+
"""
|
|
140
|
+
final_dest = nfc_path(final_dest)
|
|
141
|
+
dest_dir = os.path.dirname(final_dest)
|
|
142
|
+
os.makedirs(dest_dir, exist_ok=True)
|
|
143
|
+
|
|
144
|
+
tmp_dest = final_dest + ".tmp__ff"
|
|
145
|
+
|
|
146
|
+
# Clean up any stale temp file
|
|
147
|
+
try:
|
|
148
|
+
if os.path.exists(tmp_dest):
|
|
149
|
+
os.remove(tmp_dest)
|
|
150
|
+
except Exception:
|
|
151
|
+
pass
|
|
152
|
+
|
|
153
|
+
# Replace destination with temp path
|
|
154
|
+
cmd = list(cmd)
|
|
155
|
+
cmd[-1] = tmp_dest
|
|
156
|
+
|
|
157
|
+
logger.info("► %s", " ".join(cmd))
|
|
158
|
+
proc = subprocess.run(cmd, capture_output=True)
|
|
159
|
+
rc = proc.returncode
|
|
160
|
+
|
|
161
|
+
if rc == 0:
|
|
162
|
+
try:
|
|
163
|
+
os.replace(tmp_dest, final_dest)
|
|
164
|
+
return 0
|
|
165
|
+
except Exception as e:
|
|
166
|
+
logger.error("Atomic replace failed for %s: %s", final_dest, e)
|
|
167
|
+
_cleanup_temp(tmp_dest)
|
|
168
|
+
return 1
|
|
169
|
+
|
|
170
|
+
# Handle failure
|
|
171
|
+
logger.error("FFmpeg failed (rc=%s) for %s", rc, final_dest)
|
|
172
|
+
stderr = proc.stderr.decode("utf-8", errors="ignore") if proc.stderr else ""
|
|
173
|
+
_cleanup_temp(tmp_dest)
|
|
174
|
+
|
|
175
|
+
# Retry without artwork if error seems artwork-related
|
|
176
|
+
artwork_error_hints = ["vf#", "vist#", "VipsJpeg", "png", "mjpeg", "decode"]
|
|
177
|
+
if retry_without_artwork and any(h in stderr.lower() for h in artwork_error_hints):
|
|
178
|
+
logger.warning("Retrying without cover art for %s", final_dest)
|
|
179
|
+
|
|
180
|
+
filtered_cmd = _remove_artwork_from_command(cmd)
|
|
181
|
+
filtered_cmd[-1] = tmp_dest
|
|
182
|
+
|
|
183
|
+
logger.info("► (retry) %s", " ".join(filtered_cmd))
|
|
184
|
+
proc2 = subprocess.run(filtered_cmd, capture_output=True)
|
|
185
|
+
rc = proc2.returncode
|
|
186
|
+
|
|
187
|
+
if rc == 0:
|
|
188
|
+
try:
|
|
189
|
+
os.replace(tmp_dest, final_dest)
|
|
190
|
+
return 0
|
|
191
|
+
except Exception as e:
|
|
192
|
+
logger.error("Atomic replace failed for %s: %s", final_dest, e)
|
|
193
|
+
_cleanup_temp(tmp_dest)
|
|
194
|
+
return 1
|
|
195
|
+
|
|
196
|
+
logger.error("FFmpeg retry also failed (rc=%s) for %s", rc, final_dest)
|
|
197
|
+
_cleanup_temp(tmp_dest)
|
|
198
|
+
|
|
199
|
+
return rc
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _cleanup_temp(path: str) -> None:
|
|
203
|
+
"""Clean up a temporary file."""
|
|
204
|
+
try:
|
|
205
|
+
if os.path.exists(path):
|
|
206
|
+
os.remove(path)
|
|
207
|
+
except Exception:
|
|
208
|
+
pass
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""Automatic lyrics fetching with syncedlyrics and Whisper fallback."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
import mutagen
|
|
11
|
+
import syncedlyrics
|
|
12
|
+
|
|
13
|
+
from .utils import nfc, nfc_path
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
# Lazy-loaded Whisper model (heavyweight, only load once when needed)
|
|
18
|
+
_whisper_model = None
|
|
19
|
+
_whisper_load_failed = False
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _get_whisper_model(model_name: str = "base"):
|
|
23
|
+
"""Lazy-load and cache the Whisper model."""
|
|
24
|
+
global _whisper_model, _whisper_load_failed
|
|
25
|
+
if _whisper_load_failed:
|
|
26
|
+
return None
|
|
27
|
+
if _whisper_model is not None:
|
|
28
|
+
return _whisper_model
|
|
29
|
+
try:
|
|
30
|
+
import whisper
|
|
31
|
+
|
|
32
|
+
logger.info("Loading Whisper model '%s' (first use, may take a moment)...", model_name)
|
|
33
|
+
_whisper_model = whisper.load_model(model_name)
|
|
34
|
+
logger.info("Whisper model '%s' loaded", model_name)
|
|
35
|
+
return _whisper_model
|
|
36
|
+
except Exception:
|
|
37
|
+
logger.warning("Failed to load Whisper model", exc_info=True)
|
|
38
|
+
_whisper_load_failed = True
|
|
39
|
+
return None
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _segments_to_lrc(segments: list[dict]) -> str:
|
|
43
|
+
"""Convert Whisper transcript segments to LRC format."""
|
|
44
|
+
lines = []
|
|
45
|
+
for seg in segments:
|
|
46
|
+
start = seg.get("start", 0.0)
|
|
47
|
+
text = seg.get("text", "").strip()
|
|
48
|
+
if not text:
|
|
49
|
+
continue
|
|
50
|
+
mins = int(start // 60)
|
|
51
|
+
secs = start % 60
|
|
52
|
+
lines.append(f"[{mins:02d}:{secs:05.2f}] {text}")
|
|
53
|
+
return "\n".join(lines)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _transcribe_with_whisper(filepath: str, model_name: str = "base") -> str | None:
|
|
57
|
+
"""
|
|
58
|
+
Transcribe audio to synced lyrics using Whisper.
|
|
59
|
+
|
|
60
|
+
Args:
|
|
61
|
+
filepath: Path to the audio file.
|
|
62
|
+
model_name: Whisper model size (tiny, base, small, medium, large).
|
|
63
|
+
|
|
64
|
+
Returns:
|
|
65
|
+
LRC-formatted string, or None on failure.
|
|
66
|
+
"""
|
|
67
|
+
model = _get_whisper_model(model_name)
|
|
68
|
+
if model is None:
|
|
69
|
+
return None
|
|
70
|
+
|
|
71
|
+
try:
|
|
72
|
+
logger.info("Transcribing with Whisper: %s", Path(filepath).name)
|
|
73
|
+
result = model.transcribe(filepath, verbose=False)
|
|
74
|
+
segments = result.get("segments", [])
|
|
75
|
+
if not segments:
|
|
76
|
+
logger.info("Whisper produced no segments for: %s", Path(filepath).name)
|
|
77
|
+
return None
|
|
78
|
+
lrc = _segments_to_lrc(segments)
|
|
79
|
+
logger.info(
|
|
80
|
+
"Whisper transcribed %d segments for: %s", len(segments), Path(filepath).name
|
|
81
|
+
)
|
|
82
|
+
return lrc
|
|
83
|
+
except Exception:
|
|
84
|
+
logger.warning("Whisper transcription failed for: %s", filepath, exc_info=True)
|
|
85
|
+
return None
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def extract_metadata(filepath: str) -> tuple[str, str] | None:
|
|
89
|
+
"""
|
|
90
|
+
Extract artist and title from an audio file.
|
|
91
|
+
|
|
92
|
+
Tries embedded metadata first (mutagen), then falls back to parsing
|
|
93
|
+
the filename as "Artist - Title.ext".
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
Tuple of (artist, title) or None if not extractable.
|
|
97
|
+
"""
|
|
98
|
+
# Try embedded metadata via mutagen
|
|
99
|
+
try:
|
|
100
|
+
audio = mutagen.File(filepath, easy=True)
|
|
101
|
+
if audio and audio.tags:
|
|
102
|
+
artists = audio.tags.get("artist", [])
|
|
103
|
+
titles = audio.tags.get("title", [])
|
|
104
|
+
if artists and titles:
|
|
105
|
+
artist = artists[0].strip()
|
|
106
|
+
title = titles[0].strip()
|
|
107
|
+
if artist and title:
|
|
108
|
+
return artist, title
|
|
109
|
+
except Exception:
|
|
110
|
+
logger.debug("Could not read metadata from %s", filepath)
|
|
111
|
+
|
|
112
|
+
# Fallback: parse filename "Artist - Title.ext"
|
|
113
|
+
stem = Path(filepath).stem
|
|
114
|
+
# Strip leading track numbers like "01 - ", "01. ", "1 "
|
|
115
|
+
stem = re.sub(r"^\d+[\s.\-]+\s*", "", stem).strip()
|
|
116
|
+
if " - " in stem:
|
|
117
|
+
artist, title = stem.split(" - ", 1)
|
|
118
|
+
artist = artist.strip()
|
|
119
|
+
title = title.strip()
|
|
120
|
+
if artist and title:
|
|
121
|
+
return artist, title
|
|
122
|
+
|
|
123
|
+
return None
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def fetch_lyrics_for_file(
|
|
127
|
+
filepath: str,
|
|
128
|
+
whisper_fallback: bool = True,
|
|
129
|
+
whisper_model: str = "base",
|
|
130
|
+
) -> str | None:
|
|
131
|
+
"""
|
|
132
|
+
Fetch synced lyrics (.lrc) for an audio file if not already present.
|
|
133
|
+
|
|
134
|
+
Strategy:
|
|
135
|
+
1. Check if .lrc sidecar already exists -> skip
|
|
136
|
+
2. Try syncedlyrics providers (Musixmatch, LRCLIB, NetEase)
|
|
137
|
+
3. If nothing found and whisper_fallback enabled, transcribe locally
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
filepath: Path to the audio file.
|
|
141
|
+
whisper_fallback: Use Whisper local transcription as fallback.
|
|
142
|
+
whisper_model: Whisper model size (tiny, base, small, medium, large).
|
|
143
|
+
|
|
144
|
+
Returns:
|
|
145
|
+
Path to the written .lrc file, or None if lyrics were not found
|
|
146
|
+
or already existed.
|
|
147
|
+
"""
|
|
148
|
+
filepath = nfc_path(filepath)
|
|
149
|
+
lrc_path = nfc_path(str(Path(filepath).with_suffix(".lrc")))
|
|
150
|
+
|
|
151
|
+
# Already has lyrics
|
|
152
|
+
if os.path.isfile(lrc_path):
|
|
153
|
+
return None
|
|
154
|
+
|
|
155
|
+
lrc_content: str | None = None
|
|
156
|
+
|
|
157
|
+
# Step 1: Try syncedlyrics
|
|
158
|
+
meta = extract_metadata(filepath)
|
|
159
|
+
if meta is not None:
|
|
160
|
+
artist, title = meta
|
|
161
|
+
query = f"{artist} {title}"
|
|
162
|
+
try:
|
|
163
|
+
lrc_content = syncedlyrics.search(query)
|
|
164
|
+
except Exception:
|
|
165
|
+
logger.warning("syncedlyrics search failed for: %s", query, exc_info=True)
|
|
166
|
+
|
|
167
|
+
if lrc_content:
|
|
168
|
+
return _write_lrc(lrc_path, lrc_content, f"{artist} - {title}", "syncedlyrics")
|
|
169
|
+
|
|
170
|
+
logger.info("No lyrics found via syncedlyrics for: %s - %s", artist, title)
|
|
171
|
+
else:
|
|
172
|
+
logger.debug("Cannot extract metadata for lyrics: %s", filepath)
|
|
173
|
+
|
|
174
|
+
# Step 2: Whisper fallback
|
|
175
|
+
if whisper_fallback:
|
|
176
|
+
lrc_content = _transcribe_with_whisper(filepath, whisper_model)
|
|
177
|
+
if lrc_content:
|
|
178
|
+
label = f"{meta[0]} - {meta[1]}" if meta else Path(filepath).stem
|
|
179
|
+
return _write_lrc(lrc_path, lrc_content, label, "whisper")
|
|
180
|
+
|
|
181
|
+
return None
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _write_lrc(lrc_path: str, content: str, label: str, source: str) -> str | None:
|
|
185
|
+
"""Write LRC content to disk."""
|
|
186
|
+
try:
|
|
187
|
+
with open(lrc_path, "w", encoding="utf-8") as f:
|
|
188
|
+
f.write(content)
|
|
189
|
+
logger.info("♫ lyrics saved (%s): %s → %s", source, label, lrc_path)
|
|
190
|
+
return lrc_path
|
|
191
|
+
except Exception:
|
|
192
|
+
logger.error("Failed to write lyrics file: %s", lrc_path, exc_info=True)
|
|
193
|
+
return None
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Main entry point for audio-transcode-watcher."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import gc
|
|
6
|
+
import logging
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
import time
|
|
10
|
+
|
|
11
|
+
from .config import load_config
|
|
12
|
+
from .sync import initial_sync
|
|
13
|
+
from .utils import nfc_path
|
|
14
|
+
from .watcher import start_watcher
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def setup_logging() -> None:
|
|
18
|
+
"""Configure logging for the application."""
|
|
19
|
+
# Ensure Unicode output works
|
|
20
|
+
try:
|
|
21
|
+
if hasattr(sys.stdout, "reconfigure"):
|
|
22
|
+
sys.stdout.reconfigure(encoding="utf-8")
|
|
23
|
+
if hasattr(sys.stderr, "reconfigure"):
|
|
24
|
+
sys.stderr.reconfigure(encoding="utf-8")
|
|
25
|
+
except Exception:
|
|
26
|
+
pass
|
|
27
|
+
|
|
28
|
+
logging.basicConfig(
|
|
29
|
+
level=logging.INFO,
|
|
30
|
+
format="%(asctime)s %(levelname)s: %(message)s",
|
|
31
|
+
datefmt="%H:%M:%S",
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def main() -> int:
|
|
36
|
+
"""Main entry point."""
|
|
37
|
+
setup_logging()
|
|
38
|
+
logger = logging.getLogger(__name__)
|
|
39
|
+
|
|
40
|
+
# Load configuration
|
|
41
|
+
try:
|
|
42
|
+
config = load_config()
|
|
43
|
+
except ValueError as e:
|
|
44
|
+
logger.error("Configuration error: %s", e)
|
|
45
|
+
return 1
|
|
46
|
+
|
|
47
|
+
# Normalize paths
|
|
48
|
+
config.source_path = nfc_path(config.source_path)
|
|
49
|
+
for output in config.outputs:
|
|
50
|
+
output.path = nfc_path(output.path)
|
|
51
|
+
|
|
52
|
+
# Validate source exists
|
|
53
|
+
if not os.path.isdir(config.source_path):
|
|
54
|
+
logger.error("Source directory does not exist: %s", config.source_path)
|
|
55
|
+
return 1
|
|
56
|
+
|
|
57
|
+
# Log configuration
|
|
58
|
+
logger.info("Source: %s", config.source_path)
|
|
59
|
+
for output in config.outputs:
|
|
60
|
+
logger.info(
|
|
61
|
+
"Output: %s (%s%s) -> %s",
|
|
62
|
+
output.name,
|
|
63
|
+
output.codec,
|
|
64
|
+
f" {output.bitrate}" if output.bitrate else "",
|
|
65
|
+
output.path,
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
# Perform initial sync
|
|
69
|
+
initial_sync(config)
|
|
70
|
+
|
|
71
|
+
# Start watcher
|
|
72
|
+
observer = start_watcher(config)
|
|
73
|
+
logger.info("Watching %s …", config.source_path)
|
|
74
|
+
|
|
75
|
+
# Periodic sync interval (check for missing outputs every 5 minutes)
|
|
76
|
+
sync_interval = 300 # 5 minutes
|
|
77
|
+
last_sync = time.time()
|
|
78
|
+
|
|
79
|
+
try:
|
|
80
|
+
while True:
|
|
81
|
+
time.sleep(10)
|
|
82
|
+
|
|
83
|
+
# Periodic sync to catch deleted outputs
|
|
84
|
+
if time.time() - last_sync >= sync_interval:
|
|
85
|
+
logger.info("Periodic sync check…")
|
|
86
|
+
initial_sync(config)
|
|
87
|
+
gc.collect()
|
|
88
|
+
last_sync = time.time()
|
|
89
|
+
except KeyboardInterrupt:
|
|
90
|
+
observer.stop()
|
|
91
|
+
|
|
92
|
+
observer.join()
|
|
93
|
+
return 0
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
if __name__ == "__main__":
|
|
97
|
+
sys.exit(main())
|