mpup 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mpup/__init__.py +3 -0
- mpup/__main__.py +5 -0
- mpup/background.py +122 -0
- mpup/cli.py +159 -0
- mpup/config.py +187 -0
- mpup/cover.py +244 -0
- mpup/discovery.py +33 -0
- mpup/enriched_audio.py +147 -0
- mpup/itunes.py +112 -0
- mpup/lyrics.py +190 -0
- mpup/lyrics_from_qq.py +46 -0
- mpup/metadata.py +159 -0
- mpup/pipeline.py +305 -0
- mpup/remotion_bridge.py +130 -0
- mpup/render_job.py +118 -0
- mpup/video.py +160 -0
- mpup/workspace.py +92 -0
- mpup-0.1.0.dist-info/METADATA +135 -0
- mpup-0.1.0.dist-info/RECORD +21 -0
- mpup-0.1.0.dist-info/WHEEL +4 -0
- mpup-0.1.0.dist-info/entry_points.txt +2 -0
mpup/lyrics.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""Decode and normalize synchronized LRC lyrics."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import re
|
|
7
|
+
from collections.abc import Callable
|
|
8
|
+
from dataclasses import asdict, dataclass
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Literal, Protocol, Sequence
|
|
11
|
+
|
|
12
|
+
from mpup.metadata import AudioMetadata
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
_TIMESTAMP = re.compile(r"\[(?P<minutes>\d+):(?P<seconds>[0-5]\d)(?:[.:](?P<fraction>\d{2,3}))?\]")
|
|
16
|
+
_OFFSET = re.compile(r"^\s*\[offset:(?P<offset>[+-]?\d+)\]\s*$", re.IGNORECASE)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class LyricCue:
|
|
21
|
+
text: str
|
|
22
|
+
start_ms: int
|
|
23
|
+
end_ms: int
|
|
24
|
+
timestamp_ms: None = None
|
|
25
|
+
confidence: None = None
|
|
26
|
+
|
|
27
|
+
def as_caption(self) -> dict[str, str | int | None]:
|
|
28
|
+
values = asdict(self)
|
|
29
|
+
return {
|
|
30
|
+
"text": values["text"],
|
|
31
|
+
"startMs": values["start_ms"],
|
|
32
|
+
"endMs": values["end_ms"],
|
|
33
|
+
"timestampMs": None,
|
|
34
|
+
"confidence": None,
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class LyricsProvider(Protocol):
|
|
39
|
+
def fetch(self, metadata: AudioMetadata) -> str | None:
|
|
40
|
+
"""Return LRC text or no match."""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class NullLyricsProvider:
|
|
44
|
+
"""Default provider that deliberately performs no network access."""
|
|
45
|
+
|
|
46
|
+
def fetch(self, metadata: AudioMetadata) -> None:
|
|
47
|
+
del metadata
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class PlatformLyricsProvider:
|
|
52
|
+
"""Retrieve lyrics from a selected music platform without leaking failures."""
|
|
53
|
+
|
|
54
|
+
def __init__(self, source: str = "qq") -> None:
|
|
55
|
+
if source != "qq":
|
|
56
|
+
raise ValueError(f"Unsupported lyrics source: {source}")
|
|
57
|
+
self.source = source
|
|
58
|
+
|
|
59
|
+
def fetch(self, metadata: AudioMetadata) -> str | None:
|
|
60
|
+
try:
|
|
61
|
+
from mpup.lyrics_from_qq import fetch_lyrics
|
|
62
|
+
|
|
63
|
+
return fetch_lyrics(metadata)
|
|
64
|
+
except Exception:
|
|
65
|
+
return None
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
LyricsSource = Literal["local", "embedded", "provider", "none"]
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass(frozen=True)
|
|
72
|
+
class LyricsResolution:
|
|
73
|
+
cues: tuple[LyricCue, ...] | None
|
|
74
|
+
source: LyricsSource
|
|
75
|
+
warnings: tuple[str, ...] = ()
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def decode_lrc(payload: bytes) -> str:
|
|
79
|
+
"""Decode local lyrics using the supported deterministic encoding order."""
|
|
80
|
+
for encoding in ("utf-8-sig", "gb18030"):
|
|
81
|
+
try:
|
|
82
|
+
return payload.decode(encoding)
|
|
83
|
+
except UnicodeDecodeError:
|
|
84
|
+
continue
|
|
85
|
+
raise RuntimeError("歌词文件不是 UTF-8 或 GB18030 编码")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _timestamp_ms(match: re.Match[str]) -> int:
|
|
89
|
+
fraction = match.group("fraction") or ""
|
|
90
|
+
milliseconds = int(fraction) * (10 if len(fraction) == 2 else 1) if fraction else 0
|
|
91
|
+
return (int(match.group("minutes")) * 60 + int(match.group("seconds"))) * 1000 + milliseconds
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def parse_lrc(text: str, duration_ms: int) -> tuple[LyricCue, ...]:
|
|
95
|
+
"""Convert LRC text into sorted, bounded caption-compatible cues."""
|
|
96
|
+
if duration_ms <= 0:
|
|
97
|
+
raise ValueError("音频时长必须大于 0")
|
|
98
|
+
|
|
99
|
+
offset_ms = 0
|
|
100
|
+
for line in text.splitlines():
|
|
101
|
+
match = _OFFSET.match(line)
|
|
102
|
+
if match is not None:
|
|
103
|
+
offset_ms = int(match.group("offset"))
|
|
104
|
+
|
|
105
|
+
events: list[tuple[int, int, str]] = []
|
|
106
|
+
ordinal = 0
|
|
107
|
+
for line in text.splitlines():
|
|
108
|
+
timestamps = list(_TIMESTAMP.finditer(line))
|
|
109
|
+
if not timestamps:
|
|
110
|
+
continue
|
|
111
|
+
lyric = _TIMESTAMP.sub("", line).strip()
|
|
112
|
+
if not lyric:
|
|
113
|
+
continue
|
|
114
|
+
for timestamp in timestamps:
|
|
115
|
+
start_ms = max(0, _timestamp_ms(timestamp) + offset_ms)
|
|
116
|
+
if start_ms < duration_ms:
|
|
117
|
+
events.append((start_ms, ordinal, lyric))
|
|
118
|
+
ordinal += 1
|
|
119
|
+
|
|
120
|
+
events.sort(key=lambda event: (event[0], event[1]))
|
|
121
|
+
cues: list[LyricCue] = []
|
|
122
|
+
for index, (start_ms, _ordinal, lyric) in enumerate(events):
|
|
123
|
+
if index + 1 < len(events):
|
|
124
|
+
end_ms = min(duration_ms, events[index + 1][0])
|
|
125
|
+
else:
|
|
126
|
+
end_ms = min(duration_ms, start_ms + 6000)
|
|
127
|
+
cues.append(LyricCue(lyric, start_ms, max(start_ms, end_ms)))
|
|
128
|
+
return tuple(cues)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def lyrics_json(cues: Sequence[LyricCue]) -> str:
|
|
132
|
+
"""Serialize cues with the shared camelCase renderer contract."""
|
|
133
|
+
return json.dumps(
|
|
134
|
+
[cue.as_caption() for cue in cues],
|
|
135
|
+
ensure_ascii=False,
|
|
136
|
+
separators=(",", ":"),
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _lyrics_warning(prefix: str, error: Exception | None = None) -> str:
|
|
141
|
+
if error is None:
|
|
142
|
+
return prefix
|
|
143
|
+
detail = str(error).strip() or error.__class__.__name__
|
|
144
|
+
return f"{prefix}:{detail[-1000:]}"
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def resolve_lyrics(
|
|
148
|
+
audio_path: Path,
|
|
149
|
+
metadata: AudioMetadata,
|
|
150
|
+
*,
|
|
151
|
+
provider: LyricsProvider | None = None,
|
|
152
|
+
is_file: Callable[[Path], bool] = Path.is_file,
|
|
153
|
+
reader: Callable[[Path], bytes] = Path.read_bytes,
|
|
154
|
+
) -> LyricsResolution:
|
|
155
|
+
"""Resolve local, embedded, then provider LRC without failing the song."""
|
|
156
|
+
duration_ms = max(1, round(metadata.duration * 1000))
|
|
157
|
+
warnings: list[str] = []
|
|
158
|
+
|
|
159
|
+
for suffix in (".lrc", ".LRC"):
|
|
160
|
+
candidate = audio_path.with_suffix(suffix)
|
|
161
|
+
if not is_file(candidate):
|
|
162
|
+
continue
|
|
163
|
+
try:
|
|
164
|
+
cues = parse_lrc(decode_lrc(reader(candidate)), duration_ms)
|
|
165
|
+
except Exception as error:
|
|
166
|
+
warnings.append(_lyrics_warning(f"歌词文件 {candidate.name} 不可用", error))
|
|
167
|
+
continue
|
|
168
|
+
if cues:
|
|
169
|
+
return LyricsResolution(cues, "local", tuple(warnings))
|
|
170
|
+
warnings.append(_lyrics_warning(f"歌词文件 {candidate.name} 没有同步时间戳"))
|
|
171
|
+
|
|
172
|
+
if metadata.embedded_lyrics:
|
|
173
|
+
cues = parse_lrc(metadata.embedded_lyrics, duration_ms)
|
|
174
|
+
if cues:
|
|
175
|
+
return LyricsResolution(cues, "embedded", tuple(warnings))
|
|
176
|
+
warnings.append(_lyrics_warning("内嵌歌词没有同步时间戳"))
|
|
177
|
+
|
|
178
|
+
active_provider = provider or NullLyricsProvider()
|
|
179
|
+
if metadata.title and metadata.artist:
|
|
180
|
+
try:
|
|
181
|
+
provider_text = active_provider.fetch(metadata)
|
|
182
|
+
if provider_text:
|
|
183
|
+
cues = parse_lrc(provider_text, duration_ms)
|
|
184
|
+
if cues:
|
|
185
|
+
return LyricsResolution(cues, "provider", tuple(warnings))
|
|
186
|
+
warnings.append(_lyrics_warning("歌词 Provider 返回内容没有同步时间戳"))
|
|
187
|
+
except Exception as error:
|
|
188
|
+
warnings.append(_lyrics_warning("歌词 Provider 不可用", error))
|
|
189
|
+
|
|
190
|
+
return LyricsResolution(None, "none", tuple(warnings))
|
mpup/lyrics_from_qq.py
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""Resolve raw synchronized lyrics from QQ Music metadata."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import base64, json, re
|
|
4
|
+
from urllib.parse import urlencode
|
|
5
|
+
from urllib.request import Request, urlopen
|
|
6
|
+
from mpup.metadata import AudioMetadata
|
|
7
|
+
|
|
8
|
+
_SEARCH = "https://u.y.qq.com/cgi-bin/musicu.fcg"
|
|
9
|
+
_LYRIC = "https://c.y.qq.com/lyric/fcgi-bin/fcg_query_lyric_new.fcg"
|
|
10
|
+
_JSONP = re.compile(r"^[^(]+\((?P<payload>.*)\)\s*$", re.DOTALL)
|
|
11
|
+
_BRACKETS = re.compile(r"[((].*?[))]")
|
|
12
|
+
_SPLIT = re.compile(r"[/,&,、]")
|
|
13
|
+
_CLEAN = re.compile(r"[\s\W_]+", re.UNICODE)
|
|
14
|
+
|
|
15
|
+
def fetch_lyrics(metadata: AudioMetadata, source: str = "qq") -> str | None:
|
|
16
|
+
if source != "qq": raise ValueError(f"Unsupported lyrics source: {source}")
|
|
17
|
+
if not metadata.title or not metadata.artist: return None
|
|
18
|
+
song = _find_song(metadata)
|
|
19
|
+
return _fetch_raw_lrc(song["mid"]) if song else None
|
|
20
|
+
|
|
21
|
+
def _find_song(metadata: AudioMetadata) -> dict[str, object] | None:
|
|
22
|
+
data = {"req_1": {"method":"DoSearchForQQMusicDesktop", "module":"music.search.SearchCgiService", "param":{"num_per_page":20,"page_num":1,"query":f"{metadata.title} {metadata.artist}","search_type":0}}}
|
|
23
|
+
try: songs = _request_json(_SEARCH, json.dumps(data).encode(), {"Referer":"https://y.qq.com","Content-Type":"application/json"})["req_1"]["data"]["body"]["song"]["list"]
|
|
24
|
+
except (KeyError, TypeError): return None
|
|
25
|
+
expected_title, expected_artists = _title(metadata.title), _artists(metadata.artist)
|
|
26
|
+
for song in songs:
|
|
27
|
+
if not isinstance(song, dict) or _title(str(song.get("title", ""))) != expected_title: continue
|
|
28
|
+
artists = _artists("/".join(str(x.get("name", "")) for x in song.get("singer", []) if isinstance(x, dict)))
|
|
29
|
+
if expected_artists & artists and isinstance(song.get("mid"), str): return song
|
|
30
|
+
return None
|
|
31
|
+
|
|
32
|
+
def _fetch_raw_lrc(mid: str) -> str | None:
|
|
33
|
+
query = urlencode({"songmid":mid,"g_tk":"5381","loginUin":"0","hostUin":"0","inCharset":"utf8","outCharset":"utf-8","notice":"0","platform":"yqq","needNewCode":"0"})
|
|
34
|
+
match = _JSONP.match(_request_text(f"{_LYRIC}?{query}", {"Referer":"https://y.qq.com","Cookie":"uin="}))
|
|
35
|
+
if match is None: return None
|
|
36
|
+
try:
|
|
37
|
+
value = json.loads(match.group("payload")).get("lyric", "")
|
|
38
|
+
return base64.b64decode(value, validate=True).decode("utf-8") if value else None
|
|
39
|
+
except (UnicodeDecodeError, ValueError, json.JSONDecodeError, AttributeError): return None
|
|
40
|
+
|
|
41
|
+
def _request_json(url: str, data: bytes, headers: dict[str, str]) -> object: return json.loads(_request(url, data, headers))
|
|
42
|
+
def _request_text(url: str, headers: dict[str, str]) -> str: return _request(url, None, headers)
|
|
43
|
+
def _request(url: str, data: bytes | None, headers: dict[str, str]) -> str:
|
|
44
|
+
with urlopen(Request(url, data=data, headers=headers, method="POST" if data else "GET"), timeout=10) as response: return response.read().decode("utf-8")
|
|
45
|
+
def _title(value: str) -> str: return _CLEAN.sub("", _BRACKETS.sub("", value).casefold())
|
|
46
|
+
def _artists(value: str) -> set[str]: return {_CLEAN.sub("", x.casefold()) for x in _SPLIT.split(value) if _CLEAN.sub("", x.casefold())}
|
mpup/metadata.py
ADDED
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Read audio metadata and validate media tool availability."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import math
|
|
7
|
+
import shutil
|
|
8
|
+
import subprocess
|
|
9
|
+
from collections.abc import Callable
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
Runner = Callable[..., subprocess.CompletedProcess[str]]
|
|
16
|
+
REMOTION_ENTRY = Path(__file__).resolve().parents[2] / "renderer" / "src" / "index.ts"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class AudioMetadata:
|
|
21
|
+
duration: float
|
|
22
|
+
title: str | None
|
|
23
|
+
artist: str | None
|
|
24
|
+
cover_stream_index: int | None = None
|
|
25
|
+
embedded_lyrics: str | None = None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def require_media_tools(
|
|
29
|
+
which: Callable[[str], str | None] = shutil.which,
|
|
30
|
+
) -> tuple[str, str]:
|
|
31
|
+
"""Return FFmpeg and FFprobe paths or report the first missing tool."""
|
|
32
|
+
resolved: list[str] = []
|
|
33
|
+
for name in ("ffmpeg", "ffprobe"):
|
|
34
|
+
path = which(name)
|
|
35
|
+
if path is None:
|
|
36
|
+
raise RuntimeError(f"找不到 {name},请先安装 FFmpeg 并加入 PATH")
|
|
37
|
+
resolved.append(path)
|
|
38
|
+
return resolved[0], resolved[1]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def require_remotion_runtime(
|
|
42
|
+
entry: Path = REMOTION_ENTRY,
|
|
43
|
+
*,
|
|
44
|
+
which: Callable[[str], str | None] = shutil.which,
|
|
45
|
+
is_file: Callable[[Path], bool] = Path.is_file,
|
|
46
|
+
) -> tuple[str, Path]:
|
|
47
|
+
"""Return the Node executable and Remotion composition entrypoint."""
|
|
48
|
+
node = which("node")
|
|
49
|
+
if node is None:
|
|
50
|
+
raise RuntimeError("找不到 Node.js,请先安装 Node.js 并加入 PATH")
|
|
51
|
+
if not is_file(entry):
|
|
52
|
+
raise RuntimeError(f"找不到 Remotion 渲染入口:{entry}")
|
|
53
|
+
return node, entry
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _normalized_tags(container: object) -> dict[str, str]:
|
|
57
|
+
if not isinstance(container, dict):
|
|
58
|
+
return {}
|
|
59
|
+
raw_tags = container.get("tags", {})
|
|
60
|
+
if not isinstance(raw_tags, dict):
|
|
61
|
+
return {}
|
|
62
|
+
tags: dict[str, str] = {}
|
|
63
|
+
for key, value in raw_tags.items():
|
|
64
|
+
if not isinstance(key, str) or not isinstance(value, str):
|
|
65
|
+
continue
|
|
66
|
+
cleaned = value.strip()
|
|
67
|
+
if cleaned:
|
|
68
|
+
tags[key.casefold()] = cleaned
|
|
69
|
+
return tags
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _load_payload(stdout: str) -> dict[str, Any]:
|
|
73
|
+
try:
|
|
74
|
+
payload = json.loads(stdout)
|
|
75
|
+
except (json.JSONDecodeError, TypeError) as error:
|
|
76
|
+
raise RuntimeError("FFprobe 返回了无效 JSON") from error
|
|
77
|
+
if not isinstance(payload, dict):
|
|
78
|
+
raise RuntimeError("FFprobe 返回了无效 JSON 对象")
|
|
79
|
+
return payload
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _cover_stream_index(streams: list[object]) -> int | None:
|
|
83
|
+
for stream in streams:
|
|
84
|
+
if not isinstance(stream, dict) or stream.get("codec_type") != "video":
|
|
85
|
+
continue
|
|
86
|
+
disposition = stream.get("disposition")
|
|
87
|
+
if not isinstance(disposition, dict) or disposition.get("attached_pic") != 1:
|
|
88
|
+
continue
|
|
89
|
+
index = stream.get("index")
|
|
90
|
+
if type(index) is int and index >= 0:
|
|
91
|
+
return index
|
|
92
|
+
return None
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _embedded_lyrics(tags: dict[str, str]) -> str | None:
|
|
96
|
+
for name in ("syncedlyrics", "synced lyrics", "lyrics", "unsyncedlyrics"):
|
|
97
|
+
if name in tags:
|
|
98
|
+
return tags[name]
|
|
99
|
+
for name, value in tags.items():
|
|
100
|
+
if name.startswith(("lyrics-", "syncedlyrics-", "unsyncedlyrics-")):
|
|
101
|
+
return value
|
|
102
|
+
return None
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def probe_audio(
|
|
106
|
+
path: Path,
|
|
107
|
+
ffprobe: str = "ffprobe",
|
|
108
|
+
runner: Runner = subprocess.run,
|
|
109
|
+
) -> AudioMetadata:
|
|
110
|
+
"""Read duration, title, and artist from an audio file header."""
|
|
111
|
+
command = [
|
|
112
|
+
ffprobe,
|
|
113
|
+
"-v",
|
|
114
|
+
"error",
|
|
115
|
+
"-show_entries",
|
|
116
|
+
"format=duration:format_tags:stream=index,codec_type:"
|
|
117
|
+
"stream_tags:stream_disposition=attached_pic",
|
|
118
|
+
"-of",
|
|
119
|
+
"json",
|
|
120
|
+
str(path),
|
|
121
|
+
]
|
|
122
|
+
result = runner(command, capture_output=True, text=True, check=False)
|
|
123
|
+
if result.returncode != 0:
|
|
124
|
+
detail = result.stderr.strip() or "未知错误"
|
|
125
|
+
raise RuntimeError(f"FFprobe 读取失败:{detail[-2000:]}")
|
|
126
|
+
|
|
127
|
+
payload = _load_payload(result.stdout)
|
|
128
|
+
raw_streams = payload.get("streams", [])
|
|
129
|
+
streams = raw_streams if isinstance(raw_streams, list) else []
|
|
130
|
+
audio_stream = next(
|
|
131
|
+
(
|
|
132
|
+
stream
|
|
133
|
+
for stream in streams
|
|
134
|
+
if isinstance(stream, dict) and stream.get("codec_type") == "audio"
|
|
135
|
+
),
|
|
136
|
+
None,
|
|
137
|
+
)
|
|
138
|
+
if audio_stream is None:
|
|
139
|
+
raise RuntimeError("输入文件不包含音频流")
|
|
140
|
+
|
|
141
|
+
format_info = payload.get("format", {})
|
|
142
|
+
if not isinstance(format_info, dict):
|
|
143
|
+
format_info = {}
|
|
144
|
+
try:
|
|
145
|
+
duration = float(format_info.get("duration"))
|
|
146
|
+
except (TypeError, ValueError) as error:
|
|
147
|
+
raise RuntimeError("输入文件缺少有效时长") from error
|
|
148
|
+
if not math.isfinite(duration) or duration <= 0:
|
|
149
|
+
raise RuntimeError("输入文件缺少有效时长")
|
|
150
|
+
|
|
151
|
+
tags = _normalized_tags(audio_stream)
|
|
152
|
+
tags.update(_normalized_tags(format_info))
|
|
153
|
+
return AudioMetadata(
|
|
154
|
+
duration=duration,
|
|
155
|
+
title=tags.get("title"),
|
|
156
|
+
artist=tags.get("artist"),
|
|
157
|
+
cover_stream_index=_cover_stream_index(streams),
|
|
158
|
+
embedded_lyrics=_embedded_lyrics(tags),
|
|
159
|
+
)
|
mpup/pipeline.py
ADDED
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
"""Process audio files independently and summarize batch outcomes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Callable, Sequence
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Literal
|
|
9
|
+
|
|
10
|
+
from mpup.background import render_background
|
|
11
|
+
from mpup.config import TemplateConfig
|
|
12
|
+
from mpup.cover import BatchCoverResolver
|
|
13
|
+
from mpup.enriched_audio import enriched_mp3_path, write_enriched_mp3
|
|
14
|
+
from mpup.lyrics import resolve_lyrics
|
|
15
|
+
from mpup.metadata import AudioMetadata, probe_audio
|
|
16
|
+
from mpup.remotion_bridge import RenderRequest, run_remotion_batch
|
|
17
|
+
from mpup.video import publish_remotion_video
|
|
18
|
+
from mpup.workspace import BatchWorkspace
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
Status = Literal["success", "skipped", "failed"]
|
|
22
|
+
ProbeFn = Callable[[Path, str], AudioMetadata]
|
|
23
|
+
BackgroundRenderer = Callable[..., None]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True)
|
|
27
|
+
class SongResult:
|
|
28
|
+
input_path: Path
|
|
29
|
+
status: Status
|
|
30
|
+
message: str
|
|
31
|
+
background_path: Path | None = None
|
|
32
|
+
video_path: Path | None = None
|
|
33
|
+
cover_source: str | None = None
|
|
34
|
+
lyrics_source: str | None = None
|
|
35
|
+
warnings: tuple[str, ...] = ()
|
|
36
|
+
enriched_audio_path: Path | None = None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True)
|
|
40
|
+
class BatchResult:
|
|
41
|
+
results: tuple[SongResult, ...]
|
|
42
|
+
|
|
43
|
+
@property
|
|
44
|
+
def successes(self) -> tuple[SongResult, ...]:
|
|
45
|
+
return tuple(item for item in self.results if item.status == "success")
|
|
46
|
+
|
|
47
|
+
@property
|
|
48
|
+
def skipped(self) -> tuple[SongResult, ...]:
|
|
49
|
+
return tuple(item for item in self.results if item.status == "skipped")
|
|
50
|
+
|
|
51
|
+
@property
|
|
52
|
+
def failed(self) -> tuple[SongResult, ...]:
|
|
53
|
+
return tuple(item for item in self.results if item.status == "failed")
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def exit_code(self) -> int:
|
|
57
|
+
return 0 if len(self.successes) == len(self.results) else 1
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass(frozen=True)
|
|
61
|
+
class ProgressEvent:
|
|
62
|
+
"""A visible processing milestone for one song or the entire batch."""
|
|
63
|
+
|
|
64
|
+
stage: str
|
|
65
|
+
input_path: Path | None = None
|
|
66
|
+
status: Status | None = None
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
ProgressReporter = Callable[[ProgressEvent], None]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _report(
|
|
73
|
+
reporter: ProgressReporter,
|
|
74
|
+
stage: str,
|
|
75
|
+
input_path: Path | None = None,
|
|
76
|
+
status: Status | None = None,
|
|
77
|
+
) -> None:
|
|
78
|
+
reporter(ProgressEvent(stage, input_path, status))
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _missing_fields(metadata: AudioMetadata) -> list[str]:
|
|
82
|
+
missing: list[str] = []
|
|
83
|
+
if not metadata.title:
|
|
84
|
+
missing.append("title")
|
|
85
|
+
if not metadata.artist:
|
|
86
|
+
missing.append("artist")
|
|
87
|
+
return missing
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass(frozen=True)
|
|
91
|
+
class _PreparedSong:
|
|
92
|
+
index: int
|
|
93
|
+
request: RenderRequest
|
|
94
|
+
background_path: Path
|
|
95
|
+
video_path: Path
|
|
96
|
+
cover_source: str
|
|
97
|
+
lyrics_source: str
|
|
98
|
+
warnings: tuple[str, ...]
|
|
99
|
+
enriched_audio_path: Path
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def process_batch(
|
|
103
|
+
audio_paths: Sequence[Path],
|
|
104
|
+
background_path: Path,
|
|
105
|
+
config: TemplateConfig,
|
|
106
|
+
output_dir: Path,
|
|
107
|
+
font_path: Path,
|
|
108
|
+
ffmpeg: str,
|
|
109
|
+
ffprobe: str,
|
|
110
|
+
overwrite: bool,
|
|
111
|
+
*,
|
|
112
|
+
probe: ProbeFn = probe_audio,
|
|
113
|
+
background_renderer: BackgroundRenderer = render_background,
|
|
114
|
+
node: str = "node",
|
|
115
|
+
cover_resolver: object | None = None,
|
|
116
|
+
lyrics_resolver: Callable[..., object] = resolve_lyrics,
|
|
117
|
+
batch_renderer: Callable[..., object] = run_remotion_batch,
|
|
118
|
+
publisher: Callable[..., None] = publish_remotion_video,
|
|
119
|
+
audio_writer: Callable[..., Path] = write_enriched_mp3,
|
|
120
|
+
workspace_factory: Callable[..., BatchWorkspace] = BatchWorkspace,
|
|
121
|
+
progress: ProgressReporter | None = None,
|
|
122
|
+
) -> BatchResult:
|
|
123
|
+
"""Preprocess songs, render one Remotion batch, then validate and publish."""
|
|
124
|
+
reporter = progress or (lambda _event: None)
|
|
125
|
+
_report(reporter, "batch_start")
|
|
126
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
127
|
+
resolver = cover_resolver or BatchCoverResolver()
|
|
128
|
+
slots: list[SongResult | None] = [None] * len(audio_paths)
|
|
129
|
+
prepared: list[_PreparedSong] = []
|
|
130
|
+
with workspace_factory(output_dir) as workspace:
|
|
131
|
+
renders_dir = workspace.root / "renders"
|
|
132
|
+
renders_dir.mkdir()
|
|
133
|
+
for index, audio_path in enumerate(audio_paths):
|
|
134
|
+
job_id = f"song-{index + 1:04d}"
|
|
135
|
+
try:
|
|
136
|
+
_report(reporter, "metadata", audio_path)
|
|
137
|
+
metadata = probe(audio_path, ffprobe)
|
|
138
|
+
missing = _missing_fields(metadata)
|
|
139
|
+
if missing:
|
|
140
|
+
slots[index] = SongResult(
|
|
141
|
+
audio_path,
|
|
142
|
+
"skipped",
|
|
143
|
+
f"缺少文件头字段:{', '.join(missing)}",
|
|
144
|
+
)
|
|
145
|
+
_report(reporter, "completed", audio_path, "skipped")
|
|
146
|
+
continue
|
|
147
|
+
|
|
148
|
+
enriched_audio_path = enriched_mp3_path(
|
|
149
|
+
audio_path,
|
|
150
|
+
metadata.title,
|
|
151
|
+
metadata.artist,
|
|
152
|
+
)
|
|
153
|
+
video_output = output_dir / f"{enriched_audio_path.stem}.mp4"
|
|
154
|
+
if video_output.exists() and not overwrite:
|
|
155
|
+
slots[index] = SongResult(
|
|
156
|
+
audio_path,
|
|
157
|
+
"skipped",
|
|
158
|
+
f"视频已存在:{video_output}",
|
|
159
|
+
video_path=video_output,
|
|
160
|
+
enriched_audio_path=enriched_audio_path
|
|
161
|
+
)
|
|
162
|
+
_report(reporter, "completed", audio_path, "skipped")
|
|
163
|
+
continue
|
|
164
|
+
|
|
165
|
+
_report(reporter, "cover", audio_path)
|
|
166
|
+
cover = resolver.resolve(
|
|
167
|
+
audio_path,
|
|
168
|
+
metadata,
|
|
169
|
+
workspace.root / "covers" / f"{job_id}.png",
|
|
170
|
+
ffmpeg=ffmpeg,
|
|
171
|
+
)
|
|
172
|
+
_report(reporter, "lyrics", audio_path)
|
|
173
|
+
lyrics = lyrics_resolver(audio_path, metadata)
|
|
174
|
+
if not enriched_audio_path.exists():
|
|
175
|
+
_report(reporter, "enriched_audio", audio_path)
|
|
176
|
+
enriched_audio_path = audio_writer(
|
|
177
|
+
audio_path,
|
|
178
|
+
metadata.title,
|
|
179
|
+
metadata.artist,
|
|
180
|
+
cover.cover_path,
|
|
181
|
+
lyrics.cues,
|
|
182
|
+
ffmpeg=ffmpeg,
|
|
183
|
+
overwrite=False,
|
|
184
|
+
)
|
|
185
|
+
background_output = output_dir / f"{enriched_audio_path.stem}-background.png"
|
|
186
|
+
if not background_output.exists():
|
|
187
|
+
_report(reporter, "background", audio_path)
|
|
188
|
+
background_renderer(
|
|
189
|
+
background_path,
|
|
190
|
+
background_output,
|
|
191
|
+
config,
|
|
192
|
+
metadata.title,
|
|
193
|
+
metadata.artist,
|
|
194
|
+
font_path,
|
|
195
|
+
False,
|
|
196
|
+
)
|
|
197
|
+
_report(reporter, "render_prepare", audio_path)
|
|
198
|
+
job = workspace.stage_job(
|
|
199
|
+
job_id,
|
|
200
|
+
enriched_audio_path,
|
|
201
|
+
background_path,
|
|
202
|
+
cover.cover_path,
|
|
203
|
+
font_path,
|
|
204
|
+
metadata,
|
|
205
|
+
lyrics.cues,
|
|
206
|
+
)
|
|
207
|
+
request = RenderRequest(job, renders_dir / f"{job_id}.mp4")
|
|
208
|
+
prepared.append(
|
|
209
|
+
_PreparedSong(
|
|
210
|
+
index,
|
|
211
|
+
request,
|
|
212
|
+
background_output,
|
|
213
|
+
video_output,
|
|
214
|
+
cover.source,
|
|
215
|
+
lyrics.source,
|
|
216
|
+
tuple(cover.warnings) + tuple(lyrics.warnings),
|
|
217
|
+
enriched_audio_path,
|
|
218
|
+
)
|
|
219
|
+
)
|
|
220
|
+
except Exception as error:
|
|
221
|
+
slots[index] = SongResult(audio_path, "failed", str(error))
|
|
222
|
+
_report(reporter, "completed", audio_path, "failed")
|
|
223
|
+
|
|
224
|
+
if prepared:
|
|
225
|
+
try:
|
|
226
|
+
for item in prepared:
|
|
227
|
+
_report(reporter, "remotion", audio_paths[item.index])
|
|
228
|
+
outcomes = batch_renderer(
|
|
229
|
+
tuple(item.request for item in prepared),
|
|
230
|
+
workspace.public_dir,
|
|
231
|
+
node=node,
|
|
232
|
+
)
|
|
233
|
+
outcome_by_id = {outcome.job_id: outcome for outcome in outcomes}
|
|
234
|
+
except Exception as error:
|
|
235
|
+
outcome_by_id = {}
|
|
236
|
+
process_error = str(error)
|
|
237
|
+
else:
|
|
238
|
+
process_error = None
|
|
239
|
+
|
|
240
|
+
for item in prepared:
|
|
241
|
+
audio_path = audio_paths[item.index]
|
|
242
|
+
outcome = outcome_by_id.get(item.request.job.job_id)
|
|
243
|
+
if process_error is not None:
|
|
244
|
+
slots[item.index] = SongResult(
|
|
245
|
+
audio_path,
|
|
246
|
+
"failed",
|
|
247
|
+
process_error,
|
|
248
|
+
item.background_path,
|
|
249
|
+
cover_source=item.cover_source,
|
|
250
|
+
lyrics_source=item.lyrics_source,
|
|
251
|
+
warnings=item.warnings,
|
|
252
|
+
)
|
|
253
|
+
continue
|
|
254
|
+
if outcome is None or outcome.status == "failed":
|
|
255
|
+
message = (
|
|
256
|
+
"Remotion 未返回该歌曲的渲染结果"
|
|
257
|
+
if outcome is None
|
|
258
|
+
else outcome.error or "Remotion 渲染失败"
|
|
259
|
+
)
|
|
260
|
+
slots[item.index] = SongResult(
|
|
261
|
+
audio_path,
|
|
262
|
+
"failed",
|
|
263
|
+
message,
|
|
264
|
+
item.background_path,
|
|
265
|
+
cover_source=item.cover_source,
|
|
266
|
+
lyrics_source=item.lyrics_source,
|
|
267
|
+
warnings=item.warnings,
|
|
268
|
+
)
|
|
269
|
+
continue
|
|
270
|
+
try:
|
|
271
|
+
_report(reporter, "publish", audio_path)
|
|
272
|
+
publisher(
|
|
273
|
+
outcome.output_path,
|
|
274
|
+
item.video_path,
|
|
275
|
+
ffmpeg,
|
|
276
|
+
ffprobe,
|
|
277
|
+
overwrite,
|
|
278
|
+
)
|
|
279
|
+
item.background_path.unlink(missing_ok=True)
|
|
280
|
+
slots[item.index] = SongResult(
|
|
281
|
+
audio_path,
|
|
282
|
+
"success",
|
|
283
|
+
"生成完成",
|
|
284
|
+
item.background_path,
|
|
285
|
+
item.video_path,
|
|
286
|
+
item.cover_source,
|
|
287
|
+
item.lyrics_source,
|
|
288
|
+
item.warnings,
|
|
289
|
+
item.enriched_audio_path,
|
|
290
|
+
)
|
|
291
|
+
except Exception as error:
|
|
292
|
+
slots[item.index] = SongResult(
|
|
293
|
+
audio_path,
|
|
294
|
+
"failed",
|
|
295
|
+
str(error),
|
|
296
|
+
item.background_path,
|
|
297
|
+
cover_source=item.cover_source,
|
|
298
|
+
lyrics_source=item.lyrics_source,
|
|
299
|
+
warnings=item.warnings,
|
|
300
|
+
)
|
|
301
|
+
_report(reporter, "completed", audio_path, slots[item.index].status)
|
|
302
|
+
|
|
303
|
+
result = BatchResult(tuple(item for item in slots if item is not None))
|
|
304
|
+
_report(reporter, "batch_completed")
|
|
305
|
+
return result
|