lemony-lrc-parser 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lemony_lrc_parser/__init__.py +85 -0
- lemony_lrc_parser/exceptions.py +14 -0
- lemony_lrc_parser/models.py +269 -0
- lemony_lrc_parser/offset.py +53 -0
- lemony_lrc_parser/parser.py +297 -0
- lemony_lrc_parser/py.typed +0 -0
- lemony_lrc_parser/regex.py +159 -0
- lemony_lrc_parser/serializer.py +142 -0
- lemony_lrc_parser/timetag.py +104 -0
- lemony_lrc_parser-0.3.0.dist-info/METADATA +293 -0
- lemony_lrc_parser-0.3.0.dist-info/RECORD +13 -0
- lemony_lrc_parser-0.3.0.dist-info/WHEEL +4 -0
- lemony_lrc_parser-0.3.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""lemony-lrc-parser —— 简洁的 Python LRC 歌词解析器.
|
|
2
|
+
|
|
3
|
+
公共 API 分成三层:
|
|
4
|
+
|
|
5
|
+
1. **面向对象入口** (推荐) : :class:`Lyrics` 及其 :meth:`~Lyrics.loads` /
|
|
6
|
+
:meth:`~Lyrics.dumps` 方法.
|
|
7
|
+
|
|
8
|
+
.. code-block:: python
|
|
9
|
+
|
|
10
|
+
from lemony_lrc_parser import Lyrics
|
|
11
|
+
|
|
12
|
+
lyrics = Lyrics.loads(lrc_text)
|
|
13
|
+
for line in lyrics:
|
|
14
|
+
print(line.start, line.text)
|
|
15
|
+
lrc_out = lyrics.dumps()
|
|
16
|
+
|
|
17
|
+
2. **顶层便捷函数** (等价于 :class:`Lyrics` 的方法) : :func:`loads` /
|
|
18
|
+
:func:`dumps`, 风格对齐 ``json`` / ``pickle``.
|
|
19
|
+
|
|
20
|
+
.. code-block:: python
|
|
21
|
+
|
|
22
|
+
import lemony_lrc_parser as llp
|
|
23
|
+
|
|
24
|
+
lyrics = llp.loads(lrc_text)
|
|
25
|
+
out = llp.dumps(lyrics)
|
|
26
|
+
|
|
27
|
+
3. **底层函数与工具**: :func:`parse_lrc` / :func:`parse_line` / :func:`dump_lrc`
|
|
28
|
+
以及时间标签工具 :func:`format_timetag` / :func:`parse_timetag`.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
from .exceptions import (
|
|
34
|
+
InvalidLyricsError,
|
|
35
|
+
LyricsParserError,
|
|
36
|
+
ProgrammingError,
|
|
37
|
+
TimestampUnderflowError,
|
|
38
|
+
)
|
|
39
|
+
from .models import (
|
|
40
|
+
BasicLyricLine,
|
|
41
|
+
LyricLine,
|
|
42
|
+
Lyrics,
|
|
43
|
+
LyricToken,
|
|
44
|
+
ParseOptions,
|
|
45
|
+
SerializationOptions,
|
|
46
|
+
)
|
|
47
|
+
from .parser import parse_line, parse_lrc
|
|
48
|
+
from .serializer import dump_lrc
|
|
49
|
+
from .timetag import format_timetag, parse_timetag
|
|
50
|
+
|
|
51
|
+
__all__ = [
|
|
52
|
+
"BasicLyricLine",
|
|
53
|
+
"LyricLine",
|
|
54
|
+
"LyricToken",
|
|
55
|
+
"Lyrics",
|
|
56
|
+
"InvalidLyricsError",
|
|
57
|
+
"LyricsParserError",
|
|
58
|
+
"ProgrammingError",
|
|
59
|
+
"TimestampUnderflowError",
|
|
60
|
+
"ParseOptions",
|
|
61
|
+
"SerializationOptions",
|
|
62
|
+
"dumps",
|
|
63
|
+
"loads",
|
|
64
|
+
"dump_lrc",
|
|
65
|
+
"parse_line",
|
|
66
|
+
"parse_lrc",
|
|
67
|
+
"format_timetag",
|
|
68
|
+
"parse_timetag",
|
|
69
|
+
]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def loads(s: str, *, options: ParseOptions | None = None) -> Lyrics:
|
|
73
|
+
"""从 LRC 字符串解析出一份 :class:`Lyrics`.
|
|
74
|
+
|
|
75
|
+
等价于 :meth:`Lyrics.loads`.
|
|
76
|
+
"""
|
|
77
|
+
return Lyrics.loads(s, options=options)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def dumps(lyrics: Lyrics, *, options: SerializationOptions | None = None) -> str:
|
|
81
|
+
"""把 :class:`Lyrics` 序列化为 LRC 字符串.
|
|
82
|
+
|
|
83
|
+
等价于 ``lyrics.dumps(options=options)``
|
|
84
|
+
"""
|
|
85
|
+
return lyrics.dumps(options=options)
|
|
@@ -0,0 +1,269 @@
|
|
|
1
|
+
"""数据模型.
|
|
2
|
+
|
|
3
|
+
定义 LRC 歌词的核心数据结构: :class:`LyricToken`、:class:`LyricLine` 和
|
|
4
|
+
作为顶层容器的 :class:`Lyrics`.
|
|
5
|
+
|
|
6
|
+
本模块只承载数据层语义; 解析 (LRC 文本 → :class:`Lyrics`) 与序列化
|
|
7
|
+
(:class:`Lyrics` → LRC 文本) 的实现分别位于 :mod:`.parser` 与
|
|
8
|
+
:mod:`.serializer`, 这里仅通过延迟导入把它们暴露成 :class:`Lyrics` 的方法.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from collections.abc import Iterator
|
|
14
|
+
from copy import deepcopy
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from typing import overload
|
|
17
|
+
|
|
18
|
+
from .exceptions import ProgrammingError, TimestampUnderflowError
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
"BasicLyricLine",
|
|
22
|
+
"LyricLine",
|
|
23
|
+
"LyricToken",
|
|
24
|
+
"Lyrics",
|
|
25
|
+
"ParseOptions",
|
|
26
|
+
"SerializationOptions",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class ParseOptions:
|
|
32
|
+
"""解析选项.
|
|
33
|
+
|
|
34
|
+
Attributes:
|
|
35
|
+
fill_implicit_line_end: 若为 ``True``, 则当某行没有显式结束时间时,
|
|
36
|
+
自动用下一行的开始时间作为其结束时间.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
fill_implicit_line_end: bool = False
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class SerializationOptions:
|
|
44
|
+
"""序列化选项.
|
|
45
|
+
|
|
46
|
+
Attributes:
|
|
47
|
+
with_metadata: 是否输出 metadata 段.
|
|
48
|
+
use_bracket_for_byword_tag: 逐字标签使用 ``[...]`` 而非 ``<...>``. 在 foobar2000 等老式播放器上可能会有用.
|
|
49
|
+
line_tag_decimal_length: 行标签毫秒位数 (默认 2).
|
|
50
|
+
word_tag_decimal_length: 逐字标签毫秒位数 (默认 2).
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
with_metadata: bool = True
|
|
54
|
+
use_bracket_for_byword_tag: bool = False
|
|
55
|
+
line_tag_decimal_length: int = 2
|
|
56
|
+
word_tag_decimal_length: int = 2
|
|
57
|
+
|
|
58
|
+
def __post_init__(self) -> None:
|
|
59
|
+
"""校验参数合法性."""
|
|
60
|
+
from .timetag import MAX_TAIL_DIGITS, MIN_TAIL_DIGITS
|
|
61
|
+
|
|
62
|
+
for f in ("line_tag_decimal_length", "word_tag_decimal_length"):
|
|
63
|
+
val = getattr(self, f)
|
|
64
|
+
if not MIN_TAIL_DIGITS <= val <= MAX_TAIL_DIGITS:
|
|
65
|
+
raise ProgrammingError(
|
|
66
|
+
f"{f} must be between {MIN_TAIL_DIGITS} and {MAX_TAIL_DIGITS}, got {val}"
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass
|
|
71
|
+
class LyricToken:
|
|
72
|
+
"""一个歌词词元 (可以是一个字、一个词, 或整行纯文本) .
|
|
73
|
+
|
|
74
|
+
Attributes:
|
|
75
|
+
content: 词元的文本内容.
|
|
76
|
+
start: 开始时间 (毫秒) , 未知时为 ``None``.
|
|
77
|
+
end: 结束时间 (毫秒) , 未知时为 ``None``.
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
content: str = ""
|
|
81
|
+
start: int | None = None
|
|
82
|
+
end: int | None = None
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
#: 一行歌词主体 (由若干 :class:`LyricToken` 组成的线性序列) .
|
|
86
|
+
#:
|
|
87
|
+
#: 对于单段整行歌词, 此列表长度通常为 1; 对于逐字歌词, 长度为各词元数量.
|
|
88
|
+
BasicLyricLine = list[LyricToken]
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass
|
|
92
|
+
class LyricLine:
|
|
93
|
+
"""一行歌词.
|
|
94
|
+
|
|
95
|
+
Attributes:
|
|
96
|
+
start: 行开始时间 (毫秒) .
|
|
97
|
+
end: 行结束时间 (毫秒) .
|
|
98
|
+
content: 主语言行内容, 见 :data:`BasicLyricLine`.
|
|
99
|
+
reference_lines: 参考行列表, 常用于存放翻译/音译等辅助行.
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
start: int | None = None
|
|
103
|
+
end: int | None = None
|
|
104
|
+
content: BasicLyricLine = field(default_factory=list)
|
|
105
|
+
reference_lines: list[BasicLyricLine] = field(default_factory=list)
|
|
106
|
+
|
|
107
|
+
@property
|
|
108
|
+
def text(self) -> str:
|
|
109
|
+
"""拼接整行主语言的纯文本 (便于日志与简单展示) ."""
|
|
110
|
+
return "".join(word.content for word in self.content)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@dataclass
|
|
114
|
+
class Lyrics:
|
|
115
|
+
"""一份完整的歌词.
|
|
116
|
+
|
|
117
|
+
Attributes:
|
|
118
|
+
lines: 按时间顺序排列的歌词行.
|
|
119
|
+
metadata: 元数据键值对 (如 ``ti``、``ar``、``offset`` 等) .
|
|
120
|
+
|
|
121
|
+
:class:`Lyrics` 同时是序列容器, 可直接 ``for line in lyrics`` 迭代、
|
|
122
|
+
``len(lyrics)`` 取行数, 或通过下标/切片访问具体行.
|
|
123
|
+
"""
|
|
124
|
+
|
|
125
|
+
lines: list[LyricLine] = field(default_factory=list)
|
|
126
|
+
metadata: dict[str, str] = field(default_factory=dict)
|
|
127
|
+
|
|
128
|
+
def __iter__(self) -> Iterator[LyricLine]:
|
|
129
|
+
return iter(self.lines)
|
|
130
|
+
|
|
131
|
+
def __len__(self) -> int:
|
|
132
|
+
return len(self.lines)
|
|
133
|
+
|
|
134
|
+
@overload
|
|
135
|
+
def __getitem__(self, index: int) -> LyricLine: ...
|
|
136
|
+
|
|
137
|
+
@overload
|
|
138
|
+
def __getitem__(self, index: slice) -> list[LyricLine]: ...
|
|
139
|
+
|
|
140
|
+
def __getitem__(self, index: int | slice) -> LyricLine | list[LyricLine]:
|
|
141
|
+
return self.lines[index]
|
|
142
|
+
|
|
143
|
+
def __add__(self, other: Lyrics) -> Lyrics:
|
|
144
|
+
if not isinstance(other, Lyrics):
|
|
145
|
+
return NotImplemented
|
|
146
|
+
return self.combine(other)
|
|
147
|
+
|
|
148
|
+
def combine(self, other: Lyrics, *, other_as_refline_only: bool = True) -> Lyrics:
|
|
149
|
+
"""将另一份 :class:`Lyrics` 合并进当前对象, 返回新实例.
|
|
150
|
+
|
|
151
|
+
常见用途是把翻译版本合并到主歌词: 翻译的每一行会被挂在 ``self`` 中
|
|
152
|
+
同 ``start`` 行的 :attr:`LyricLine.reference_lines` 列表里.
|
|
153
|
+
|
|
154
|
+
Args:
|
|
155
|
+
other: 要合并进来的另一份歌词.
|
|
156
|
+
other_as_refline_only: 若为 ``True`` (默认) , ``other`` 中在
|
|
157
|
+
``self`` 里找不到对应时间点的行会被丢弃; 若为 ``False``,
|
|
158
|
+
这些行会被保留为新行.
|
|
159
|
+
|
|
160
|
+
Returns:
|
|
161
|
+
合并后的新 :class:`Lyrics` 对象; ``self`` 与 ``other`` 均不受影响.
|
|
162
|
+
"""
|
|
163
|
+
new = Lyrics()
|
|
164
|
+
# metadata 以 self 为准, other 作为补充
|
|
165
|
+
new.metadata.update(other.metadata)
|
|
166
|
+
new.metadata.update(self.metadata)
|
|
167
|
+
|
|
168
|
+
pool: dict[int, LyricLine] = {}
|
|
169
|
+
for line in self.lines:
|
|
170
|
+
if line.start is None:
|
|
171
|
+
continue
|
|
172
|
+
# 深拷贝 self 的行, 避免污染原始对象
|
|
173
|
+
pool[line.start] = deepcopy(line)
|
|
174
|
+
for line in other.lines:
|
|
175
|
+
if line.start is None:
|
|
176
|
+
continue
|
|
177
|
+
if line.start in pool:
|
|
178
|
+
# 深拷贝 other 的行内容, 避免共享引用
|
|
179
|
+
pool[line.start].reference_lines.append(deepcopy(line.content))
|
|
180
|
+
pool[line.start].reference_lines.extend(
|
|
181
|
+
deepcopy(rl) for rl in line.reference_lines
|
|
182
|
+
)
|
|
183
|
+
elif not other_as_refline_only:
|
|
184
|
+
pool[line.start] = deepcopy(line)
|
|
185
|
+
|
|
186
|
+
new.lines = sorted(pool.values(), key=lambda line: line.start or 0)
|
|
187
|
+
return new
|
|
188
|
+
|
|
189
|
+
@classmethod
|
|
190
|
+
def loads(cls, s: str, *, options: ParseOptions | None = None) -> Lyrics:
|
|
191
|
+
"""从 LRC 字符串解析出一份 :class:`Lyrics`.
|
|
192
|
+
|
|
193
|
+
Args:
|
|
194
|
+
s: LRC 源文本.
|
|
195
|
+
options: 解析选项.
|
|
196
|
+
"""
|
|
197
|
+
from .parser import parse_lrc
|
|
198
|
+
|
|
199
|
+
return parse_lrc(s, options=options)
|
|
200
|
+
|
|
201
|
+
def dumps(self, *, options: SerializationOptions | None = None) -> str:
|
|
202
|
+
"""把当前对象序列化为 LRC 字符串.
|
|
203
|
+
|
|
204
|
+
Args:
|
|
205
|
+
options: 序列化选项.
|
|
206
|
+
"""
|
|
207
|
+
from .serializer import dump_lrc
|
|
208
|
+
|
|
209
|
+
return dump_lrc(self, options=options)
|
|
210
|
+
|
|
211
|
+
def apply_delta(self, ms: int) -> Lyrics:
|
|
212
|
+
"""深拷贝当前对象并在新副本上应用时间偏移, 返回新对象.
|
|
213
|
+
|
|
214
|
+
该方法**不修改**原始对象, 而是返回一个时间戳已被整体偏移的新
|
|
215
|
+
:class:`Lyrics`.
|
|
216
|
+
|
|
217
|
+
这里的 ms 直接加在时间戳上, 所以用了 delta 这个词而不是 offset.
|
|
218
|
+
|
|
219
|
+
若偏移后存在任何时间戳 < 0, 直接抛出 :class:`TimestampUnderflowError`,
|
|
220
|
+
由调用方自行处理 (例如先从 ``metadata.offset`` 读取值再传入).
|
|
221
|
+
|
|
222
|
+
Args:
|
|
223
|
+
ms: 要加到每个时间戳上的毫秒数.
|
|
224
|
+
|
|
225
|
+
Returns:
|
|
226
|
+
应用了偏移后的新 :class:`Lyrics` 对象.
|
|
227
|
+
|
|
228
|
+
Raises:
|
|
229
|
+
TimestampUnderflowError: 偏移后会导致某个时间戳变为负数.
|
|
230
|
+
"""
|
|
231
|
+
from .offset import apply_delta, iter_all_timestamps
|
|
232
|
+
|
|
233
|
+
if ms == 0:
|
|
234
|
+
return deepcopy(self)
|
|
235
|
+
|
|
236
|
+
# 先做下溢检测, 避免异常路径上的 deepcopy 浪费
|
|
237
|
+
all_times = list(iter_all_timestamps(self))
|
|
238
|
+
if all_times:
|
|
239
|
+
min_time = min(all_times)
|
|
240
|
+
if min_time + ms < 0:
|
|
241
|
+
raise TimestampUnderflowError(
|
|
242
|
+
f"Applying offset={ms}ms would make minimum "
|
|
243
|
+
f"timestamp {min_time}ms negative"
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
result = deepcopy(self)
|
|
247
|
+
apply_delta(result, ms)
|
|
248
|
+
return result
|
|
249
|
+
|
|
250
|
+
def __lshift__(self, ms: int) -> Lyrics:
|
|
251
|
+
"""左移运算符: ``lyrics << ms`` 等价于 ``lyrics.apply_delta(-ms)``.
|
|
252
|
+
|
|
253
|
+
语义: 正数 → 歌词提前出现.
|
|
254
|
+
"""
|
|
255
|
+
if not isinstance(ms, int):
|
|
256
|
+
return NotImplemented
|
|
257
|
+
return self.apply_delta(-ms)
|
|
258
|
+
|
|
259
|
+
def __rshift__(self, ms: int) -> Lyrics:
|
|
260
|
+
"""右移运算符: ``lyrics >> ms`` 等价于 ``lyrics.apply_delta(ms)``.
|
|
261
|
+
|
|
262
|
+
语义: 正数 → 歌词延后出现.
|
|
263
|
+
"""
|
|
264
|
+
if not isinstance(ms, int):
|
|
265
|
+
return NotImplemented
|
|
266
|
+
return self.apply_delta(ms)
|
|
267
|
+
|
|
268
|
+
def __str__(self) -> str:
|
|
269
|
+
return self.dumps()
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Iterator
|
|
4
|
+
|
|
5
|
+
from .models import BasicLyricLine, Lyrics
|
|
6
|
+
|
|
7
|
+
__all__ = [
|
|
8
|
+
"apply_delta",
|
|
9
|
+
"iter_all_timestamps",
|
|
10
|
+
]
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def apply_delta(lyrics: Lyrics, delta: int) -> None:
|
|
14
|
+
"""将所有时间戳增加 ``delta`` ms
|
|
15
|
+
|
|
16
|
+
注意此处为底层函数没有下溢保护, 且为原地修改"""
|
|
17
|
+
for line in lyrics.lines:
|
|
18
|
+
if line.start is not None:
|
|
19
|
+
line.start += delta
|
|
20
|
+
if line.end is not None:
|
|
21
|
+
line.end += delta
|
|
22
|
+
_apply_word_delta(line.content, delta)
|
|
23
|
+
for refline in line.reference_lines:
|
|
24
|
+
_apply_word_delta(refline, delta)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _apply_word_delta(words: BasicLyricLine, delta: int) -> None:
|
|
28
|
+
for word in words:
|
|
29
|
+
if word.start is not None:
|
|
30
|
+
word.start += delta
|
|
31
|
+
if word.end is not None:
|
|
32
|
+
word.end += delta
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def iter_all_timestamps(lyrics: Lyrics) -> Iterator[int]:
|
|
36
|
+
"""迭代 :class:`Lyrics` 中出现过的所有时间戳 (含参考行)."""
|
|
37
|
+
for line in lyrics.lines:
|
|
38
|
+
if line.start is not None:
|
|
39
|
+
yield line.start
|
|
40
|
+
if line.end is not None:
|
|
41
|
+
yield line.end
|
|
42
|
+
yield from _iter_word_timestamps(line.content)
|
|
43
|
+
for refline in line.reference_lines:
|
|
44
|
+
yield from _iter_word_timestamps(refline)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _iter_word_timestamps(words: BasicLyricLine) -> Iterator[int]:
|
|
48
|
+
"""从一个 word 序列中迭代出所有非空时间戳."""
|
|
49
|
+
for word in words:
|
|
50
|
+
if word.start is not None:
|
|
51
|
+
yield word.start
|
|
52
|
+
if word.end is not None:
|
|
53
|
+
yield word.end
|