lemony-lrc-parser 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,85 @@
1
+ """lemony-lrc-parser —— 简洁的 Python LRC 歌词解析器.
2
+
3
+ 公共 API 分成三层:
4
+
5
+ 1. **面向对象入口** (推荐) : :class:`Lyrics` 及其 :meth:`~Lyrics.loads` /
6
+ :meth:`~Lyrics.dumps` 方法.
7
+
8
+ .. code-block:: python
9
+
10
+ from lemony_lrc_parser import Lyrics
11
+
12
+ lyrics = Lyrics.loads(lrc_text)
13
+ for line in lyrics:
14
+ print(line.start, line.text)
15
+ lrc_out = lyrics.dumps()
16
+
17
+ 2. **顶层便捷函数** (等价于 :class:`Lyrics` 的方法) : :func:`loads` /
18
+ :func:`dumps`, 风格对齐 ``json`` / ``pickle``.
19
+
20
+ .. code-block:: python
21
+
22
+ import lemony_lrc_parser as llp
23
+
24
+ lyrics = llp.loads(lrc_text)
25
+ out = llp.dumps(lyrics)
26
+
27
+ 3. **底层函数与工具**: :func:`parse_lrc` / :func:`parse_line` / :func:`dump_lrc`
28
+ 以及时间标签工具 :func:`format_timetag` / :func:`parse_timetag`.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ from .exceptions import (
34
+ InvalidLyricsError,
35
+ LyricsParserError,
36
+ ProgrammingError,
37
+ TimestampUnderflowError,
38
+ )
39
+ from .models import (
40
+ BasicLyricLine,
41
+ LyricLine,
42
+ Lyrics,
43
+ LyricToken,
44
+ ParseOptions,
45
+ SerializationOptions,
46
+ )
47
+ from .parser import parse_line, parse_lrc
48
+ from .serializer import dump_lrc
49
+ from .timetag import format_timetag, parse_timetag
50
+
51
+ __all__ = [
52
+ "BasicLyricLine",
53
+ "LyricLine",
54
+ "LyricToken",
55
+ "Lyrics",
56
+ "InvalidLyricsError",
57
+ "LyricsParserError",
58
+ "ProgrammingError",
59
+ "TimestampUnderflowError",
60
+ "ParseOptions",
61
+ "SerializationOptions",
62
+ "dumps",
63
+ "loads",
64
+ "dump_lrc",
65
+ "parse_line",
66
+ "parse_lrc",
67
+ "format_timetag",
68
+ "parse_timetag",
69
+ ]
70
+
71
+
72
+ def loads(s: str, *, options: ParseOptions | None = None) -> Lyrics:
73
+ """从 LRC 字符串解析出一份 :class:`Lyrics`.
74
+
75
+ 等价于 :meth:`Lyrics.loads`.
76
+ """
77
+ return Lyrics.loads(s, options=options)
78
+
79
+
80
+ def dumps(lyrics: Lyrics, *, options: SerializationOptions | None = None) -> str:
81
+ """把 :class:`Lyrics` 序列化为 LRC 字符串.
82
+
83
+ 等价于 ``lyrics.dumps(options=options)``
84
+ """
85
+ return lyrics.dumps(options=options)
@@ -0,0 +1,14 @@
1
+ class LyricsParserError(Exception):
2
+ pass
3
+
4
+
5
+ class InvalidLyricsError(LyricsParserError):
6
+ pass
7
+
8
+
9
+ class TimestampUnderflowError(LyricsParserError):
10
+ pass
11
+
12
+
13
+ class ProgrammingError(LyricsParserError):
14
+ pass
@@ -0,0 +1,269 @@
1
+ """数据模型.
2
+
3
+ 定义 LRC 歌词的核心数据结构: :class:`LyricToken`、:class:`LyricLine` 和
4
+ 作为顶层容器的 :class:`Lyrics`.
5
+
6
+ 本模块只承载数据层语义; 解析 (LRC 文本 → :class:`Lyrics`) 与序列化
7
+ (:class:`Lyrics` → LRC 文本) 的实现分别位于 :mod:`.parser` 与
8
+ :mod:`.serializer`, 这里仅通过延迟导入把它们暴露成 :class:`Lyrics` 的方法.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ from collections.abc import Iterator
14
+ from copy import deepcopy
15
+ from dataclasses import dataclass, field
16
+ from typing import overload
17
+
18
+ from .exceptions import ProgrammingError, TimestampUnderflowError
19
+
20
+ __all__ = [
21
+ "BasicLyricLine",
22
+ "LyricLine",
23
+ "LyricToken",
24
+ "Lyrics",
25
+ "ParseOptions",
26
+ "SerializationOptions",
27
+ ]
28
+
29
+
30
+ @dataclass
31
+ class ParseOptions:
32
+ """解析选项.
33
+
34
+ Attributes:
35
+ fill_implicit_line_end: 若为 ``True``, 则当某行没有显式结束时间时,
36
+ 自动用下一行的开始时间作为其结束时间.
37
+ """
38
+
39
+ fill_implicit_line_end: bool = False
40
+
41
+
42
+ @dataclass
43
+ class SerializationOptions:
44
+ """序列化选项.
45
+
46
+ Attributes:
47
+ with_metadata: 是否输出 metadata 段.
48
+ use_bracket_for_byword_tag: 逐字标签使用 ``[...]`` 而非 ``<...>``. 在 foobar2000 等老式播放器上可能会有用.
49
+ line_tag_decimal_length: 行标签毫秒位数 (默认 2).
50
+ word_tag_decimal_length: 逐字标签毫秒位数 (默认 2).
51
+ """
52
+
53
+ with_metadata: bool = True
54
+ use_bracket_for_byword_tag: bool = False
55
+ line_tag_decimal_length: int = 2
56
+ word_tag_decimal_length: int = 2
57
+
58
+ def __post_init__(self) -> None:
59
+ """校验参数合法性."""
60
+ from .timetag import MAX_TAIL_DIGITS, MIN_TAIL_DIGITS
61
+
62
+ for f in ("line_tag_decimal_length", "word_tag_decimal_length"):
63
+ val = getattr(self, f)
64
+ if not MIN_TAIL_DIGITS <= val <= MAX_TAIL_DIGITS:
65
+ raise ProgrammingError(
66
+ f"{f} must be between {MIN_TAIL_DIGITS} and {MAX_TAIL_DIGITS}, got {val}"
67
+ )
68
+
69
+
70
+ @dataclass
71
+ class LyricToken:
72
+ """一个歌词词元 (可以是一个字、一个词, 或整行纯文本) .
73
+
74
+ Attributes:
75
+ content: 词元的文本内容.
76
+ start: 开始时间 (毫秒) , 未知时为 ``None``.
77
+ end: 结束时间 (毫秒) , 未知时为 ``None``.
78
+ """
79
+
80
+ content: str = ""
81
+ start: int | None = None
82
+ end: int | None = None
83
+
84
+
85
+ #: 一行歌词主体 (由若干 :class:`LyricToken` 组成的线性序列) .
86
+ #:
87
+ #: 对于单段整行歌词, 此列表长度通常为 1; 对于逐字歌词, 长度为各词元数量.
88
+ BasicLyricLine = list[LyricToken]
89
+
90
+
91
+ @dataclass
92
+ class LyricLine:
93
+ """一行歌词.
94
+
95
+ Attributes:
96
+ start: 行开始时间 (毫秒) .
97
+ end: 行结束时间 (毫秒) .
98
+ content: 主语言行内容, 见 :data:`BasicLyricLine`.
99
+ reference_lines: 参考行列表, 常用于存放翻译/音译等辅助行.
100
+ """
101
+
102
+ start: int | None = None
103
+ end: int | None = None
104
+ content: BasicLyricLine = field(default_factory=list)
105
+ reference_lines: list[BasicLyricLine] = field(default_factory=list)
106
+
107
+ @property
108
+ def text(self) -> str:
109
+ """拼接整行主语言的纯文本 (便于日志与简单展示) ."""
110
+ return "".join(word.content for word in self.content)
111
+
112
+
113
+ @dataclass
114
+ class Lyrics:
115
+ """一份完整的歌词.
116
+
117
+ Attributes:
118
+ lines: 按时间顺序排列的歌词行.
119
+ metadata: 元数据键值对 (如 ``ti``、``ar``、``offset`` 等) .
120
+
121
+ :class:`Lyrics` 同时是序列容器, 可直接 ``for line in lyrics`` 迭代、
122
+ ``len(lyrics)`` 取行数, 或通过下标/切片访问具体行.
123
+ """
124
+
125
+ lines: list[LyricLine] = field(default_factory=list)
126
+ metadata: dict[str, str] = field(default_factory=dict)
127
+
128
+ def __iter__(self) -> Iterator[LyricLine]:
129
+ return iter(self.lines)
130
+
131
+ def __len__(self) -> int:
132
+ return len(self.lines)
133
+
134
+ @overload
135
+ def __getitem__(self, index: int) -> LyricLine: ...
136
+
137
+ @overload
138
+ def __getitem__(self, index: slice) -> list[LyricLine]: ...
139
+
140
+ def __getitem__(self, index: int | slice) -> LyricLine | list[LyricLine]:
141
+ return self.lines[index]
142
+
143
+ def __add__(self, other: Lyrics) -> Lyrics:
144
+ if not isinstance(other, Lyrics):
145
+ return NotImplemented
146
+ return self.combine(other)
147
+
148
+ def combine(self, other: Lyrics, *, other_as_refline_only: bool = True) -> Lyrics:
149
+ """将另一份 :class:`Lyrics` 合并进当前对象, 返回新实例.
150
+
151
+ 常见用途是把翻译版本合并到主歌词: 翻译的每一行会被挂在 ``self`` 中
152
+ 同 ``start`` 行的 :attr:`LyricLine.reference_lines` 列表里.
153
+
154
+ Args:
155
+ other: 要合并进来的另一份歌词.
156
+ other_as_refline_only: 若为 ``True`` (默认) , ``other`` 中在
157
+ ``self`` 里找不到对应时间点的行会被丢弃; 若为 ``False``,
158
+ 这些行会被保留为新行.
159
+
160
+ Returns:
161
+ 合并后的新 :class:`Lyrics` 对象; ``self`` 与 ``other`` 均不受影响.
162
+ """
163
+ new = Lyrics()
164
+ # metadata 以 self 为准, other 作为补充
165
+ new.metadata.update(other.metadata)
166
+ new.metadata.update(self.metadata)
167
+
168
+ pool: dict[int, LyricLine] = {}
169
+ for line in self.lines:
170
+ if line.start is None:
171
+ continue
172
+ # 深拷贝 self 的行, 避免污染原始对象
173
+ pool[line.start] = deepcopy(line)
174
+ for line in other.lines:
175
+ if line.start is None:
176
+ continue
177
+ if line.start in pool:
178
+ # 深拷贝 other 的行内容, 避免共享引用
179
+ pool[line.start].reference_lines.append(deepcopy(line.content))
180
+ pool[line.start].reference_lines.extend(
181
+ deepcopy(rl) for rl in line.reference_lines
182
+ )
183
+ elif not other_as_refline_only:
184
+ pool[line.start] = deepcopy(line)
185
+
186
+ new.lines = sorted(pool.values(), key=lambda line: line.start or 0)
187
+ return new
188
+
189
+ @classmethod
190
+ def loads(cls, s: str, *, options: ParseOptions | None = None) -> Lyrics:
191
+ """从 LRC 字符串解析出一份 :class:`Lyrics`.
192
+
193
+ Args:
194
+ s: LRC 源文本.
195
+ options: 解析选项.
196
+ """
197
+ from .parser import parse_lrc
198
+
199
+ return parse_lrc(s, options=options)
200
+
201
+ def dumps(self, *, options: SerializationOptions | None = None) -> str:
202
+ """把当前对象序列化为 LRC 字符串.
203
+
204
+ Args:
205
+ options: 序列化选项.
206
+ """
207
+ from .serializer import dump_lrc
208
+
209
+ return dump_lrc(self, options=options)
210
+
211
+ def apply_delta(self, ms: int) -> Lyrics:
212
+ """深拷贝当前对象并在新副本上应用时间偏移, 返回新对象.
213
+
214
+ 该方法**不修改**原始对象, 而是返回一个时间戳已被整体偏移的新
215
+ :class:`Lyrics`.
216
+
217
+ 这里的 ms 直接加在时间戳上, 所以用了 delta 这个词而不是 offset.
218
+
219
+ 若偏移后存在任何时间戳 < 0, 直接抛出 :class:`TimestampUnderflowError`,
220
+ 由调用方自行处理 (例如先从 ``metadata.offset`` 读取值再传入).
221
+
222
+ Args:
223
+ ms: 要加到每个时间戳上的毫秒数.
224
+
225
+ Returns:
226
+ 应用了偏移后的新 :class:`Lyrics` 对象.
227
+
228
+ Raises:
229
+ TimestampUnderflowError: 偏移后会导致某个时间戳变为负数.
230
+ """
231
+ from .offset import apply_delta, iter_all_timestamps
232
+
233
+ if ms == 0:
234
+ return deepcopy(self)
235
+
236
+ # 先做下溢检测, 避免异常路径上的 deepcopy 浪费
237
+ all_times = list(iter_all_timestamps(self))
238
+ if all_times:
239
+ min_time = min(all_times)
240
+ if min_time + ms < 0:
241
+ raise TimestampUnderflowError(
242
+ f"Applying offset={ms}ms would make minimum "
243
+ f"timestamp {min_time}ms negative"
244
+ )
245
+
246
+ result = deepcopy(self)
247
+ apply_delta(result, ms)
248
+ return result
249
+
250
+ def __lshift__(self, ms: int) -> Lyrics:
251
+ """左移运算符: ``lyrics << ms`` 等价于 ``lyrics.apply_delta(-ms)``.
252
+
253
+ 语义: 正数 → 歌词提前出现.
254
+ """
255
+ if not isinstance(ms, int):
256
+ return NotImplemented
257
+ return self.apply_delta(-ms)
258
+
259
+ def __rshift__(self, ms: int) -> Lyrics:
260
+ """右移运算符: ``lyrics >> ms`` 等价于 ``lyrics.apply_delta(ms)``.
261
+
262
+ 语义: 正数 → 歌词延后出现.
263
+ """
264
+ if not isinstance(ms, int):
265
+ return NotImplemented
266
+ return self.apply_delta(ms)
267
+
268
+ def __str__(self) -> str:
269
+ return self.dumps()
@@ -0,0 +1,53 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Iterator
4
+
5
+ from .models import BasicLyricLine, Lyrics
6
+
7
+ __all__ = [
8
+ "apply_delta",
9
+ "iter_all_timestamps",
10
+ ]
11
+
12
+
13
+ def apply_delta(lyrics: Lyrics, delta: int) -> None:
14
+ """将所有时间戳增加 ``delta`` ms
15
+
16
+ 注意此处为底层函数没有下溢保护, 且为原地修改"""
17
+ for line in lyrics.lines:
18
+ if line.start is not None:
19
+ line.start += delta
20
+ if line.end is not None:
21
+ line.end += delta
22
+ _apply_word_delta(line.content, delta)
23
+ for refline in line.reference_lines:
24
+ _apply_word_delta(refline, delta)
25
+
26
+
27
+ def _apply_word_delta(words: BasicLyricLine, delta: int) -> None:
28
+ for word in words:
29
+ if word.start is not None:
30
+ word.start += delta
31
+ if word.end is not None:
32
+ word.end += delta
33
+
34
+
35
+ def iter_all_timestamps(lyrics: Lyrics) -> Iterator[int]:
36
+ """迭代 :class:`Lyrics` 中出现过的所有时间戳 (含参考行)."""
37
+ for line in lyrics.lines:
38
+ if line.start is not None:
39
+ yield line.start
40
+ if line.end is not None:
41
+ yield line.end
42
+ yield from _iter_word_timestamps(line.content)
43
+ for refline in line.reference_lines:
44
+ yield from _iter_word_timestamps(refline)
45
+
46
+
47
+ def _iter_word_timestamps(words: BasicLyricLine) -> Iterator[int]:
48
+ """从一个 word 序列中迭代出所有非空时间戳."""
49
+ for word in words:
50
+ if word.start is not None:
51
+ yield word.start
52
+ if word.end is not None:
53
+ yield word.end