lemony-lrc-parser 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lemony_lrc_parser/__init__.py +79 -0
- lemony_lrc_parser/exceptions.py +6 -0
- lemony_lrc_parser/models.py +184 -0
- lemony_lrc_parser/parser.py +291 -0
- lemony_lrc_parser/py.typed +0 -0
- lemony_lrc_parser/regex.py +159 -0
- lemony_lrc_parser/serializer.py +215 -0
- lemony_lrc_parser/timetag.py +86 -0
- lemony_lrc_parser-0.2.1.dist-info/METADATA +229 -0
- lemony_lrc_parser-0.2.1.dist-info/RECORD +12 -0
- lemony_lrc_parser-0.2.1.dist-info/WHEEL +4 -0
- lemony_lrc_parser-0.2.1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""lemony-lrc-parser —— 简洁的 Python LRC 歌词解析器.
|
|
2
|
+
|
|
3
|
+
公共 API 分成三层:
|
|
4
|
+
|
|
5
|
+
1. **面向对象入口** (推荐) : :class:`Lyrics` 及其 :meth:`~Lyrics.loads` /
|
|
6
|
+
:meth:`~Lyrics.dumps` 方法.
|
|
7
|
+
|
|
8
|
+
.. code-block:: python
|
|
9
|
+
|
|
10
|
+
from lemony_lrc_parser import Lyrics
|
|
11
|
+
|
|
12
|
+
lyrics = Lyrics.loads(lrc_text)
|
|
13
|
+
for line in lyrics:
|
|
14
|
+
print(line.start, line.text)
|
|
15
|
+
lrc_out = lyrics.dumps()
|
|
16
|
+
|
|
17
|
+
2. **顶层便捷函数** (等价于 :class:`Lyrics` 的方法) : :func:`loads` /
|
|
18
|
+
:func:`dumps`, 风格对齐 ``json`` / ``pickle``.
|
|
19
|
+
|
|
20
|
+
.. code-block:: python
|
|
21
|
+
|
|
22
|
+
import lemony_lrc_parser as llp
|
|
23
|
+
|
|
24
|
+
lyrics = llp.loads(lrc_text)
|
|
25
|
+
out = llp.dumps(lyrics)
|
|
26
|
+
|
|
27
|
+
3. **底层函数与工具**: :func:`parse_lrc` / :func:`parse_line` / :func:`dump_lrc`
|
|
28
|
+
以及时间标签工具 :func:`format_timetag` / :func:`parse_timetag`.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
from .exceptions import InvalidLyricsError, LyricsParserError
|
|
34
|
+
from .models import BasicLyricLine, LyricLine, Lyrics, LyricWord
|
|
35
|
+
from .parser import parse_line, parse_lrc
|
|
36
|
+
from .serializer import dump_lrc
|
|
37
|
+
from .timetag import format_timetag, parse_timetag
|
|
38
|
+
|
|
39
|
+
__all__ = [
|
|
40
|
+
"BasicLyricLine",
|
|
41
|
+
"LyricLine",
|
|
42
|
+
"LyricWord",
|
|
43
|
+
"Lyrics",
|
|
44
|
+
"InvalidLyricsError",
|
|
45
|
+
"LyricsParserError",
|
|
46
|
+
"dumps",
|
|
47
|
+
"loads",
|
|
48
|
+
"dump_lrc",
|
|
49
|
+
"parse_line",
|
|
50
|
+
"parse_lrc",
|
|
51
|
+
"format_timetag",
|
|
52
|
+
"parse_timetag",
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def loads(s: str, *, fill_implicit_line_end: bool = False) -> Lyrics:
|
|
57
|
+
"""从 LRC 字符串解析出一份 :class:`Lyrics`.
|
|
58
|
+
|
|
59
|
+
等价于 :meth:`Lyrics.loads`.
|
|
60
|
+
"""
|
|
61
|
+
return Lyrics.loads(s, fill_implicit_line_end=fill_implicit_line_end)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def dumps(
|
|
65
|
+
lyrics: Lyrics,
|
|
66
|
+
*,
|
|
67
|
+
with_metadata: bool = True,
|
|
68
|
+
use_bracket_for_byword_tag: bool = False,
|
|
69
|
+
apply_offset_from_metadata: bool = False,
|
|
70
|
+
) -> str:
|
|
71
|
+
"""把 :class:`Lyrics` 序列化为 LRC 字符串.
|
|
72
|
+
|
|
73
|
+
等价于 ``lyrics.dumps(**kwargs)``
|
|
74
|
+
"""
|
|
75
|
+
return lyrics.dumps(
|
|
76
|
+
with_metadata=with_metadata,
|
|
77
|
+
use_bracket_for_byword_tag=use_bracket_for_byword_tag,
|
|
78
|
+
apply_offset_from_metadata=apply_offset_from_metadata,
|
|
79
|
+
)
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
"""数据模型.
|
|
2
|
+
|
|
3
|
+
定义 LRC 歌词的核心数据结构: :class:`LyricWord`、:class:`LyricLine` 和
|
|
4
|
+
作为顶层容器的 :class:`Lyrics`.
|
|
5
|
+
|
|
6
|
+
本模块只承载数据层语义; 解析 (LRC 文本 → :class:`Lyrics`) 与序列化
|
|
7
|
+
(:class:`Lyrics` → LRC 文本) 的实现分别位于 :mod:`.parser` 与
|
|
8
|
+
:mod:`.serializer`, 这里仅通过延迟导入把它们暴露成 :class:`Lyrics` 的方法.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from collections.abc import Iterator
|
|
14
|
+
from copy import deepcopy
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from typing import overload
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"BasicLyricLine",
|
|
20
|
+
"LyricLine",
|
|
21
|
+
"LyricWord",
|
|
22
|
+
"Lyrics",
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class LyricWord:
|
|
28
|
+
"""一个歌词词元 (可以是一个字、一个词, 或整行纯文本) .
|
|
29
|
+
|
|
30
|
+
Attributes:
|
|
31
|
+
content: 词元的文本内容.
|
|
32
|
+
start: 开始时间 (毫秒) , 未知时为 ``None``.
|
|
33
|
+
end: 结束时间 (毫秒) , 未知时为 ``None``.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
content: str = ""
|
|
37
|
+
start: int | None = None
|
|
38
|
+
end: int | None = None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
#: 一行歌词主体 (由若干 :class:`LyricWord` 组成的线性序列) .
|
|
42
|
+
#:
|
|
43
|
+
#: 对于单段整行歌词, 此列表长度通常为 1; 对于逐字歌词, 长度为各词元数量.
|
|
44
|
+
BasicLyricLine = list[LyricWord]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass
|
|
48
|
+
class LyricLine:
|
|
49
|
+
"""一行歌词.
|
|
50
|
+
|
|
51
|
+
Attributes:
|
|
52
|
+
start: 行开始时间 (毫秒) .
|
|
53
|
+
end: 行结束时间 (毫秒) .
|
|
54
|
+
content: 主语言行内容, 见 :data:`BasicLyricLine`.
|
|
55
|
+
reference_lines: 参考行列表, 常用于存放翻译/音译等辅助行.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
start: int | None = None
|
|
59
|
+
end: int | None = None
|
|
60
|
+
content: BasicLyricLine = field(default_factory=list)
|
|
61
|
+
reference_lines: list[BasicLyricLine] = field(default_factory=list)
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def text(self) -> str:
|
|
65
|
+
"""拼接整行主语言的纯文本 (便于日志与简单展示) ."""
|
|
66
|
+
return "".join(word.content for word in self.content)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
@dataclass
|
|
70
|
+
class Lyrics:
|
|
71
|
+
"""一份完整的歌词.
|
|
72
|
+
|
|
73
|
+
Attributes:
|
|
74
|
+
lines: 按时间顺序排列的歌词行.
|
|
75
|
+
metadata: 元数据键值对 (如 ``ti``、``ar``、``offset`` 等) .
|
|
76
|
+
|
|
77
|
+
:class:`Lyrics` 同时是序列容器, 可直接 ``for line in lyrics`` 迭代、
|
|
78
|
+
``len(lyrics)`` 取行数, 或通过下标/切片访问具体行.
|
|
79
|
+
"""
|
|
80
|
+
|
|
81
|
+
lines: list[LyricLine] = field(default_factory=list)
|
|
82
|
+
metadata: dict[str, str] = field(default_factory=dict)
|
|
83
|
+
|
|
84
|
+
def __iter__(self) -> Iterator[LyricLine]:
|
|
85
|
+
return iter(self.lines)
|
|
86
|
+
|
|
87
|
+
def __len__(self) -> int:
|
|
88
|
+
return len(self.lines)
|
|
89
|
+
|
|
90
|
+
@overload
|
|
91
|
+
def __getitem__(self, index: int) -> LyricLine: ...
|
|
92
|
+
|
|
93
|
+
@overload
|
|
94
|
+
def __getitem__(self, index: slice) -> list[LyricLine]: ...
|
|
95
|
+
|
|
96
|
+
def __getitem__(self, index: int | slice) -> LyricLine | list[LyricLine]:
|
|
97
|
+
return self.lines[index]
|
|
98
|
+
|
|
99
|
+
def __add__(self, other: Lyrics) -> Lyrics:
|
|
100
|
+
if not isinstance(other, Lyrics):
|
|
101
|
+
return NotImplemented
|
|
102
|
+
return self.combine(other)
|
|
103
|
+
|
|
104
|
+
def combine(self, other: Lyrics, *, other_as_refline_only: bool = True) -> Lyrics:
|
|
105
|
+
"""将另一份 :class:`Lyrics` 合并进当前对象, 返回新实例.
|
|
106
|
+
|
|
107
|
+
常见用途是把翻译版本合并到主歌词: 翻译的每一行会被挂在 ``self`` 中
|
|
108
|
+
同 ``start`` 行的 :attr:`LyricLine.reference_lines` 列表里.
|
|
109
|
+
|
|
110
|
+
Args:
|
|
111
|
+
other: 要合并进来的另一份歌词.
|
|
112
|
+
other_as_refline_only: 若为 ``True`` (默认) , ``other`` 中在
|
|
113
|
+
``self`` 里找不到对应时间点的行会被丢弃; 若为 ``False``,
|
|
114
|
+
这些行会被保留为新行.
|
|
115
|
+
|
|
116
|
+
Returns:
|
|
117
|
+
合并后的新 :class:`Lyrics` 对象; ``self`` 与 ``other`` 均不受影响.
|
|
118
|
+
"""
|
|
119
|
+
new = Lyrics()
|
|
120
|
+
# metadata 以 self 为准, other 作为补充
|
|
121
|
+
new.metadata.update(other.metadata)
|
|
122
|
+
new.metadata.update(self.metadata)
|
|
123
|
+
|
|
124
|
+
pool: dict[int, LyricLine] = {}
|
|
125
|
+
for line in self.lines:
|
|
126
|
+
if line.start is None:
|
|
127
|
+
continue
|
|
128
|
+
# 深拷贝 self 的行, 避免污染原始对象
|
|
129
|
+
pool[line.start] = deepcopy(line)
|
|
130
|
+
for line in other.lines:
|
|
131
|
+
if line.start is None:
|
|
132
|
+
continue
|
|
133
|
+
if line.start in pool:
|
|
134
|
+
# 深拷贝 other 的行内容, 避免共享引用
|
|
135
|
+
pool[line.start].reference_lines.append(deepcopy(line.content))
|
|
136
|
+
pool[line.start].reference_lines.extend(
|
|
137
|
+
deepcopy(rl) for rl in line.reference_lines
|
|
138
|
+
)
|
|
139
|
+
elif not other_as_refline_only:
|
|
140
|
+
pool[line.start] = deepcopy(line)
|
|
141
|
+
|
|
142
|
+
new.lines = list(pool.values())
|
|
143
|
+
return new
|
|
144
|
+
|
|
145
|
+
@classmethod
|
|
146
|
+
def loads(cls, s: str, *, fill_implicit_line_end: bool = False) -> Lyrics:
|
|
147
|
+
"""从 LRC 字符串解析出一份 :class:`Lyrics`.
|
|
148
|
+
|
|
149
|
+
Args:
|
|
150
|
+
s: LRC 源文本.
|
|
151
|
+
fill_implicit_line_end: 若为 ``True``, 则当某行没有显式结束时间时,
|
|
152
|
+
自动用下一行的开始时间作为其结束时间.
|
|
153
|
+
"""
|
|
154
|
+
from .parser import parse_lrc
|
|
155
|
+
|
|
156
|
+
return parse_lrc(s, fill_implicit_line_end=fill_implicit_line_end)
|
|
157
|
+
|
|
158
|
+
def dumps(
|
|
159
|
+
self,
|
|
160
|
+
*,
|
|
161
|
+
with_metadata: bool = True,
|
|
162
|
+
use_bracket_for_byword_tag: bool = False,
|
|
163
|
+
apply_offset_from_metadata: bool = False,
|
|
164
|
+
) -> str:
|
|
165
|
+
"""把当前对象序列化为 LRC 字符串.
|
|
166
|
+
|
|
167
|
+
Args:
|
|
168
|
+
with_metadata: 是否写出 metadata 段.
|
|
169
|
+
use_bracket_for_byword_tag: 逐字标签是否使用 ``[...]``
|
|
170
|
+
(默认 ``False`` 使用 ``<...>``) .
|
|
171
|
+
apply_offset_from_metadata: 是否读取并应用 ``metadata.offset``;
|
|
172
|
+
见 :func:`.serializer.dump_lrc` 的完整语义说明.
|
|
173
|
+
"""
|
|
174
|
+
from .serializer import dump_lrc
|
|
175
|
+
|
|
176
|
+
return dump_lrc(
|
|
177
|
+
self,
|
|
178
|
+
with_metadata=with_metadata,
|
|
179
|
+
use_bracket_for_byword_tag=use_bracket_for_byword_tag,
|
|
180
|
+
apply_offset_from_metadata=apply_offset_from_metadata,
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
def __str__(self) -> str:
|
|
184
|
+
return self.dumps()
|
|
@@ -0,0 +1,291 @@
|
|
|
1
|
+
"""LRC 解析器.
|
|
2
|
+
|
|
3
|
+
将 LRC 文本转换为 :class:`.models.Lyrics`. 公共入口有:
|
|
4
|
+
|
|
5
|
+
* :func:`parse_line` —— 解析单行歌词 (不含行首的重复时间标签) .
|
|
6
|
+
* :func:`parse_lrc` —— 解析整份 LRC 文本.
|
|
7
|
+
|
|
8
|
+
其它以下划线开头的函数均为内部实现细节, 后续版本可能调整.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from copy import deepcopy
|
|
15
|
+
from logging import getLogger
|
|
16
|
+
|
|
17
|
+
from .exceptions import LyricsParserError
|
|
18
|
+
from .models import BasicLyricLine, LyricLine, Lyrics, LyricWord
|
|
19
|
+
from .regex import (
|
|
20
|
+
GENERIC_TIMETAG_REGEX,
|
|
21
|
+
LINE_TIMETAG_REGEX,
|
|
22
|
+
METATAG_REGEX,
|
|
23
|
+
compile_regex,
|
|
24
|
+
)
|
|
25
|
+
from .timetag import _match_to_ms
|
|
26
|
+
|
|
27
|
+
logger = getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
__all__ = [
|
|
30
|
+
"parse_line",
|
|
31
|
+
"parse_lrc",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def parse_line(line: str) -> BasicLyricLine | None:
|
|
36
|
+
"""解析单行歌词为 :data:`.models.BasicLyricLine`.
|
|
37
|
+
|
|
38
|
+
输入行应已经去除了行首的重复行时间标签 (见 :func:`_split_leading_line_timetags`) .
|
|
39
|
+
若该行为空或只含空白, 返回 ``None``.
|
|
40
|
+
|
|
41
|
+
内部算法:
|
|
42
|
+
1. 用通用时间标签正则把原行拆成 ``text/match`` 交替序列.
|
|
43
|
+
2. 把序列拆成 ``texts`` (长度 N) 与 ``times`` (长度 N-1) 两条平行数组.
|
|
44
|
+
3. 丢弃非单调递增的时间标签 (视为误写并合并相邻文本) .
|
|
45
|
+
4. 用滑动窗口方式把 ``texts[i]`` 和 ``times[i-1] / times[i]`` 绑成
|
|
46
|
+
单个 :class:`LyricWord`.
|
|
47
|
+
5. 去掉首尾的空词, 使 ``result[0].start`` 成为行首、
|
|
48
|
+
``result[-1].end`` 成为行尾.
|
|
49
|
+
"""
|
|
50
|
+
if not line.strip():
|
|
51
|
+
return None
|
|
52
|
+
|
|
53
|
+
sequence = _split_on_timetags(line)
|
|
54
|
+
if not sequence or (len(sequence) == 1 and not sequence[0]):
|
|
55
|
+
return None
|
|
56
|
+
|
|
57
|
+
if len(sequence) % 2 != 1:
|
|
58
|
+
raise LyricsParserError(
|
|
59
|
+
f"Unexpected sequence length (expected odd): {len(sequence)}"
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
texts, times = _unzip_sequence(sequence)
|
|
63
|
+
|
|
64
|
+
# 只有一段纯文本、没有任何时间标签
|
|
65
|
+
if len(texts) == 1:
|
|
66
|
+
if times:
|
|
67
|
+
raise LyricsParserError(
|
|
68
|
+
"Inconsistent state: single text segment should not have time tags"
|
|
69
|
+
)
|
|
70
|
+
return [LyricWord(content=texts[0])]
|
|
71
|
+
|
|
72
|
+
diff = len(texts) - len(times)
|
|
73
|
+
if diff != 1:
|
|
74
|
+
raise LyricsParserError(
|
|
75
|
+
f"text/time length mismatch: expected diff=1, got {diff}"
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
texts, times = _drop_nonmonotonic_times(texts, times)
|
|
79
|
+
|
|
80
|
+
result: BasicLyricLine = []
|
|
81
|
+
last_idx = len(texts) - 1
|
|
82
|
+
for idx, content in enumerate(texts):
|
|
83
|
+
word = LyricWord(content=content)
|
|
84
|
+
if idx > 0:
|
|
85
|
+
word.start = times[idx - 1]
|
|
86
|
+
if idx < last_idx:
|
|
87
|
+
word.end = times[idx]
|
|
88
|
+
result.append(word)
|
|
89
|
+
|
|
90
|
+
if len(result) < 2:
|
|
91
|
+
raise LyricsParserError(
|
|
92
|
+
f"Expected at least 2 preprocessed elements, got {len(result)}"
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
# 去除空头/空尾, 使 [0].start 与 [-1].end 分别对应行首/行尾
|
|
96
|
+
if not result[0].content:
|
|
97
|
+
result.pop(0)
|
|
98
|
+
if not result[-1].content and len(result) > 1:
|
|
99
|
+
result.pop(-1)
|
|
100
|
+
|
|
101
|
+
return result
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _split_on_timetags(text: str) -> list[str | re.Match[str]]:
|
|
105
|
+
"""按通用时间标签把一行文本拆分为 ``[text, match, text, match, ..., text]``.
|
|
106
|
+
|
|
107
|
+
返回结果长度始终为奇数: 以文本开头、以文本结尾, 中间夹杂 match 对象.
|
|
108
|
+
若相邻的两个 match 之间没有文本, 会插入空字符串, 保证 text/match 严格交替.
|
|
109
|
+
"""
|
|
110
|
+
pattern = compile_regex(GENERIC_TIMETAG_REGEX)
|
|
111
|
+
result: list[str | re.Match[str]] = []
|
|
112
|
+
last_end = 0
|
|
113
|
+
for match in pattern.finditer(text):
|
|
114
|
+
result.append(text[last_end : match.start()])
|
|
115
|
+
result.append(match)
|
|
116
|
+
last_end = match.end()
|
|
117
|
+
result.append(text[last_end:])
|
|
118
|
+
return result
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _unzip_sequence(
|
|
122
|
+
sequence: list[str | re.Match[str]],
|
|
123
|
+
) -> tuple[list[str], list[int]]:
|
|
124
|
+
"""把 :func:`_split_on_timetags` 的结果拆成两条独立数组."""
|
|
125
|
+
texts: list[str] = []
|
|
126
|
+
times: list[int] = []
|
|
127
|
+
for item in sequence:
|
|
128
|
+
if isinstance(item, str):
|
|
129
|
+
texts.append(item)
|
|
130
|
+
elif isinstance(item, re.Match):
|
|
131
|
+
times.append(_match_to_ms(item))
|
|
132
|
+
else:
|
|
133
|
+
raise LyricsParserError(
|
|
134
|
+
f"Unexpected element type in sequence: {type(item).__name__}"
|
|
135
|
+
)
|
|
136
|
+
return texts, times
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _drop_nonmonotonic_times(
|
|
140
|
+
texts: list[str], times: list[int]
|
|
141
|
+
) -> tuple[list[str], list[int]]:
|
|
142
|
+
"""丢弃非严格递增的时间标签, 并把它们前后的文本合并."""
|
|
143
|
+
texts = list(texts)
|
|
144
|
+
times = list(times)
|
|
145
|
+
removed = 0
|
|
146
|
+
for raw_idx in range(len(times)):
|
|
147
|
+
idx = raw_idx - removed
|
|
148
|
+
if idx <= 0:
|
|
149
|
+
continue
|
|
150
|
+
prev_time = times[idx - 1]
|
|
151
|
+
now_time = times[idx]
|
|
152
|
+
if prev_time < now_time:
|
|
153
|
+
continue
|
|
154
|
+
logger.warning(f"Unordered time tag dropped: prev={prev_time}, now={now_time}")
|
|
155
|
+
texts[idx] += texts[idx + 1]
|
|
156
|
+
texts.pop(idx + 1)
|
|
157
|
+
times.pop(idx)
|
|
158
|
+
removed += 1
|
|
159
|
+
return texts, times
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def parse_lrc(lrc: str, *, fill_implicit_line_end: bool = False) -> Lyrics:
|
|
163
|
+
"""解析一份完整的 LRC 文本.
|
|
164
|
+
|
|
165
|
+
Args:
|
|
166
|
+
lrc: LRC 源文本.
|
|
167
|
+
fill_implicit_line_end: 若为 ``True``, 则对没有显式结束时间的行,
|
|
168
|
+
用紧随其后的行开始时间作为隐式结束时间.
|
|
169
|
+
|
|
170
|
+
Returns:
|
|
171
|
+
组装完毕的 :class:`Lyrics` 对象.
|
|
172
|
+
"""
|
|
173
|
+
metadata: dict[str, str] = {}
|
|
174
|
+
line_pool: dict[int, LyricLine] = {}
|
|
175
|
+
last_tag: int | None = None
|
|
176
|
+
|
|
177
|
+
for raw_line in lrc.strip().splitlines():
|
|
178
|
+
line_str = raw_line.strip()
|
|
179
|
+
|
|
180
|
+
# 1. metadata 行 (如 [ti: ...]、[offset: 500])
|
|
181
|
+
if meta := _extract_metadata(line_str):
|
|
182
|
+
metadata.update(meta)
|
|
183
|
+
logger.debug(f"Metadata line: {line_str!r}")
|
|
184
|
+
continue
|
|
185
|
+
|
|
186
|
+
logger.debug(f"Parsing lyric line: {line_str!r}")
|
|
187
|
+
|
|
188
|
+
# 2. 切出行首的重复时间标签
|
|
189
|
+
time_tags, line_str = _split_leading_line_timetags(line_str)
|
|
190
|
+
line = parse_line(line_str)
|
|
191
|
+
|
|
192
|
+
# 2a. 行首没有时间标签 → 要么是参考行, 要么是分隔符
|
|
193
|
+
if not time_tags:
|
|
194
|
+
if not line:
|
|
195
|
+
# 空分隔行, 重置参考行锚点
|
|
196
|
+
last_tag = None
|
|
197
|
+
logger.debug("Reference line marker reset")
|
|
198
|
+
continue
|
|
199
|
+
if last_tag is None:
|
|
200
|
+
logger.warning(
|
|
201
|
+
f"Orphaned lyric line (no anchor): {line!r} (raw={raw_line!r})"
|
|
202
|
+
)
|
|
203
|
+
continue
|
|
204
|
+
logger.debug(f"Adding {line!r} as reference of {line_pool[last_tag]!r}")
|
|
205
|
+
line_pool[last_tag].reference_lines.append(line)
|
|
206
|
+
continue
|
|
207
|
+
|
|
208
|
+
# 2b. 行首有时间标签但没有正文 → 占位符 (例如清空当前歌词)
|
|
209
|
+
if not line:
|
|
210
|
+
t = time_tags[0]
|
|
211
|
+
if t not in line_pool:
|
|
212
|
+
line_pool[t] = LyricLine(start=t, content=[LyricWord(content="")])
|
|
213
|
+
continue
|
|
214
|
+
|
|
215
|
+
# 2c. 常规行: 可能有多个重复时间标签, 每个都生成一行
|
|
216
|
+
_register_line_at_tags(line_pool, line, time_tags)
|
|
217
|
+
last_tag = time_tags[0] if len(time_tags) == 1 else None
|
|
218
|
+
|
|
219
|
+
return _finalize_lyrics(
|
|
220
|
+
metadata, line_pool, fill_implicit_line_end=fill_implicit_line_end
|
|
221
|
+
)
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _register_line_at_tags(
|
|
225
|
+
line_pool: dict[int, LyricLine],
|
|
226
|
+
line: BasicLyricLine,
|
|
227
|
+
time_tags: list[int],
|
|
228
|
+
) -> None:
|
|
229
|
+
"""把同一行歌词注册到 ``line_pool`` 中所有 ``time_tags`` 对应的时间点上."""
|
|
230
|
+
word_start = line[0].start
|
|
231
|
+
for tag in time_tags:
|
|
232
|
+
if word_start is not None and word_start < tag:
|
|
233
|
+
logger.warning(
|
|
234
|
+
f"Invalid duplicate line tag {tag}ms "
|
|
235
|
+
f"(later than first word start {word_start}ms)"
|
|
236
|
+
)
|
|
237
|
+
continue
|
|
238
|
+
if tag in line_pool:
|
|
239
|
+
# 同一个时间点已有行 → 当前行变为参考行
|
|
240
|
+
line_pool[tag].reference_lines.append(line)
|
|
241
|
+
else:
|
|
242
|
+
# 深拷贝 word 列表, 避免多个 LyricLine 共享同一 LyricWord 实例
|
|
243
|
+
line_pool[tag] = LyricLine(content=[deepcopy(word) for word in line])
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _finalize_lyrics(
|
|
247
|
+
metadata: dict[str, str],
|
|
248
|
+
line_pool: dict[int, LyricLine],
|
|
249
|
+
*,
|
|
250
|
+
fill_implicit_line_end: bool,
|
|
251
|
+
) -> Lyrics:
|
|
252
|
+
"""把 ``line_pool`` 按时间排序、补全行首/行尾时间并装进 :class:`Lyrics`."""
|
|
253
|
+
lyrics = Lyrics(metadata=metadata)
|
|
254
|
+
sorted_items = sorted(line_pool.items(), key=lambda kv: kv[0])
|
|
255
|
+
|
|
256
|
+
for idx, (line_start, line) in enumerate(sorted_items):
|
|
257
|
+
line.start = line_start
|
|
258
|
+
|
|
259
|
+
# 把最后一个 word 的 end 提升为整行 end
|
|
260
|
+
last_word = line.content[-1]
|
|
261
|
+
if last_word.end is not None:
|
|
262
|
+
line.end, last_word.end = last_word.end, None
|
|
263
|
+
|
|
264
|
+
# 可选: 用下一行的开始时间作为当前行的隐式结束
|
|
265
|
+
if fill_implicit_line_end and line.end is None and idx + 1 < len(sorted_items):
|
|
266
|
+
line.end = sorted_items[idx + 1][0]
|
|
267
|
+
|
|
268
|
+
lyrics.lines.append(line)
|
|
269
|
+
|
|
270
|
+
return lyrics
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _split_leading_line_timetags(raw_line: str) -> tuple[list[int], str]:
|
|
274
|
+
"""从一行开头连续剥离 ``[mm:ss.xxx]`` 行时间标签.
|
|
275
|
+
|
|
276
|
+
Returns:
|
|
277
|
+
``(times, remainder)``, ``times`` 为毫秒列表, ``remainder`` 为剥离后
|
|
278
|
+
的剩余文本.
|
|
279
|
+
"""
|
|
280
|
+
pattern = compile_regex(f"^{LINE_TIMETAG_REGEX}")
|
|
281
|
+
times: list[int] = []
|
|
282
|
+
while (match := pattern.match(raw_line)) is not None:
|
|
283
|
+
times.append(_match_to_ms(match))
|
|
284
|
+
raw_line = raw_line[match.end() :]
|
|
285
|
+
return times, raw_line
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _extract_metadata(line: str) -> dict[str, str]:
|
|
289
|
+
"""从一行字符串中提取 metadata 标签 ``[key: value]``."""
|
|
290
|
+
pattern = compile_regex(METATAG_REGEX)
|
|
291
|
+
return {match["key"]: match["value"] for match in pattern.finditer(line)}
|
|
File without changes
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""正则表达式常量与编译缓存.
|
|
2
|
+
|
|
3
|
+
本模块集中存放 LRC 语法所需的正则模式, 并提供带缓存的 :func:`compile_regex`.
|
|
4
|
+
这些常量会被 :mod:`.parser`、:mod:`.timetag` 等模块消费, 不建议外部直接依赖.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"GENERIC_TIMETAG_REGEX",
|
|
13
|
+
"LINE_TIMETAG_REGEX",
|
|
14
|
+
"METATAG_REGEX",
|
|
15
|
+
"TIMETAG_REGEX_STRICT",
|
|
16
|
+
"WORD_TIMETAG_REGEX",
|
|
17
|
+
"compile_regex",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
_REGEX_PATTERN_CACHE: dict[str, re.Pattern[str]] = {}
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def compile_regex(pattern: str) -> re.Pattern[str]:
|
|
24
|
+
"""使用 ``re.VERBOSE`` 编译正则, 并以模式字符串为 key 做进程级缓存."""
|
|
25
|
+
compiled = _REGEX_PATTERN_CACHE.get(pattern)
|
|
26
|
+
if compiled is None:
|
|
27
|
+
compiled = re.compile(pattern, flags=re.VERBOSE)
|
|
28
|
+
_REGEX_PATTERN_CACHE[pattern] = compiled
|
|
29
|
+
return compiled
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
#: 行时间标签 ``[mm:ss.xxx]``, 命名组: ``min`` / ``sec`` / ``tail``.
|
|
33
|
+
LINE_TIMETAG_REGEX: str = r"""
|
|
34
|
+
(?:
|
|
35
|
+
\[
|
|
36
|
+
\s*
|
|
37
|
+
(?P<min>\d{1,4})
|
|
38
|
+
\s*
|
|
39
|
+
:
|
|
40
|
+
\s*
|
|
41
|
+
(?P<sec>\d{1,2})
|
|
42
|
+
\s*
|
|
43
|
+
(?:
|
|
44
|
+
[:\.]
|
|
45
|
+
\s*
|
|
46
|
+
(?P<tail>\d{1,6})
|
|
47
|
+
\s*
|
|
48
|
+
)?
|
|
49
|
+
\]
|
|
50
|
+
)
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
#: 逐字时间标签 ``<mm:ss.xxx>``, 命名组: ``min`` / ``sec`` / ``tail``.
|
|
54
|
+
WORD_TIMETAG_REGEX: str = r"""
|
|
55
|
+
(?:
|
|
56
|
+
\<
|
|
57
|
+
\s*
|
|
58
|
+
(?P<min>\d{1,4})
|
|
59
|
+
\s*
|
|
60
|
+
:
|
|
61
|
+
\s*
|
|
62
|
+
(?P<sec>\d{1,2})
|
|
63
|
+
\s*
|
|
64
|
+
(?:
|
|
65
|
+
[:\.]
|
|
66
|
+
\s*
|
|
67
|
+
(?P<tail>\d{1,6})
|
|
68
|
+
\s*
|
|
69
|
+
)?
|
|
70
|
+
\>
|
|
71
|
+
)
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
#: 严格行时间标签 ``[mm:ss.xxx]``, 要求三段齐全、毫秒 1-3 位、无多余空白.
|
|
75
|
+
TIMETAG_REGEX_STRICT: str = r"""
|
|
76
|
+
(?:
|
|
77
|
+
\[
|
|
78
|
+
(?P<min>\d{1,4})
|
|
79
|
+
:
|
|
80
|
+
(?P<sec>\d{1,2})
|
|
81
|
+
\.
|
|
82
|
+
(?P<tail>\d{1,3})
|
|
83
|
+
\]
|
|
84
|
+
)
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
#: 元数据标签 ``[key: value]``, 命名组: ``key`` / ``value``.
|
|
88
|
+
METATAG_REGEX: str = r"""
|
|
89
|
+
(?:
|
|
90
|
+
\[
|
|
91
|
+
\s*
|
|
92
|
+
(?P<key>[a-zA-Z#]{2,16}) # `#` 用于注释标签, 见 LRC 规范
|
|
93
|
+
\s*
|
|
94
|
+
:
|
|
95
|
+
\s*
|
|
96
|
+
(?P<value>.+?)
|
|
97
|
+
\s*
|
|
98
|
+
\]
|
|
99
|
+
)
|
|
100
|
+
"""
|
|
101
|
+
|
|
102
|
+
#: 通用时间标签 (同时匹配方括号行标签与尖括号逐字标签) .
|
|
103
|
+
#:
|
|
104
|
+
#: 为避免同名命名组冲突, 方括号分支使用 ``line_*`` 前缀,
|
|
105
|
+
#: 尖括号分支使用 ``word_*`` 前缀. 消费方应使用 :func:`._match_to_ms` 抹平差异.
|
|
106
|
+
GENERIC_TIMETAG_REGEX: str = r"""
|
|
107
|
+
(?:
|
|
108
|
+
(?:
|
|
109
|
+
\[
|
|
110
|
+
\s*
|
|
111
|
+
(?P<line_min>\d{1,4})
|
|
112
|
+
\s*
|
|
113
|
+
:
|
|
114
|
+
\s*
|
|
115
|
+
(?P<line_sec>\d{1,2})
|
|
116
|
+
\s*
|
|
117
|
+
(?:
|
|
118
|
+
[:\.]
|
|
119
|
+
\s*
|
|
120
|
+
(?P<line_tail>\d{1,6})
|
|
121
|
+
\s*
|
|
122
|
+
)?
|
|
123
|
+
\]
|
|
124
|
+
)
|
|
125
|
+
|
|
|
126
|
+
(?:
|
|
127
|
+
\<
|
|
128
|
+
\s*
|
|
129
|
+
(?P<word_min>\d{1,4})
|
|
130
|
+
\s*
|
|
131
|
+
:
|
|
132
|
+
\s*
|
|
133
|
+
(?P<word_sec>\d{1,2})
|
|
134
|
+
\s*
|
|
135
|
+
(?:
|
|
136
|
+
[:\.]
|
|
137
|
+
\s*
|
|
138
|
+
(?P<word_tail>\d{1,6})
|
|
139
|
+
\s*
|
|
140
|
+
)?
|
|
141
|
+
\>
|
|
142
|
+
)
|
|
143
|
+
)
|
|
144
|
+
"""
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _warmup_cache() -> None:
|
|
148
|
+
"""在模块导入期预热编译缓存, 避免首次使用时的抖动."""
|
|
149
|
+
for pattern in (
|
|
150
|
+
LINE_TIMETAG_REGEX,
|
|
151
|
+
WORD_TIMETAG_REGEX,
|
|
152
|
+
TIMETAG_REGEX_STRICT,
|
|
153
|
+
METATAG_REGEX,
|
|
154
|
+
GENERIC_TIMETAG_REGEX,
|
|
155
|
+
):
|
|
156
|
+
compile_regex(pattern)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
_warmup_cache()
|
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
"""LRC 序列化器.
|
|
2
|
+
|
|
3
|
+
把 :class:`.models.Lyrics` 对象序列化为 LRC 文本. 公共入口是
|
|
4
|
+
:func:`dump_lrc`
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Iterator
|
|
10
|
+
from io import StringIO
|
|
11
|
+
from logging import getLogger
|
|
12
|
+
|
|
13
|
+
from .models import BasicLyricLine, Lyrics
|
|
14
|
+
from .timetag import format_timetag
|
|
15
|
+
|
|
16
|
+
logger = getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"dump_lrc",
|
|
20
|
+
]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def dump_lrc(
|
|
24
|
+
lyrics: Lyrics,
|
|
25
|
+
*,
|
|
26
|
+
with_metadata: bool = True,
|
|
27
|
+
use_bracket_for_byword_tag: bool = False,
|
|
28
|
+
apply_offset_from_metadata: bool = False,
|
|
29
|
+
) -> str:
|
|
30
|
+
"""把一份 :class:`Lyrics` 序列化为 LRC 文本.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
lyrics: 待序列化的歌词对象.
|
|
34
|
+
with_metadata: 是否写出 metadata 段 (不影响 offset 的处理逻辑) .
|
|
35
|
+
use_bracket_for_byword_tag: 若为 ``True``, 逐字标签使用 ``[...]``;
|
|
36
|
+
否则使用 ``<...>``.
|
|
37
|
+
apply_offset_from_metadata: 是否读取并应用 ``metadata.offset``.
|
|
38
|
+
见下方“offset 语义”.
|
|
39
|
+
|
|
40
|
+
offset 语义 (与 LRC 规范一致) :
|
|
41
|
+
正 offset 会让歌词显示提前, 即 ``display_time = tag_time - offset``.
|
|
42
|
+
当 ``offset > 0`` 时, 最早的时间戳可能变为负数; 此时函数只应用
|
|
43
|
+
``min(all_times)`` 这部分“安全” offset, 把剩余量写回
|
|
44
|
+
``metadata.offset`` 交给播放器处理, 确保输出的所有时间标签 ``>= 0``.
|
|
45
|
+
"""
|
|
46
|
+
buffer = StringIO()
|
|
47
|
+
|
|
48
|
+
# 拷贝 metadata 以免污染调用方传入的对象
|
|
49
|
+
metadata = dict(lyrics.metadata)
|
|
50
|
+
offset = _resolve_offset(
|
|
51
|
+
lyrics, metadata, apply_offset_from_metadata=apply_offset_from_metadata
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
if with_metadata:
|
|
55
|
+
for key, value in metadata.items():
|
|
56
|
+
buffer.write(f"[{key}: {value}]\n")
|
|
57
|
+
|
|
58
|
+
for idx, line in enumerate(lyrics.lines):
|
|
59
|
+
if idx > 0:
|
|
60
|
+
buffer.write("\n")
|
|
61
|
+
|
|
62
|
+
line_start = line.start
|
|
63
|
+
if line_start is None:
|
|
64
|
+
logger.warning(f"Skipping line with unknown start time: {line}")
|
|
65
|
+
continue
|
|
66
|
+
|
|
67
|
+
# 写主行
|
|
68
|
+
buffer.write(format_timetag(line_start - offset))
|
|
69
|
+
buffer.write(
|
|
70
|
+
_format_words(
|
|
71
|
+
line.content,
|
|
72
|
+
line_start=line_start,
|
|
73
|
+
offset=offset,
|
|
74
|
+
use_bracket_for_byword_tag=use_bracket_for_byword_tag,
|
|
75
|
+
)
|
|
76
|
+
)
|
|
77
|
+
if line.end is not None:
|
|
78
|
+
buffer.write(format_timetag(line.end - offset))
|
|
79
|
+
buffer.write("\n")
|
|
80
|
+
|
|
81
|
+
# 写参考行 (共享主行的 start)
|
|
82
|
+
for refline in line.reference_lines:
|
|
83
|
+
buffer.write(format_timetag(line_start - offset))
|
|
84
|
+
buffer.write(
|
|
85
|
+
_format_words(
|
|
86
|
+
refline,
|
|
87
|
+
line_start=line_start,
|
|
88
|
+
offset=offset,
|
|
89
|
+
use_bracket_for_byword_tag=use_bracket_for_byword_tag,
|
|
90
|
+
)
|
|
91
|
+
)
|
|
92
|
+
buffer.write("\n")
|
|
93
|
+
|
|
94
|
+
return buffer.getvalue()
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _format_words(
|
|
98
|
+
words: BasicLyricLine,
|
|
99
|
+
*,
|
|
100
|
+
line_start: int | None,
|
|
101
|
+
offset: int,
|
|
102
|
+
use_bracket_for_byword_tag: bool,
|
|
103
|
+
) -> str:
|
|
104
|
+
"""把一行 :data:`BasicLyricLine` 格式化为字符串 (不含行首/行末标签) .
|
|
105
|
+
|
|
106
|
+
逐字标签在以下情形会被省略:
|
|
107
|
+
|
|
108
|
+
* ``idx == 0`` 且 ``word.start == line_start`` —— 行首时间已由调用方
|
|
109
|
+
输出过, 不重复.
|
|
110
|
+
* ``idx > 0`` 且 ``words[idx - 1].end == word.start`` —— 与前一词的
|
|
111
|
+
结束时间相接, 可省略前缀.
|
|
112
|
+
"""
|
|
113
|
+
use_angle = not use_bracket_for_byword_tag
|
|
114
|
+
parts: list[str] = []
|
|
115
|
+
|
|
116
|
+
for idx, word in enumerate(words):
|
|
117
|
+
prefix = ""
|
|
118
|
+
suffix = ""
|
|
119
|
+
|
|
120
|
+
if word.start is not None:
|
|
121
|
+
if idx == 0:
|
|
122
|
+
if word.start != line_start:
|
|
123
|
+
prefix = format_timetag(
|
|
124
|
+
word.start - offset, use_angle_bracket=use_angle
|
|
125
|
+
)
|
|
126
|
+
elif words[idx - 1].end != word.start:
|
|
127
|
+
prefix = format_timetag(
|
|
128
|
+
word.start - offset, use_angle_bracket=use_angle
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
if word.end is not None:
|
|
132
|
+
suffix = format_timetag(word.end - offset, use_angle_bracket=use_angle)
|
|
133
|
+
|
|
134
|
+
parts.append(f"{prefix}{word.content}{suffix}")
|
|
135
|
+
|
|
136
|
+
return "".join(parts)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _resolve_offset(
|
|
140
|
+
lyrics: Lyrics,
|
|
141
|
+
metadata: dict[str, str],
|
|
142
|
+
*,
|
|
143
|
+
apply_offset_from_metadata: bool,
|
|
144
|
+
) -> int:
|
|
145
|
+
"""决定实际应用的 offset 值, 必要时把剩余部分写回 ``metadata``.
|
|
146
|
+
|
|
147
|
+
Args:
|
|
148
|
+
lyrics: 原始歌词对象 (只读, 用于收集时间戳) .
|
|
149
|
+
metadata: 调用方已经拷贝出的 metadata 字典.
|
|
150
|
+
``offset`` 键可能被 pop 出来 (``apply_offset_from_metadata=True``
|
|
151
|
+
时) 或更新为剩余 offset.
|
|
152
|
+
apply_offset_from_metadata: 是否启用 offset 处理.
|
|
153
|
+
|
|
154
|
+
Returns:
|
|
155
|
+
实际应从每个时间戳中扣除的毫秒数, 保证不会让任何时间戳变为负.
|
|
156
|
+
"""
|
|
157
|
+
if not apply_offset_from_metadata:
|
|
158
|
+
return 0
|
|
159
|
+
|
|
160
|
+
offset_str = metadata.pop("offset", None)
|
|
161
|
+
if not offset_str:
|
|
162
|
+
return 0
|
|
163
|
+
|
|
164
|
+
try:
|
|
165
|
+
offset = int(offset_str)
|
|
166
|
+
except ValueError:
|
|
167
|
+
logger.warning(
|
|
168
|
+
f"Cannot parse metadata.offset as integer, ignoring: {offset_str!r}"
|
|
169
|
+
)
|
|
170
|
+
return 0
|
|
171
|
+
|
|
172
|
+
logger.info(f"Applying global time offset: {offset}ms (from metadata.offset)")
|
|
173
|
+
|
|
174
|
+
# 负 offset 只会让时间戳整体变大, 无需做越界保护
|
|
175
|
+
if offset <= 0:
|
|
176
|
+
return offset
|
|
177
|
+
|
|
178
|
+
all_times = list(_iter_all_timestamps(lyrics))
|
|
179
|
+
if not all_times:
|
|
180
|
+
return offset
|
|
181
|
+
|
|
182
|
+
min_time = min(all_times)
|
|
183
|
+
if min_time - offset >= 0:
|
|
184
|
+
return offset
|
|
185
|
+
|
|
186
|
+
# 正 offset 超过最小时间戳 → 只应用“安全”部分, 剩余写回 metadata
|
|
187
|
+
remaining = offset - min_time
|
|
188
|
+
logger.warning(
|
|
189
|
+
f"Applying offset={offset}ms would make minimum timestamp {min_time}ms "
|
|
190
|
+
f"negative; only applying {min_time}ms, remaining {remaining}ms kept in "
|
|
191
|
+
f"metadata.offset"
|
|
192
|
+
)
|
|
193
|
+
metadata["offset"] = str(remaining)
|
|
194
|
+
return min_time
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def _iter_all_timestamps(lyrics: Lyrics) -> Iterator[int]:
|
|
198
|
+
"""迭代 :class:`Lyrics` 中出现过的所有时间戳 (含参考行) ."""
|
|
199
|
+
for line in lyrics.lines:
|
|
200
|
+
if line.start is not None:
|
|
201
|
+
yield line.start
|
|
202
|
+
if line.end is not None:
|
|
203
|
+
yield line.end
|
|
204
|
+
yield from _iter_word_timestamps(line.content)
|
|
205
|
+
for refline in line.reference_lines:
|
|
206
|
+
yield from _iter_word_timestamps(refline)
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _iter_word_timestamps(words: BasicLyricLine) -> Iterator[int]:
|
|
210
|
+
"""从一个 word 序列中迭代出所有非空时间戳."""
|
|
211
|
+
for word in words:
|
|
212
|
+
if word.start is not None:
|
|
213
|
+
yield word.start
|
|
214
|
+
if word.end is not None:
|
|
215
|
+
yield word.end
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""时间标签处理工具.
|
|
2
|
+
|
|
3
|
+
集中管理 LRC 时间标签 (``[mm:ss.xxx]`` / ``<mm:ss.xxx>``) 的解析与格式化.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import re
|
|
9
|
+
from logging import getLogger
|
|
10
|
+
|
|
11
|
+
from .regex import TIMETAG_REGEX_STRICT, compile_regex
|
|
12
|
+
|
|
13
|
+
logger = getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"format_timetag",
|
|
17
|
+
"parse_timetag",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def format_timetag(
|
|
22
|
+
ms: int,
|
|
23
|
+
*,
|
|
24
|
+
use_angle_bracket: bool = False,
|
|
25
|
+
tail_digits: int = 3,
|
|
26
|
+
) -> str:
|
|
27
|
+
"""将毫秒数格式化为 LRC 时间标签字符串.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
ms: 毫秒时间戳 (允许为负, 调用方应自行保证语义合理) .
|
|
31
|
+
use_angle_bracket: True 使用 ``<...>`` (逐字标签) , False 使用 ``[...]`` (行标签) .
|
|
32
|
+
tail_digits: 毫秒尾部补齐的位数, 默认为 3.
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
形如 ``[01:23.456]`` 或 ``<01:23.456>`` 的字符串.
|
|
36
|
+
"""
|
|
37
|
+
if ms < 0:
|
|
38
|
+
raise ValueError(f"Negative timestamp is not allowed: {ms}ms")
|
|
39
|
+
minutes = ms // 60_000
|
|
40
|
+
seconds = (ms % 60_000) // 1000
|
|
41
|
+
millis = ms % 1000
|
|
42
|
+
body = f"{minutes:02d}:{seconds:02d}.{millis:0{tail_digits}d}"
|
|
43
|
+
return f"<{body}>" if use_angle_bracket else f"[{body}]"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def parse_timetag(s: str) -> int | None:
|
|
47
|
+
"""解析一个严格格式的时间标签字符串, 返回对应毫秒数.
|
|
48
|
+
|
|
49
|
+
严格格式要求形如 ``[mm:ss.xxx]`` (方括号、三段齐全、毫秒 1-3 位) .
|
|
50
|
+
解析失败返回 ``None``.
|
|
51
|
+
"""
|
|
52
|
+
match = compile_regex(rf"^{TIMETAG_REGEX_STRICT}$").match(s)
|
|
53
|
+
return _match_to_ms(match) if match else None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _match_to_ms(match: re.Match[str]) -> int:
|
|
57
|
+
"""从正则匹配对象中提取毫秒数.
|
|
58
|
+
|
|
59
|
+
兼容两类命名组:
|
|
60
|
+
|
|
61
|
+
* 标准命名组 ``min`` / ``sec`` / ``tail`` (见 ``LINE_TIMETAG_REGEX`` 等) .
|
|
62
|
+
* 前缀命名组 ``line_min`` / ``word_min`` 等 (见 ``GENERIC_TIMETAG_REGEX``) .
|
|
63
|
+
|
|
64
|
+
该函数是包内部工具, 不导出到公共 API.
|
|
65
|
+
"""
|
|
66
|
+
groups = match.groupdict()
|
|
67
|
+
|
|
68
|
+
# 优先使用前缀命名组, 再退回到标准命名组
|
|
69
|
+
min_val = groups.get("line_min") or groups.get("word_min") or groups.get("min")
|
|
70
|
+
sec_val = groups.get("line_sec") or groups.get("word_sec") or groups.get("sec")
|
|
71
|
+
tail_val = groups.get("line_tail") or groups.get("word_tail") or groups.get("tail")
|
|
72
|
+
|
|
73
|
+
minutes = int(min_val or 0)
|
|
74
|
+
seconds = int(sec_val or 0)
|
|
75
|
+
|
|
76
|
+
if tail_val:
|
|
77
|
+
# 将毫秒标准化到 3 位
|
|
78
|
+
if len(tail_val) > 3:
|
|
79
|
+
tail_val = tail_val[:3] # 截断: "123456" -> "123"
|
|
80
|
+
elif len(tail_val) < 3:
|
|
81
|
+
tail_val = tail_val.ljust(3, "0") # 补齐: "1" -> "100"
|
|
82
|
+
millis = int(tail_val)
|
|
83
|
+
else:
|
|
84
|
+
millis = 0
|
|
85
|
+
|
|
86
|
+
return millis + seconds * 1000 + minutes * 60_000
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: lemony-lrc-parser
|
|
3
|
+
Version: 0.2.1
|
|
4
|
+
Summary: A simple LRC parser for Python.
|
|
5
|
+
Project-URL: Homepage, https://github.com/NingmengLemon/lemony-lrc-parser
|
|
6
|
+
Project-URL: Repository, https://github.com/NingmengLemon/lemony-lrc-parser.git
|
|
7
|
+
Project-URL: Issues, https://github.com/NingmengLemon/lemony-lrc-parser/issues
|
|
8
|
+
Author-email: LemonekoUwU <lemoneko233uwu@gmail.com>
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: Enhanced LRC,LRC,SPL,lyrics,parser,serializer
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.15
|
|
24
|
+
Classifier: Topic :: Multimedia :: Sound/Audio
|
|
25
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
26
|
+
Classifier: Typing :: Typed
|
|
27
|
+
Requires-Python: <3.16,>=3.9
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
|
|
30
|
+
# Lemony LRC Parser
|
|
31
|
+
|
|
32
|
+
[](https://www.python.org/)
|
|
33
|
+
[](https://opensource.org/licenses/MIT)
|
|
34
|
+
[](https://pypi.org/project/lemony-lrc-parser/)
|
|
35
|
+
|
|
36
|
+
柠檬味的 Python LRC 歌词解析器.
|
|
37
|
+
|
|
38
|
+
Lemon-flavored LRC Parser for Python.
|
|
39
|
+
|
|
40
|
+
## Features
|
|
41
|
+
|
|
42
|
+
- 解析标准 LRC 歌词文件
|
|
43
|
+
- 支持 EnhancedLRC / SPL 的逐字歌词标签
|
|
44
|
+
- 支持 metadata 标签
|
|
45
|
+
- 支持折叠时间标签
|
|
46
|
+
- 支持参照行
|
|
47
|
+
- 支持歌词合并
|
|
48
|
+
- 完整的类型注解
|
|
49
|
+
|
|
50
|
+
## Installation
|
|
51
|
+
|
|
52
|
+
推荐使用 [uv](https://docs.astral.sh/uv/).
|
|
53
|
+
|
|
54
|
+
It's recommend to use [uv](https://docs.astral.sh/uv/).
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
uv add lemony-lrc-parser
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
用 pip 也行.
|
|
61
|
+
|
|
62
|
+
It's okay to use pip.
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
pip install lemony-lrc-parser
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
## Usage
|
|
69
|
+
|
|
70
|
+
### Quick Start
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
import lemony_lrc_parser as llp
|
|
74
|
+
|
|
75
|
+
lrc_text = """[ti: Never Gonna Give You Up]
|
|
76
|
+
[ar: Rick Astley]
|
|
77
|
+
|
|
78
|
+
[00:18.684]We're no strangers to love
|
|
79
|
+
[00:18.684]我们都是情场老手
|
|
80
|
+
|
|
81
|
+
[00:22.657]You know the rules and so do I
|
|
82
|
+
[00:22.657]你和我都知道爱情的规则
|
|
83
|
+
|
|
84
|
+
[00:27.070]A full commitment's what I'm thinking of
|
|
85
|
+
[00:27.070]我在想的正是一份实打实的承诺
|
|
86
|
+
|
|
87
|
+
[00:31.459]You wouldn't get this from any other guy
|
|
88
|
+
[00:31.459]你从其他人那里得不到的
|
|
89
|
+
"""
|
|
90
|
+
|
|
91
|
+
# 解析
|
|
92
|
+
lyrics = llp.loads(lrc_text)
|
|
93
|
+
|
|
94
|
+
# 访问 metadata
|
|
95
|
+
print(lyrics.metadata["ti"]) # "Never Gonna Give You Up"
|
|
96
|
+
print(lyrics.metadata["ar"]) # "Rick Astley"
|
|
97
|
+
|
|
98
|
+
# 遍历歌词行
|
|
99
|
+
for line in lyrics:
|
|
100
|
+
print(f"{line.start}ms: {line.text}")
|
|
101
|
+
|
|
102
|
+
# 序列化回 LRC 文本
|
|
103
|
+
output = llp.dumps(lyrics)
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
### OOP Interface
|
|
107
|
+
|
|
108
|
+
`Lyrics` 类提供面向对象的解析和序列化入口:
|
|
109
|
+
|
|
110
|
+
```python
|
|
111
|
+
from lemony_lrc_parser import Lyrics
|
|
112
|
+
|
|
113
|
+
lyrics = Lyrics.loads(lrc_text)
|
|
114
|
+
|
|
115
|
+
# Lyrics 同时是序列容器
|
|
116
|
+
print(len(lyrics)) # 行数
|
|
117
|
+
print(lyrics[0].text) # 第一行文本
|
|
118
|
+
print(lyrics[-1].text)
|
|
119
|
+
|
|
120
|
+
# 切片访问
|
|
121
|
+
first_three = lyrics[0:3]
|
|
122
|
+
|
|
123
|
+
# 序列化
|
|
124
|
+
lrc_output = lyrics.dumps()
|
|
125
|
+
|
|
126
|
+
# __str__ 等价于 dumps()
|
|
127
|
+
print(lyrics)
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
### Word-level Lyrics
|
|
131
|
+
|
|
132
|
+
解析逐字 (Enhanced LRC / SPL) 歌词:
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
lrc_text = "[00:01.000]<00:01.000>Never <00:01.500>gonna <00:02.000>give <00:02.500>you <00:03.000>up[00:03.500]"
|
|
136
|
+
|
|
137
|
+
lyrics = llp.loads(lrc_text)
|
|
138
|
+
line = lyrics[0]
|
|
139
|
+
|
|
140
|
+
for word in line.content:
|
|
141
|
+
print(f" [{word.start} -> {word.end}] {word.content!r}")
|
|
142
|
+
# [1000 -> 1500] 'Never '
|
|
143
|
+
# [1500 -> 2000] 'gonna '
|
|
144
|
+
# [2000 -> 2500] 'give '
|
|
145
|
+
# [2500 -> 3000] 'you '
|
|
146
|
+
# [3000 -> None] 'up'
|
|
147
|
+
|
|
148
|
+
# 行级时间: line.start=1000, line.end=3500
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
### Reference Lines (Translation / Transliteration)
|
|
152
|
+
|
|
153
|
+
LRC 文件中, 紧跟在带时间标签行后面的无标签行, 或与主行的时间戳相同的行, 会被解析为参考行, 常用于存放翻译或音译:
|
|
154
|
+
|
|
155
|
+
```python
|
|
156
|
+
lrc_text = """[00:01.000]Hello
|
|
157
|
+
你好
|
|
158
|
+
[00:02.000]World
|
|
159
|
+
[00:02.000]世界
|
|
160
|
+
"""
|
|
161
|
+
|
|
162
|
+
lyrics = llp.loads(lrc_text)
|
|
163
|
+
|
|
164
|
+
line = lyrics[0]
|
|
165
|
+
print(line.text) # "Hello"
|
|
166
|
+
print(line.reference_lines[0][0].content) # "你好"
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
### Combining Lyrics
|
|
170
|
+
|
|
171
|
+
将两份歌词 (如原文和翻译) 按时间标签合并:
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
main = llp.loads("[00:01.000]Hello\n[00:02.000]World\n")
|
|
175
|
+
translation = llp.loads("[00:01.000]你好\n[00:02.000]世界\n")
|
|
176
|
+
|
|
177
|
+
# combine 方法: 翻译行挂到同时间点的 reference_lines 中
|
|
178
|
+
combined = main.combine(translation)
|
|
179
|
+
|
|
180
|
+
# 也可以用 + 运算符
|
|
181
|
+
combined = main + translation
|
|
182
|
+
|
|
183
|
+
for line in combined:
|
|
184
|
+
print(line.text) # 主歌词
|
|
185
|
+
for ref in line.reference_lines:
|
|
186
|
+
ref_text = "".join(w.content for w in ref)
|
|
187
|
+
print(f" -> {ref_text}") # 参考行
|
|
188
|
+
|
|
189
|
+
# other_as_refline_only=False 时, 翻译中找不到对应时间点的行会作为新行保留
|
|
190
|
+
combined = main.combine(translation, other_as_refline_only=False)
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
### Implicit Line End
|
|
194
|
+
|
|
195
|
+
当歌词行没有显式结束时间时, 可以自动用下一行的开始时间填充:
|
|
196
|
+
|
|
197
|
+
```python
|
|
198
|
+
lyrics = llp.loads(lrc_text, fill_implicit_line_end=True)
|
|
199
|
+
|
|
200
|
+
# lyrics[0].end == lyrics[1].start
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
### Serialization Options
|
|
204
|
+
|
|
205
|
+
`dumps` 支持以下选项:
|
|
206
|
+
|
|
207
|
+
```python
|
|
208
|
+
output = lyrics.dumps(
|
|
209
|
+
with_metadata=True, # 是否输出 metadata 段
|
|
210
|
+
use_bracket_for_byword_tag=False, # 逐字标签使用 [...] 还是 <...> (默认)
|
|
211
|
+
apply_offset_from_metadata=False, # 是否读取并应用 metadata 中的 offset
|
|
212
|
+
)
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
`apply_offset_from_metadata` 的行为: 正 offset 会让歌词显示提前 (`display_time = tag_time - offset`) . 如果应用 offset 会导致时间戳变负, 则只应用安全的部分, 剩余写回 `metadata["offset"]`.
|
|
216
|
+
|
|
217
|
+
## References
|
|
218
|
+
|
|
219
|
+
[LRC Wikipedia](https://en.wikipedia.org/wiki/LRC_(file_format))
|
|
220
|
+
|
|
221
|
+
[SPL Specification](https://moriafly.com/standards/spl.html)
|
|
222
|
+
|
|
223
|
+
## UwU?
|
|
224
|
+
|
|
225
|
+
UwU!
|
|
226
|
+
|
|
227
|
+
## License
|
|
228
|
+
|
|
229
|
+
MIT License
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
lemony_lrc_parser/__init__.py,sha256=9Uql8Fgt1FDNs3pIAGcKxLe9vqZbpyLCDxnmRiuD-Kg,2208
|
|
2
|
+
lemony_lrc_parser/exceptions.py,sha256=hh9gPOkOYm7vSocAhirlswkx8rMU7CkIiLzEoxdxt4U,107
|
|
3
|
+
lemony_lrc_parser/models.py,sha256=LdC9RxNL_ynWuM1MgnRJKQgOzOFjI7KF_xQfSPk1Vuk,6323
|
|
4
|
+
lemony_lrc_parser/parser.py,sha256=ZJkiOCzx2eJQFK_D_5GBEVFFWoH5bFySXUZyFBZQ0s4,9877
|
|
5
|
+
lemony_lrc_parser/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
6
|
+
lemony_lrc_parser/regex.py,sha256=zGoNtcgVYN6RMb1cRs3dTZfr6I13hEnwEdsON9xvD1k,3918
|
|
7
|
+
lemony_lrc_parser/serializer.py,sha256=Q-W3WrjXTaeIBj20Lp0BlfxCxf0O53mydEd_j-rHWHs,7004
|
|
8
|
+
lemony_lrc_parser/timetag.py,sha256=rXmKl2NP9QNgxDvDGdvIBYK4I2HMysVq7U8O4nUs7A8,2772
|
|
9
|
+
lemony_lrc_parser-0.2.1.dist-info/METADATA,sha256=hOxSZyhJHDLQC5eciQdAKbgNweqbHiB0TgbMU4-be3o,6076
|
|
10
|
+
lemony_lrc_parser-0.2.1.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
|
|
11
|
+
lemony_lrc_parser-0.2.1.dist-info/licenses/LICENSE,sha256=Sn85iIuJalIesbHGlJTZ69jQMraRusqPCYNvmgJ1Rvo,1068
|
|
12
|
+
lemony_lrc_parser-0.2.1.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 LemonekoUwU
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|