trackparse 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- trackparse/__init__.py +119 -0
- trackparse/_artists.py +101 -0
- trackparse/_assemble.py +141 -0
- trackparse/_classify.py +468 -0
- trackparse/_context.py +74 -0
- trackparse/_credits.py +218 -0
- trackparse/_data.py +513 -0
- trackparse/_format.py +149 -0
- trackparse/_mode.py +204 -0
- trackparse/_normalize.py +107 -0
- trackparse/_parser.py +257 -0
- trackparse/_prefix.py +143 -0
- trackparse/_scanner.py +240 -0
- trackparse/_separator.py +209 -0
- trackparse/_tables.py +228 -0
- trackparse/_title.py +246 -0
- trackparse/_words.py +344 -0
- trackparse/options.py +62 -0
- trackparse/py.typed +0 -0
- trackparse/types.py +259 -0
- trackparse-0.2.0.dist-info/METADATA +389 -0
- trackparse-0.2.0.dist-info/RECORD +24 -0
- trackparse-0.2.0.dist-info/WHEEL +4 -0
- trackparse-0.2.0.dist-info/licenses/LICENSE +21 -0
trackparse/_prefix.py
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""R4: leading timestamp and track-position prefixes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import NamedTuple, Optional
|
|
6
|
+
|
|
7
|
+
from ._context import Ctx, match_known_at
|
|
8
|
+
from ._scanner import has_spaced_dash, scan
|
|
9
|
+
from ._words import char_at, is_digit
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class PrefixResult(NamedTuple):
|
|
13
|
+
rest: str
|
|
14
|
+
#: Absolute offset of ``rest`` (prefixes are ASCII, so code points == UTF-16 units).
|
|
15
|
+
base: int
|
|
16
|
+
timestamp: Optional[tuple[str, int]]
|
|
17
|
+
position: Optional[tuple[str, int]]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _digit_run(s: str, frm: int) -> int:
|
|
21
|
+
i = frm
|
|
22
|
+
while is_digit(char_at(s, i)):
|
|
23
|
+
i += 1
|
|
24
|
+
return i - frm
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _two_digits_up_to_59(s: str, at: int) -> Optional[int]:
|
|
28
|
+
if (
|
|
29
|
+
not is_digit(char_at(s, at))
|
|
30
|
+
or not is_digit(char_at(s, at + 1))
|
|
31
|
+
or is_digit(char_at(s, at + 2))
|
|
32
|
+
):
|
|
33
|
+
return None
|
|
34
|
+
n = int(s[at : at + 2])
|
|
35
|
+
return n if n <= 59 else None
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class _TimeMatch(NamedTuple):
|
|
39
|
+
raw: str
|
|
40
|
+
seconds: int
|
|
41
|
+
end: int
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _match_time(s: str, at: int) -> Optional[_TimeMatch]:
|
|
45
|
+
"""``H:MM:SS``, ``HH:MM:SS``, ``M:SS``, ``MM:SS`` starting at ``at``."""
|
|
46
|
+
lead = _digit_run(s, at)
|
|
47
|
+
if lead < 1 or lead > 2 or char_at(s, at + lead) != ":":
|
|
48
|
+
return None
|
|
49
|
+
first = int(s[at : at + lead])
|
|
50
|
+
mid = _two_digits_up_to_59(s, at + lead + 1)
|
|
51
|
+
if mid is None:
|
|
52
|
+
return None
|
|
53
|
+
after_mid = at + lead + 3
|
|
54
|
+
if char_at(s, after_mid) == ":":
|
|
55
|
+
sec = _two_digits_up_to_59(s, after_mid + 1)
|
|
56
|
+
if sec is not None:
|
|
57
|
+
end = after_mid + 3
|
|
58
|
+
return _TimeMatch(s[at:end], first * 3600 + mid * 60 + sec, end)
|
|
59
|
+
if lead == 2 and first > 59:
|
|
60
|
+
return None # MM is 00–59
|
|
61
|
+
return _TimeMatch(s[at:after_mid], first * 60 + mid, after_mid)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _skip_space_and_dash(s: str, at: int) -> int:
|
|
65
|
+
"""After a prefix: skip one space, then an optional ``- `` (spaced dash)."""
|
|
66
|
+
i = at
|
|
67
|
+
if char_at(s, i) == " ":
|
|
68
|
+
i += 1
|
|
69
|
+
if char_at(s, i) == "-" and char_at(s, i + 1) == " ":
|
|
70
|
+
i += 2
|
|
71
|
+
return i
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _match_timestamp(s: str) -> Optional[tuple[tuple[str, int], int]]:
|
|
75
|
+
"""R4.1"""
|
|
76
|
+
first = char_at(s, 0)
|
|
77
|
+
bracket = "]" if first == "[" else ")" if first == "(" else None
|
|
78
|
+
t = _match_time(s, 1 if bracket else 0)
|
|
79
|
+
if t is None:
|
|
80
|
+
return None
|
|
81
|
+
end = t.end
|
|
82
|
+
if bracket:
|
|
83
|
+
if char_at(s, end) != bracket:
|
|
84
|
+
return None
|
|
85
|
+
end += 1
|
|
86
|
+
if end < len(s) and s[end] != " ":
|
|
87
|
+
return None
|
|
88
|
+
return (t.raw, t.seconds), _skip_space_and_dash(s, end)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _contains_spaced_dash(s: str) -> bool:
|
|
92
|
+
return has_spaced_dash(scan(s).skel.text)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _match_position(s: str, mode: str) -> Optional[int]:
|
|
96
|
+
"""R4.2: end offset of the position prefix, or None. Forms (a), (b), (c), (d)/(e)."""
|
|
97
|
+
n = _digit_run(s, 0)
|
|
98
|
+
if n < 1 or n > 3:
|
|
99
|
+
return None
|
|
100
|
+
c = char_at(s, n)
|
|
101
|
+
zero_padded = s[0] == "0" and n >= 2
|
|
102
|
+
# (a) `01. `, `1) `
|
|
103
|
+
if (c == "." or c == ")") and (char_at(s, n + 1) == " " or n + 1 == len(s)):
|
|
104
|
+
return _skip_space_and_dash(s, n + 1)
|
|
105
|
+
# (b) `01 `: a position in every mode.
|
|
106
|
+
if zero_padded and c == " ":
|
|
107
|
+
return _skip_space_and_dash(s, n)
|
|
108
|
+
# (c) `3 - Artist - Title`; in filename mode `N - ` is always a position (`311 - Amber.mp3`).
|
|
109
|
+
if c == " " and char_at(s, n + 1) == "-" and char_at(s, n + 2) == " ":
|
|
110
|
+
if mode == "filename" or _contains_spaced_dash(s[n + 3 :]):
|
|
111
|
+
return n + 3
|
|
112
|
+
return None
|
|
113
|
+
if mode == "filename":
|
|
114
|
+
# (e) `01-Track`, `03_name`, `01.Track`
|
|
115
|
+
nxt = char_at(s, n + 1)
|
|
116
|
+
if (c == "-" or c == "_" or c == ".") and nxt is not None and nxt != " ":
|
|
117
|
+
return n + 1
|
|
118
|
+
# (e) `3 My Song.mp3`: only when the remainder has no spaced dash.
|
|
119
|
+
if c == " " and not _contains_spaced_dash(s[n + 1 :]):
|
|
120
|
+
return _skip_space_and_dash(s, n)
|
|
121
|
+
# (d) `2 Unlimited`, `50 Cent`: never a position outside filename mode.
|
|
122
|
+
return None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def strip_prefixes(s: str, base: int, mode: str, ctx: Ctx) -> PrefixResult:
|
|
126
|
+
rest = s
|
|
127
|
+
offset = base
|
|
128
|
+
timestamp: Optional[tuple[str, int]] = None
|
|
129
|
+
position: Optional[tuple[str, int]] = None
|
|
130
|
+
|
|
131
|
+
ts = _match_timestamp(rest)
|
|
132
|
+
if ts is not None:
|
|
133
|
+
timestamp = ts[0]
|
|
134
|
+
rest = rest[ts[1] :]
|
|
135
|
+
offset += ts[1]
|
|
136
|
+
if match_known_at(rest, 0, ctx, "-") == 0:
|
|
137
|
+
end = _match_position(rest, mode)
|
|
138
|
+
if end is not None and end < len(rest):
|
|
139
|
+
raw = rest[: _digit_run(rest, 0)]
|
|
140
|
+
position = (raw, int(raw))
|
|
141
|
+
rest = rest[end:]
|
|
142
|
+
offset += end
|
|
143
|
+
return PrefixResult(rest, offset, timestamp, position)
|
trackparse/_scanner.py
ADDED
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
"""R3: top-level group scanner and skeleton.
|
|
2
|
+
|
|
3
|
+
A ``Skel`` is a skeleton string plus, for every code point, its absolute offset in the
|
|
4
|
+
normalized input. Groups are looked up by the offset of their placeholder, so any slice of a
|
|
5
|
+
skeleton still knows which groups it contains. Offsets also give every extracted item (junk,
|
|
6
|
+
versions, credits) its position for the "order of appearance" rules.
|
|
7
|
+
|
|
8
|
+
Offsets are counted in UTF-16 units (an astral code point advances by 2) so that every
|
|
9
|
+
ordering comparison is identical to the JS reference; indices into ``text`` are code points.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from typing import NamedTuple, Optional
|
|
15
|
+
|
|
16
|
+
from ._words import char_at, collapse_spaces, is_whitespace
|
|
17
|
+
|
|
18
|
+
#: R3.4 placeholder code point.
|
|
19
|
+
PH = ""
|
|
20
|
+
|
|
21
|
+
_CLOSER = {"(": ")", "[": "]", "{": "}"}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Group:
|
|
25
|
+
__slots__ = ("closed", "end", "inner", "open", "pos", "raw")
|
|
26
|
+
|
|
27
|
+
def __init__(self, open: str, inner: str, raw: str, closed: bool, pos: int, end: int) -> None:
|
|
28
|
+
self.open = open
|
|
29
|
+
#: R3.5: inner text, trimmed and whitespace-collapsed.
|
|
30
|
+
self.inner = inner
|
|
31
|
+
#: The group exactly as written, brackets included.
|
|
32
|
+
self.raw = raw
|
|
33
|
+
self.closed = closed
|
|
34
|
+
#: Absolute offset of the opener.
|
|
35
|
+
self.pos = pos
|
|
36
|
+
#: Absolute offset just past the group.
|
|
37
|
+
self.end = end
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class Skel:
|
|
41
|
+
__slots__ = ("groups", "pos", "text")
|
|
42
|
+
|
|
43
|
+
def __init__(self, text: str, pos: list[int], groups: dict[int, Group]) -> None:
|
|
44
|
+
self.text = text
|
|
45
|
+
self.pos = pos
|
|
46
|
+
self.groups = groups
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class ScanResult(NamedTuple):
|
|
50
|
+
skel: Skel
|
|
51
|
+
unbalanced: bool
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def scan(s: str, base: int = 0) -> ScanResult:
|
|
55
|
+
"""R3.1–R3.5: scan ``s`` (whose first unit sits at absolute offset ``base``)."""
|
|
56
|
+
text: list[str] = []
|
|
57
|
+
pos: list[int] = []
|
|
58
|
+
groups: dict[int, Group] = {}
|
|
59
|
+
stack: list[str] = []
|
|
60
|
+
group_start = -1
|
|
61
|
+
group_off = 0
|
|
62
|
+
# Absolute UTF-16 offset of each code point of `s` (plus the end), for group bookkeeping.
|
|
63
|
+
offs: list[int] = []
|
|
64
|
+
off = base
|
|
65
|
+
for ch in s:
|
|
66
|
+
offs.append(off)
|
|
67
|
+
off += 2 if ch > "" else 1
|
|
68
|
+
offs.append(off)
|
|
69
|
+
for i, ch in enumerate(s):
|
|
70
|
+
if not stack:
|
|
71
|
+
if ch in _CLOSER:
|
|
72
|
+
stack.append(_CLOSER[ch])
|
|
73
|
+
group_start = i
|
|
74
|
+
group_off = offs[i]
|
|
75
|
+
else:
|
|
76
|
+
text.append(ch)
|
|
77
|
+
pos.append(offs[i])
|
|
78
|
+
continue
|
|
79
|
+
if ch in _CLOSER:
|
|
80
|
+
stack.append(_CLOSER[ch])
|
|
81
|
+
elif ch == stack[-1]:
|
|
82
|
+
stack.pop()
|
|
83
|
+
if not stack:
|
|
84
|
+
_add_group(s, group_start, i + 1, True, offs, groups)
|
|
85
|
+
text.append(PH)
|
|
86
|
+
pos.append(group_off)
|
|
87
|
+
# R3.2: a non-matching closer is literal text inside the group.
|
|
88
|
+
unbalanced = len(stack) > 0
|
|
89
|
+
if unbalanced:
|
|
90
|
+
# R3.3: the group runs to the end of the string.
|
|
91
|
+
_add_group(s, group_start, len(s), False, offs, groups)
|
|
92
|
+
text.append(PH)
|
|
93
|
+
pos.append(group_off)
|
|
94
|
+
return ScanResult(Skel("".join(text), pos, groups), unbalanced)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _add_group(
|
|
98
|
+
s: str, start: int, end: int, closed: bool, offs: list[int], groups: dict[int, Group]
|
|
99
|
+
) -> None:
|
|
100
|
+
inner = s[start + 1 : end - 1 if closed else end]
|
|
101
|
+
groups[offs[start]] = Group(
|
|
102
|
+
s[start], collapse_spaces(inner), s[start:end], closed, offs[start], offs[end]
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
# ---------------------------------------------------------------------------
|
|
107
|
+
# Skeleton operations
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def slice_skel(s: Skel, start: int, end: Optional[int] = None) -> Skel:
|
|
111
|
+
n = len(s.text)
|
|
112
|
+
if end is None:
|
|
113
|
+
end = n
|
|
114
|
+
a = max(0, min(start, n))
|
|
115
|
+
b = max(a, min(end, n))
|
|
116
|
+
return Skel(s.text[a:b], s.pos[a:b], s.groups)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def trim_skel(s: Skel) -> Skel:
|
|
120
|
+
t = s.text
|
|
121
|
+
a = 0
|
|
122
|
+
b = len(t)
|
|
123
|
+
while a < b and is_whitespace(t[a]):
|
|
124
|
+
a += 1
|
|
125
|
+
while b > a and is_whitespace(t[b - 1]):
|
|
126
|
+
b -= 1
|
|
127
|
+
return slice_skel(s, a, b)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def concat_skel(a: Skel, b: Skel) -> Skel:
|
|
131
|
+
return Skel(a.text + b.text, a.pos + b.pos, a.groups)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def cut_skel(s: Skel, start: int, end: int) -> Skel:
|
|
135
|
+
"""Remove the code points in [start, end)."""
|
|
136
|
+
return concat_skel(slice_skel(s, 0, start), slice_skel(s, end))
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def group_at(s: Skel, index: int) -> Optional[Group]:
|
|
140
|
+
"""The group behind the placeholder at ``index``, if it is one."""
|
|
141
|
+
if char_at(s.text, index) != PH:
|
|
142
|
+
return None
|
|
143
|
+
return s.groups.get(s.pos[index])
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def groups_of(s: Skel) -> list[tuple[int, Group]]:
|
|
147
|
+
"""Groups in the skeleton, left to right, with their index."""
|
|
148
|
+
out: list[tuple[int, Group]] = []
|
|
149
|
+
if PH not in s.text:
|
|
150
|
+
return out
|
|
151
|
+
for i, ch in enumerate(s.text):
|
|
152
|
+
if ch == PH:
|
|
153
|
+
g = s.groups.get(s.pos[i])
|
|
154
|
+
if g is not None:
|
|
155
|
+
out.append((i, g))
|
|
156
|
+
return out
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def has_group(s: Skel) -> bool:
|
|
160
|
+
return PH in s.text
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def render_skel(s: Skel) -> str:
|
|
164
|
+
"""Substitute groups back (R3.4 reassembly). Whitespace is not touched."""
|
|
165
|
+
if PH not in s.text:
|
|
166
|
+
return s.text
|
|
167
|
+
out: list[str] = []
|
|
168
|
+
for i, ch in enumerate(s.text):
|
|
169
|
+
g = s.groups.get(s.pos[i]) if ch == PH else None
|
|
170
|
+
out.append(g.raw if g is not None else ch)
|
|
171
|
+
return "".join(out)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def start_pos(s: Skel, fallback: int = 0) -> int:
|
|
175
|
+
"""Absolute offset of the first code point (or ``fallback`` for an empty skeleton)."""
|
|
176
|
+
return s.pos[0] if s.pos else fallback
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def remove_groups(s: Skel, remove: set[int]) -> Skel:
|
|
180
|
+
"""Remove the placeholders of the given groups."""
|
|
181
|
+
if not remove:
|
|
182
|
+
return s
|
|
183
|
+
text: list[str] = []
|
|
184
|
+
pos: list[int] = []
|
|
185
|
+
for ch, p in zip(s.text, s.pos):
|
|
186
|
+
if ch == PH and p in remove:
|
|
187
|
+
continue
|
|
188
|
+
text.append(ch)
|
|
189
|
+
pos.append(p)
|
|
190
|
+
return Skel("".join(text), pos, s.groups)
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def is_spaced_dash_at(text: str, i: int) -> bool:
|
|
194
|
+
"""R6.1: ``-`` with a space immediately on both sides."""
|
|
195
|
+
return 0 < i < len(text) - 1 and text[i] == "-" and text[i - 1] == " " and text[i + 1] == " "
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def has_spaced_dash(text: str) -> bool:
|
|
199
|
+
return " - " in text
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
class Segment:
|
|
203
|
+
__slots__ = ("end", "skel", "start")
|
|
204
|
+
|
|
205
|
+
def __init__(self, skel: Skel, start: int, end: int) -> None:
|
|
206
|
+
self.skel = skel
|
|
207
|
+
#: Trimmed range inside the parent skeleton.
|
|
208
|
+
self.start = start
|
|
209
|
+
self.end = end
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _trimmed_segment(s: Skel, start: int, end: int) -> Segment:
|
|
213
|
+
t = s.text
|
|
214
|
+
a = start
|
|
215
|
+
b = max(start, end)
|
|
216
|
+
while a < b and is_whitespace(char_at(t, a)):
|
|
217
|
+
a += 1
|
|
218
|
+
while b > a and is_whitespace(char_at(t, b - 1)):
|
|
219
|
+
b -= 1
|
|
220
|
+
return Segment(slice_skel(s, a, b), a, b)
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def split_at_spaced_dashes(s: Skel, allow_leading: bool = False) -> list[Segment]:
|
|
224
|
+
"""R6.1 / R8.2: split a skeleton at every spaced dash into trimmed segments. With
|
|
225
|
+
``allow_leading``, a string starting with ``- `` yields an empty first segment (R6.3)."""
|
|
226
|
+
segments: list[Segment] = []
|
|
227
|
+
t = s.text
|
|
228
|
+
start = 0
|
|
229
|
+
if allow_leading and t[:2] == "- ":
|
|
230
|
+
segments.append(Segment(slice_skel(s, 0, 0), 0, 0))
|
|
231
|
+
start = 2
|
|
232
|
+
i = max(start, 1)
|
|
233
|
+
n = len(t)
|
|
234
|
+
while i < n - 1:
|
|
235
|
+
if t[i] == "-" and t[i - 1] == " " and t[i + 1] == " ":
|
|
236
|
+
segments.append(_trimmed_segment(s, start, i - 1))
|
|
237
|
+
start = i + 2
|
|
238
|
+
i += 1
|
|
239
|
+
segments.append(_trimmed_segment(s, start, n))
|
|
240
|
+
return segments
|
trackparse/_separator.py
ADDED
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""R6: artist/title separator, including R6.2 suffix peeling (shared with R8.2)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import NamedTuple, Optional, Union
|
|
6
|
+
|
|
7
|
+
from ._classify import (
|
|
8
|
+
UNKNOWN,
|
|
9
|
+
Classification,
|
|
10
|
+
Flag,
|
|
11
|
+
Junk,
|
|
12
|
+
Site,
|
|
13
|
+
Versions,
|
|
14
|
+
Year,
|
|
15
|
+
classify,
|
|
16
|
+
)
|
|
17
|
+
from ._context import Ctx, is_known_artist, warn
|
|
18
|
+
from ._scanner import (
|
|
19
|
+
PH,
|
|
20
|
+
Segment,
|
|
21
|
+
Skel,
|
|
22
|
+
concat_skel,
|
|
23
|
+
has_group,
|
|
24
|
+
render_skel,
|
|
25
|
+
slice_skel,
|
|
26
|
+
split_at_spaced_dashes,
|
|
27
|
+
start_pos,
|
|
28
|
+
trim_skel,
|
|
29
|
+
)
|
|
30
|
+
from ._words import (
|
|
31
|
+
char_at,
|
|
32
|
+
collapse_spaces,
|
|
33
|
+
in_vocab,
|
|
34
|
+
is_letter,
|
|
35
|
+
is_whitespace,
|
|
36
|
+
split_words,
|
|
37
|
+
year_of,
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
Peelable = Union[Junk, Flag, Year, Versions]
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class Peeled(NamedTuple):
|
|
44
|
+
cls: Peelable
|
|
45
|
+
raw: str
|
|
46
|
+
pos: int
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _classify_segment(seg: Skel, ctx: Ctx) -> Classification:
|
|
50
|
+
"""R6.2 classification of a bare dash segment: placeholders make it Unknown."""
|
|
51
|
+
if has_group(seg):
|
|
52
|
+
return UNKNOWN
|
|
53
|
+
text = collapse_spaces(seg.text)
|
|
54
|
+
return classify(text, Site("title", "-", start_pos(seg)), ctx)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def peel_suffixes(
|
|
58
|
+
segments: list[Segment], relaxed: bool, ctx: Ctx
|
|
59
|
+
) -> tuple[list[Segment], list[Peeled]]:
|
|
60
|
+
"""R6.2 / R8.2: peel classifiable suffixes off the right end. ``relaxed`` drops conditions
|
|
61
|
+
(i)–(iv) (R8.2 with an artist side). Returns the remaining segments and the peeled
|
|
62
|
+
suffixes, left to right."""
|
|
63
|
+
remaining = list(segments)
|
|
64
|
+
peeled: list[Peeled] = []
|
|
65
|
+
while len(remaining) >= 2:
|
|
66
|
+
last = remaining[-1]
|
|
67
|
+
cls = _classify_segment(last.skel, ctx)
|
|
68
|
+
if not isinstance(cls, (Junk, Flag, Year, Versions)):
|
|
69
|
+
break
|
|
70
|
+
words = split_words(last.skel.text)
|
|
71
|
+
# (ii) does not apply to prefix-form versions (`Andy C b2b Hedex - Live at Printworks`).
|
|
72
|
+
prefix_form = isinstance(cls, Versions) and all(v.prefix_form for v in cls.versions)
|
|
73
|
+
ok = (
|
|
74
|
+
relaxed
|
|
75
|
+
or len(remaining) - 1 >= 2
|
|
76
|
+
or (len(words) >= 2 and not prefix_form)
|
|
77
|
+
or any(year_of(w) is not None for w in words)
|
|
78
|
+
or isinstance(cls, Junk)
|
|
79
|
+
)
|
|
80
|
+
if not ok:
|
|
81
|
+
break
|
|
82
|
+
peeled.insert(0, Peeled(cls, collapse_spaces(last.skel.text), start_pos(last.skel)))
|
|
83
|
+
remaining.pop()
|
|
84
|
+
return remaining, peeled
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class SeparatorResult(NamedTuple):
|
|
88
|
+
artist: Optional[Skel]
|
|
89
|
+
title: Skel
|
|
90
|
+
#: A separator split the string (R6.3–R6.7): R8.2 peels without conditions.
|
|
91
|
+
split: bool
|
|
92
|
+
peeled: list[Peeled]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _non_empty(s: Skel) -> Optional[Skel]:
|
|
96
|
+
t = trim_skel(s)
|
|
97
|
+
return t if len(t.text) > 0 else None
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _split_at(s: Skel, cut_start: int, cut_end: int) -> tuple[Optional[Skel], Skel]:
|
|
101
|
+
return _non_empty(slice_skel(s, 0, cut_start)), trim_skel(slice_skel(s, cut_end))
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def find_separator(skel: Skel, mode: str, uploader: Optional[Skel], ctx: Ctx) -> SeparatorResult:
|
|
105
|
+
segments = split_at_spaced_dashes(skel, True)
|
|
106
|
+
remaining, peeled = peel_suffixes(segments, False, ctx)
|
|
107
|
+
|
|
108
|
+
# R6.3
|
|
109
|
+
if len(remaining) >= 2:
|
|
110
|
+
first = remaining[0]
|
|
111
|
+
second = remaining[1]
|
|
112
|
+
last_seg = remaining[-1]
|
|
113
|
+
artist = first.skel if len(first.skel.text) > 0 else None
|
|
114
|
+
title = slice_skel(skel, second.start, last_seg.end)
|
|
115
|
+
return SeparatorResult(artist, title, artist is not None, peeled)
|
|
116
|
+
|
|
117
|
+
rest = remaining[0].skel
|
|
118
|
+
other = _split_without_spaced_dash(rest, mode, ctx)
|
|
119
|
+
if other is not None:
|
|
120
|
+
return SeparatorResult(other[0], other[1], other[0] is not None, peeled)
|
|
121
|
+
|
|
122
|
+
# R6.8: the uploader fallback does not warn noSeparator.
|
|
123
|
+
artist = uploader if mode == "youtube" else None
|
|
124
|
+
if len(rest.text) > 0 and artist is None:
|
|
125
|
+
warn(ctx, "noSeparator")
|
|
126
|
+
return SeparatorResult(artist, rest, False, peeled)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _split_without_spaced_dash(
|
|
130
|
+
s: Skel, mode: str, ctx: Ctx
|
|
131
|
+
) -> Optional[tuple[Optional[Skel], Skel]]:
|
|
132
|
+
asym = _asymmetric_dash(s.text)
|
|
133
|
+
if asym >= 0:
|
|
134
|
+
warn(ctx, "asymmetricDashSplit")
|
|
135
|
+
return _split_at(s, asym, asym + 1)
|
|
136
|
+
if mode == "youtube":
|
|
137
|
+
quoted = _quoted_title(s)
|
|
138
|
+
if quoted is not None:
|
|
139
|
+
warn(ctx, "quotedTitleSplit")
|
|
140
|
+
return quoted
|
|
141
|
+
by = _by_split(s, ctx)
|
|
142
|
+
if by >= 0:
|
|
143
|
+
warn(ctx, "bySplit")
|
|
144
|
+
left, right = _split_at(s, by, by + 4)
|
|
145
|
+
return (right if len(right.text) > 0 else None, left if left is not None else s)
|
|
146
|
+
if mode == "youtube" or mode == "filename":
|
|
147
|
+
dash = _unspaced_dash(s, ctx)
|
|
148
|
+
if dash >= 0:
|
|
149
|
+
warn(ctx, "unspacedDashSplit")
|
|
150
|
+
return _split_at(s, dash, dash + 1)
|
|
151
|
+
return None
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _asymmetric_dash(t: str) -> int:
|
|
155
|
+
"""R6.4: the first ``-`` with a space on exactly one side."""
|
|
156
|
+
for i in range(1, len(t) - 1):
|
|
157
|
+
if t[i] != "-":
|
|
158
|
+
continue
|
|
159
|
+
if is_whitespace(t[i - 1]) != is_whitespace(t[i + 1]):
|
|
160
|
+
return i
|
|
161
|
+
return -1
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _quoted_title(s: Skel) -> Optional[tuple[Optional[Skel], Skel]]:
|
|
165
|
+
"""R6.5: ``Artist "Title" (extra)``."""
|
|
166
|
+
t = s.text
|
|
167
|
+
open_ = t.find('"')
|
|
168
|
+
if open_ <= 0:
|
|
169
|
+
return None
|
|
170
|
+
close = t.find('"', open_ + 1)
|
|
171
|
+
if close < 0:
|
|
172
|
+
return None
|
|
173
|
+
artist = trim_skel(slice_skel(s, 0, open_))
|
|
174
|
+
last = artist.text[-1:]
|
|
175
|
+
if last == ":" or last == "-":
|
|
176
|
+
artist = trim_skel(slice_skel(artist, 0, len(artist.text) - 1))
|
|
177
|
+
title = trim_skel(concat_skel(slice_skel(s, open_ + 1, close), slice_skel(s, close + 1)))
|
|
178
|
+
return (artist if len(artist.text) > 0 else None, title)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _by_split(s: Skel, ctx: Ctx) -> int:
|
|
182
|
+
"""R6.6: index of the space before the last usable `` by ``. The right side must be
|
|
183
|
+
non-empty and not a single stopword (``Stand by Me``)."""
|
|
184
|
+
t = s.text
|
|
185
|
+
i = t.rfind(" by ")
|
|
186
|
+
while i > 0:
|
|
187
|
+
# Placeholder-only words (groups such as `(Official Video)`) don't count here.
|
|
188
|
+
right = [w for w in split_words(t[i + 4 :]) if w != PH]
|
|
189
|
+
if right and not (len(right) == 1 and in_vocab(right[0], ctx.t.stopwords)):
|
|
190
|
+
return i
|
|
191
|
+
# JS lastIndexOf(" by ", i - 1): the last occurrence starting at or before i - 1.
|
|
192
|
+
i = t.rfind(" by ", 0, i - 1 + 4) if i - 1 >= 0 else -1
|
|
193
|
+
return -1
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _unspaced_dash(s: Skel, ctx: Ctx) -> int:
|
|
197
|
+
"""R6.7: ``Jay-Z-Numb`` → the chosen unspaced dash index."""
|
|
198
|
+
t = s.text
|
|
199
|
+
last_candidate = -1
|
|
200
|
+
for i in range(1, len(t) - 1):
|
|
201
|
+
if t[i] != "-" or is_whitespace(t[i - 1]) or is_whitespace(t[i + 1]):
|
|
202
|
+
continue
|
|
203
|
+
nxt = char_at(t, i + 1)
|
|
204
|
+
if not (nxt == '"' or is_letter(nxt)):
|
|
205
|
+
continue
|
|
206
|
+
if ctx.known_keys and is_known_artist(render_skel(slice_skel(s, 0, i)), ctx):
|
|
207
|
+
return i
|
|
208
|
+
last_candidate = i
|
|
209
|
+
return last_candidate
|