trackparse 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- trackparse/__init__.py +119 -0
- trackparse/_artists.py +101 -0
- trackparse/_assemble.py +141 -0
- trackparse/_classify.py +468 -0
- trackparse/_context.py +74 -0
- trackparse/_credits.py +218 -0
- trackparse/_data.py +513 -0
- trackparse/_format.py +149 -0
- trackparse/_mode.py +204 -0
- trackparse/_normalize.py +107 -0
- trackparse/_parser.py +257 -0
- trackparse/_prefix.py +143 -0
- trackparse/_scanner.py +240 -0
- trackparse/_separator.py +209 -0
- trackparse/_tables.py +228 -0
- trackparse/_title.py +246 -0
- trackparse/_words.py +344 -0
- trackparse/options.py +62 -0
- trackparse/py.typed +0 -0
- trackparse/types.py +259 -0
- trackparse-0.2.0.dist-info/METADATA +389 -0
- trackparse-0.2.0.dist-info/RECORD +24 -0
- trackparse-0.2.0.dist-info/WHEEL +4 -0
- trackparse-0.2.0.dist-info/licenses/LICENSE +21 -0
trackparse/_words.py
ADDED
|
@@ -0,0 +1,344 @@
|
|
|
1
|
+
"""Low-level text helpers shared by every rule: the R0.2 whitespace set, R0.3 word keys,
|
|
2
|
+
R0.4 dedup keys, R0.6 letters/digits and token-sequence (phrase) matching.
|
|
3
|
+
|
|
4
|
+
Python strings are sequences of code points, so indices here are code-point indices (R0.5).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import unicodedata
|
|
10
|
+
from collections.abc import Iterable, Mapping
|
|
11
|
+
from collections.abc import Set as AbstractSet
|
|
12
|
+
from typing import (
|
|
13
|
+
Generic,
|
|
14
|
+
NamedTuple,
|
|
15
|
+
Optional,
|
|
16
|
+
TypeVar,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
V = TypeVar("V")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def is_whitespace_code(cp: int) -> bool:
|
|
23
|
+
"""R0.2: the exact whitespace code-point set. Never use ``str.isspace`` / ``\\s``."""
|
|
24
|
+
return (
|
|
25
|
+
(0x09 <= cp <= 0x0D)
|
|
26
|
+
or cp == 0x20
|
|
27
|
+
or cp == 0x85
|
|
28
|
+
or cp == 0xA0
|
|
29
|
+
or cp == 0x1680
|
|
30
|
+
or (0x2000 <= cp <= 0x200A)
|
|
31
|
+
or cp == 0x2028
|
|
32
|
+
or cp == 0x2029
|
|
33
|
+
or cp == 0x202F
|
|
34
|
+
or cp == 0x205F
|
|
35
|
+
or cp == 0x3000
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
_WS = frozenset(
|
|
40
|
+
chr(c)
|
|
41
|
+
for c in [*range(9, 14), 32, 133, 160, 5760, *range(8192, 8203), 8232, 8233, 8239, 8287, 12288]
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def is_whitespace(ch: Optional[str]) -> bool:
|
|
46
|
+
"""R0.2 test on one code point (``None`` / empty → False)."""
|
|
47
|
+
return ch is not None and ch in _WS
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def char_at(s: str, i: int) -> Optional[str]:
|
|
51
|
+
"""``s[i]`` or ``None`` when out of range (JS ``s[i]`` → ``undefined``; never negative)."""
|
|
52
|
+
return s[i] if 0 <= i < len(s) else None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def r0_trim(s: str) -> str:
|
|
56
|
+
"""Trim R0.2 whitespace from both ends."""
|
|
57
|
+
a = 0
|
|
58
|
+
b = len(s)
|
|
59
|
+
while a < b and s[a] in _WS:
|
|
60
|
+
a += 1
|
|
61
|
+
while b > a and s[b - 1] in _WS:
|
|
62
|
+
b -= 1
|
|
63
|
+
return s[a:b]
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def r0_trim_end(s: str) -> str:
|
|
67
|
+
b = len(s)
|
|
68
|
+
while b > 0 and s[b - 1] in _WS:
|
|
69
|
+
b -= 1
|
|
70
|
+
return s[:b]
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
# JS String.prototype.trim(): WhiteSpace (incl. U+FEFF and Zs) + LineTerminator. Used only where
|
|
74
|
+
# the JS reference trims raw, unnormalized option strings (keyword keys), to stay byte-identical.
|
|
75
|
+
_JS_TRIM = frozenset(
|
|
76
|
+
chr(c)
|
|
77
|
+
for c in [
|
|
78
|
+
9,
|
|
79
|
+
10,
|
|
80
|
+
11,
|
|
81
|
+
12,
|
|
82
|
+
13,
|
|
83
|
+
32,
|
|
84
|
+
160,
|
|
85
|
+
5760,
|
|
86
|
+
*range(8192, 8203),
|
|
87
|
+
8232,
|
|
88
|
+
8233,
|
|
89
|
+
8239,
|
|
90
|
+
8287,
|
|
91
|
+
12288,
|
|
92
|
+
65279,
|
|
93
|
+
]
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def js_trim(s: str) -> str:
|
|
98
|
+
a = 0
|
|
99
|
+
b = len(s)
|
|
100
|
+
while a < b and s[a] in _JS_TRIM:
|
|
101
|
+
a += 1
|
|
102
|
+
while b > a and s[b - 1] in _JS_TRIM:
|
|
103
|
+
b -= 1
|
|
104
|
+
return s[a:b]
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def is_digit(ch: Optional[str]) -> bool:
|
|
108
|
+
"""R0.6: ASCII digits only."""
|
|
109
|
+
return ch is not None and len(ch) == 1 and "0" <= ch <= "9"
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def is_all_digits(s: str) -> bool:
|
|
113
|
+
if len(s) == 0:
|
|
114
|
+
return False
|
|
115
|
+
return all("0" <= c <= "9" for c in s)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def is_letter(ch: Optional[str]) -> bool:
|
|
119
|
+
"""R0.6: Unicode general category L (``str.isalpha`` on one code point)."""
|
|
120
|
+
return ch is not None and len(ch) == 1 and ch.isalpha()
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
_ASCII_LOWER = {c: c + 32 for c in range(65, 91)}
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def ascii_lower(s: str) -> str:
|
|
127
|
+
"""ASCII-only lowercase (A–Z → a–z). Never ``str.lower()`` for word keys."""
|
|
128
|
+
return s.translate(_ASCII_LOWER)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
_KEY_STRIPPABLE = ".,!?:;"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def strip_key_punct(lower: str) -> str:
|
|
135
|
+
"""R0.3: the lowercased word with one trailing ``. , ! ? : ;`` removed."""
|
|
136
|
+
return lower[:-1] if lower and lower[-1] in _KEY_STRIPPABLE else lower
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def word_key(word: str) -> str:
|
|
140
|
+
"""R0.3 word key (stripped form)."""
|
|
141
|
+
return strip_key_punct(ascii_lower(word))
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def word_matches(word: str, token: str) -> bool:
|
|
145
|
+
"""R0.3: the unstripped lowercase form is compared first (entries like ``feat.``)."""
|
|
146
|
+
lower = ascii_lower(word)
|
|
147
|
+
return lower == token or strip_key_punct(lower) == token
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def in_vocab(word: str, vocab: AbstractSet[str]) -> bool:
|
|
151
|
+
lower = ascii_lower(word)
|
|
152
|
+
return lower in vocab or strip_key_punct(lower) in vocab
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def lookup_vocab(word: str, vocab: Mapping[str, V]) -> Optional[V]:
|
|
156
|
+
lower = ascii_lower(word)
|
|
157
|
+
v = vocab.get(lower)
|
|
158
|
+
return v if v is not None else vocab.get(strip_key_punct(lower))
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def collapse_spaces(s: str) -> str:
|
|
162
|
+
"""Collapse runs of R0.2 whitespace to one U+0020 and trim."""
|
|
163
|
+
out: list[str] = []
|
|
164
|
+
pending = False
|
|
165
|
+
for ch in s:
|
|
166
|
+
if ch in _WS:
|
|
167
|
+
pending = len(out) > 0
|
|
168
|
+
else:
|
|
169
|
+
if pending:
|
|
170
|
+
out.append(" ")
|
|
171
|
+
pending = False
|
|
172
|
+
out.append(ch)
|
|
173
|
+
return "".join(out)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def split_words(s: str) -> list[str]:
|
|
177
|
+
"""R0.7: maximal runs of non-whitespace code points."""
|
|
178
|
+
words: list[str] = []
|
|
179
|
+
start = -1
|
|
180
|
+
for i, ch in enumerate(s):
|
|
181
|
+
if ch in _WS:
|
|
182
|
+
if start >= 0:
|
|
183
|
+
words.append(s[start:i])
|
|
184
|
+
start = -1
|
|
185
|
+
elif start < 0:
|
|
186
|
+
start = i
|
|
187
|
+
if start >= 0:
|
|
188
|
+
words.append(s[start:])
|
|
189
|
+
return words
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
class WordSpan(NamedTuple):
|
|
193
|
+
"""A word with its code-point span inside the string it was taken from."""
|
|
194
|
+
|
|
195
|
+
text: str
|
|
196
|
+
start: int
|
|
197
|
+
end: int
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def word_spans(s: str) -> list[WordSpan]:
|
|
201
|
+
spans: list[WordSpan] = []
|
|
202
|
+
start = -1
|
|
203
|
+
for i, ch in enumerate(s):
|
|
204
|
+
if ch in _WS:
|
|
205
|
+
if start >= 0:
|
|
206
|
+
spans.append(WordSpan(s[start:i], start, i))
|
|
207
|
+
start = -1
|
|
208
|
+
elif start < 0:
|
|
209
|
+
start = i
|
|
210
|
+
if start >= 0:
|
|
211
|
+
spans.append(WordSpan(s[start:], start, len(s)))
|
|
212
|
+
return spans
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def dedup_key(s: str) -> str:
|
|
216
|
+
"""R0.4 dedup key: full lowercase → NFD → drop U+0300–U+036F → collapse/trim."""
|
|
217
|
+
decomposed = unicodedata.normalize("NFD", s.lower())
|
|
218
|
+
return collapse_spaces("".join(ch for ch in decomposed if not ("̀" <= ch <= "ͯ")))
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def year_of(word: str) -> Optional[int]:
|
|
222
|
+
"""Year word: 4 ASCII digits in 1900–2099 (word key, so ``2015,`` counts)."""
|
|
223
|
+
key = word_key(word)
|
|
224
|
+
if len(key) != 4 or not is_all_digits(key):
|
|
225
|
+
return None
|
|
226
|
+
n = int(key)
|
|
227
|
+
return n if 1900 <= n <= 2099 else None
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def utf16_len(s: str) -> int:
|
|
231
|
+
"""Length in UTF-16 code units. Positions are kept in UTF-16 units so that every
|
|
232
|
+
order-of-appearance comparison (R9) matches the JS reference exactly."""
|
|
233
|
+
n = len(s)
|
|
234
|
+
for ch in s:
|
|
235
|
+
if ch > "":
|
|
236
|
+
n += 1
|
|
237
|
+
return n
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
# ---------------------------------------------------------------------------
|
|
241
|
+
# Phrase tables: multi-word vocabulary matched as token sequences, longest first.
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
class PhraseEntry(Generic[V]):
|
|
245
|
+
__slots__ = ("key", "tokens", "value")
|
|
246
|
+
|
|
247
|
+
def __init__(self, key: str, tokens: list[str], value: V) -> None:
|
|
248
|
+
self.key = key
|
|
249
|
+
self.tokens = tokens
|
|
250
|
+
self.value = value
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
class PhraseMatch(Generic[V]):
|
|
254
|
+
__slots__ = ("entry", "length")
|
|
255
|
+
|
|
256
|
+
def __init__(self, entry: PhraseEntry[V], length: int) -> None:
|
|
257
|
+
self.entry = entry
|
|
258
|
+
#: Number of words consumed.
|
|
259
|
+
self.length = length
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
class PhraseTable(Generic[V]):
|
|
263
|
+
def __init__(self, entries: Iterable[tuple[str, V]]) -> None:
|
|
264
|
+
self._by_first: dict[str, list[PhraseEntry[V]]] = {}
|
|
265
|
+
self._by_last: dict[str, list[PhraseEntry[V]]] = {}
|
|
266
|
+
self._keys: dict[str, V] = {}
|
|
267
|
+
for raw_key, value in entries:
|
|
268
|
+
self.add(raw_key, value)
|
|
269
|
+
|
|
270
|
+
def add(self, raw_key: str, value: V) -> None:
|
|
271
|
+
key = js_trim(ascii_lower(raw_key))
|
|
272
|
+
tokens = split_words(key)
|
|
273
|
+
if not tokens or key in self._keys:
|
|
274
|
+
return
|
|
275
|
+
self._keys[key] = value
|
|
276
|
+
entry = PhraseEntry(key, tokens, value)
|
|
277
|
+
_insert_sorted(self._by_first, tokens[0], entry)
|
|
278
|
+
_insert_sorted(self._by_last, tokens[-1], entry)
|
|
279
|
+
|
|
280
|
+
def has(self, key: str) -> bool:
|
|
281
|
+
return key in self._keys
|
|
282
|
+
|
|
283
|
+
def match_at(self, words: list[str], start: int) -> Optional[PhraseMatch[V]]:
|
|
284
|
+
"""Longest entry matching ``words[start …]``."""
|
|
285
|
+
if not 0 <= start < len(words):
|
|
286
|
+
return None
|
|
287
|
+
best: Optional[PhraseMatch[V]] = None
|
|
288
|
+
for entry in _candidates(self._by_first, words[start]):
|
|
289
|
+
n = len(entry.tokens)
|
|
290
|
+
if best is not None and n <= best.length:
|
|
291
|
+
continue
|
|
292
|
+
if start + n > len(words):
|
|
293
|
+
continue
|
|
294
|
+
if _tokens_match(words, start, entry.tokens):
|
|
295
|
+
best = PhraseMatch(entry, n)
|
|
296
|
+
return best
|
|
297
|
+
|
|
298
|
+
def match_ending(self, words: list[str], end: int) -> Optional[PhraseMatch[V]]:
|
|
299
|
+
"""Longest entry matching ``words[… end-1]`` (end exclusive)."""
|
|
300
|
+
if not 0 <= end - 1 < len(words):
|
|
301
|
+
return None
|
|
302
|
+
best: Optional[PhraseMatch[V]] = None
|
|
303
|
+
for entry in _candidates(self._by_last, words[end - 1]):
|
|
304
|
+
n = len(entry.tokens)
|
|
305
|
+
if best is not None and n <= best.length:
|
|
306
|
+
continue
|
|
307
|
+
if end - n < 0:
|
|
308
|
+
continue
|
|
309
|
+
if _tokens_match(words, end - n, entry.tokens):
|
|
310
|
+
best = PhraseMatch(entry, n)
|
|
311
|
+
return best
|
|
312
|
+
|
|
313
|
+
def match_whole(self, words: list[str]) -> Optional[PhraseMatch[V]]:
|
|
314
|
+
"""The whole word list is exactly one entry."""
|
|
315
|
+
m = self.match_at(words, 0)
|
|
316
|
+
return m if m is not None and m.length == len(words) else None
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _insert_sorted(m: dict[str, list[PhraseEntry[V]]], k: str, e: PhraseEntry[V]) -> None:
|
|
320
|
+
lst = m.setdefault(k, [])
|
|
321
|
+
lst.append(e)
|
|
322
|
+
lst.sort(key=lambda x: -len(x.tokens)) # stable, like the JS sort
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _candidates(m: dict[str, list[PhraseEntry[V]]], word: str) -> list[PhraseEntry[V]]:
|
|
326
|
+
lower = ascii_lower(word)
|
|
327
|
+
stripped = strip_key_punct(lower)
|
|
328
|
+
a = m.get(lower, [])
|
|
329
|
+
if stripped == lower:
|
|
330
|
+
return a
|
|
331
|
+
b = m.get(stripped, [])
|
|
332
|
+
if not a:
|
|
333
|
+
return b
|
|
334
|
+
if not b:
|
|
335
|
+
return a
|
|
336
|
+
return a + b
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def _tokens_match(words: list[str], start: int, tokens: list[str]) -> bool:
|
|
340
|
+
for k, t in enumerate(tokens):
|
|
341
|
+
i = start + k
|
|
342
|
+
if i >= len(words) or not word_matches(words[i], t):
|
|
343
|
+
return False
|
|
344
|
+
return True
|
trackparse/options.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
"""Option types. They mirror spec/schema/options.schema.json with snake_case keys.
|
|
2
|
+
|
|
3
|
+
Fixture JSON and the JS API use camelCase (``knownArtists``); the Python API takes keyword
|
|
4
|
+
arguments (``known_artists=...``) and a ``keywords`` dict with snake_case keys.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from collections.abc import Sequence
|
|
10
|
+
from typing import Literal, TypedDict
|
|
11
|
+
|
|
12
|
+
from .types import JunkKind, ModeOption, VersionType
|
|
13
|
+
|
|
14
|
+
SplitAnd = Literal["auto", "always", "never"]
|
|
15
|
+
FeatPlacement = Literal["source", "artist", "title", "omit"]
|
|
16
|
+
JoinerStyle = Literal["original", "canonical"]
|
|
17
|
+
VersionsStyle = Literal["all", "none"]
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"FeatPlacement",
|
|
21
|
+
"FormatOptions",
|
|
22
|
+
"JoinerStyle",
|
|
23
|
+
"KeywordOptions",
|
|
24
|
+
"ParseOptions",
|
|
25
|
+
"SplitAnd",
|
|
26
|
+
"VersionsStyle",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class KeywordOptions(TypedDict, total=False):
|
|
31
|
+
"""Additive vocabulary (R12). Keys are lowercased with ASCII rules."""
|
|
32
|
+
|
|
33
|
+
#: Extra version heads: key -> version type.
|
|
34
|
+
version_heads: dict[str, VersionType]
|
|
35
|
+
descriptors: list[str]
|
|
36
|
+
genres: list[str]
|
|
37
|
+
#: Extra junk phrases: phrase -> kind.
|
|
38
|
+
junk: dict[str, JunkKind]
|
|
39
|
+
feat_markers: list[str]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class ParseOptions(TypedDict, total=False):
|
|
43
|
+
"""Keyword arguments accepted by ``parse``, ``parse_artists`` and ``create_parser``."""
|
|
44
|
+
|
|
45
|
+
mode: ModeOption
|
|
46
|
+
#: youtube mode: channel name used as the artist when no separator is found (R6.8).
|
|
47
|
+
uploader: str
|
|
48
|
+
#: Names that must never be split (R7.3a).
|
|
49
|
+
known_artists: Sequence[str]
|
|
50
|
+
split_and: SplitAnd
|
|
51
|
+
keywords: KeywordOptions
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class FormatOptions(TypedDict, total=False):
|
|
55
|
+
"""Keyword arguments accepted by ``format_track`` (R11)."""
|
|
56
|
+
|
|
57
|
+
feat: FeatPlacement
|
|
58
|
+
feat_marker: str
|
|
59
|
+
joiners: JoinerStyle
|
|
60
|
+
versions: VersionsStyle
|
|
61
|
+
producers: bool
|
|
62
|
+
position: bool
|
trackparse/py.typed
ADDED
|
File without changes
|
trackparse/types.py
ADDED
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
"""Public result types. They mirror spec/schema/parsed-track.schema.json.
|
|
2
|
+
|
|
3
|
+
Every class is a frozen dataclass with snake_case attributes. Sequences are stored as tuples so
|
|
4
|
+
results are deeply immutable and hashable. ``to_dict()`` returns the spec's canonical JSON form
|
|
5
|
+
(camelCase keys, lists, key order as in the schema), identical to the JS reference output;
|
|
6
|
+
``from_dict()`` builds an instance back from that form.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from typing import Any, Literal, Optional, cast
|
|
13
|
+
|
|
14
|
+
Mode = Literal["clean", "youtube", "filename"]
|
|
15
|
+
ModeOption = Literal["auto", "clean", "youtube", "filename"]
|
|
16
|
+
ArtistRole = Literal["primary", "featured", "remixer", "producer"]
|
|
17
|
+
ArtistSource = Literal["artist", "title", "version"]
|
|
18
|
+
VersionType = Literal[
|
|
19
|
+
"remix",
|
|
20
|
+
"bootleg",
|
|
21
|
+
"vip",
|
|
22
|
+
"edit",
|
|
23
|
+
"flip",
|
|
24
|
+
"refix",
|
|
25
|
+
"rework",
|
|
26
|
+
"mashup",
|
|
27
|
+
"blend",
|
|
28
|
+
"dub",
|
|
29
|
+
"mix",
|
|
30
|
+
"extended",
|
|
31
|
+
"radio",
|
|
32
|
+
"club",
|
|
33
|
+
"original",
|
|
34
|
+
"instrumental",
|
|
35
|
+
"acapella",
|
|
36
|
+
"live",
|
|
37
|
+
"acoustic",
|
|
38
|
+
"remaster",
|
|
39
|
+
"demo",
|
|
40
|
+
"reprise",
|
|
41
|
+
"cover",
|
|
42
|
+
"version",
|
|
43
|
+
"spedUp",
|
|
44
|
+
"slowed",
|
|
45
|
+
"nightcore",
|
|
46
|
+
]
|
|
47
|
+
VersionDelimiter = Literal["(", "[", "{", "-"]
|
|
48
|
+
JunkKind = Literal[
|
|
49
|
+
"video", "audio", "lyrics", "quality", "promo", "platform", "label", "genre", "other"
|
|
50
|
+
]
|
|
51
|
+
Warning = Literal[
|
|
52
|
+
"noSeparator",
|
|
53
|
+
"ambiguousSeparator",
|
|
54
|
+
"unspacedDashSplit",
|
|
55
|
+
"asymmetricDashSplit",
|
|
56
|
+
"bySplit",
|
|
57
|
+
"quotedTitleSplit",
|
|
58
|
+
"ambiguousMixCredit",
|
|
59
|
+
"unbalancedBrackets",
|
|
60
|
+
]
|
|
61
|
+
|
|
62
|
+
__all__ = [
|
|
63
|
+
"Artist",
|
|
64
|
+
"ArtistRole",
|
|
65
|
+
"ArtistSource",
|
|
66
|
+
"Flags",
|
|
67
|
+
"Junk",
|
|
68
|
+
"JunkKind",
|
|
69
|
+
"Mode",
|
|
70
|
+
"ModeOption",
|
|
71
|
+
"ParsedTrack",
|
|
72
|
+
"Position",
|
|
73
|
+
"Timestamp",
|
|
74
|
+
"Version",
|
|
75
|
+
"VersionDelimiter",
|
|
76
|
+
"VersionType",
|
|
77
|
+
"Warning",
|
|
78
|
+
]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass(frozen=True)
|
|
82
|
+
class Artist:
|
|
83
|
+
"""One credited name."""
|
|
84
|
+
|
|
85
|
+
name: str
|
|
86
|
+
role: ArtistRole
|
|
87
|
+
#: Raw joiner / marker text that preceded the name in its credit list; ``None`` for the first.
|
|
88
|
+
joiner: Optional[str]
|
|
89
|
+
source: ArtistSource
|
|
90
|
+
|
|
91
|
+
def to_dict(self) -> dict[str, Any]:
|
|
92
|
+
return {"name": self.name, "role": self.role, "joiner": self.joiner, "source": self.source}
|
|
93
|
+
|
|
94
|
+
@classmethod
|
|
95
|
+
def from_dict(cls, d: dict[str, Any]) -> Artist:
|
|
96
|
+
return cls(name=d["name"], role=d["role"], joiner=d["joiner"], source=d["source"])
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
@dataclass(frozen=True)
|
|
100
|
+
class Version:
|
|
101
|
+
"""A recognised version/remix group or dash suffix."""
|
|
102
|
+
|
|
103
|
+
type: VersionType
|
|
104
|
+
raw: str
|
|
105
|
+
artists: tuple[Artist, ...]
|
|
106
|
+
modifiers: tuple[str, ...]
|
|
107
|
+
descriptor: Optional[str]
|
|
108
|
+
year: Optional[int]
|
|
109
|
+
unknown_artist: bool
|
|
110
|
+
delimiter: VersionDelimiter
|
|
111
|
+
|
|
112
|
+
def to_dict(self) -> dict[str, Any]:
|
|
113
|
+
return {
|
|
114
|
+
"type": self.type,
|
|
115
|
+
"raw": self.raw,
|
|
116
|
+
"artists": [a.to_dict() for a in self.artists],
|
|
117
|
+
"modifiers": list(self.modifiers),
|
|
118
|
+
"descriptor": self.descriptor,
|
|
119
|
+
"year": self.year,
|
|
120
|
+
"unknownArtist": self.unknown_artist,
|
|
121
|
+
"delimiter": self.delimiter,
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
@classmethod
|
|
125
|
+
def from_dict(cls, d: dict[str, Any]) -> Version:
|
|
126
|
+
return cls(
|
|
127
|
+
type=d["type"],
|
|
128
|
+
raw=d["raw"],
|
|
129
|
+
artists=tuple(Artist.from_dict(a) for a in d["artists"]),
|
|
130
|
+
modifiers=tuple(d["modifiers"]),
|
|
131
|
+
descriptor=d["descriptor"],
|
|
132
|
+
year=d["year"],
|
|
133
|
+
unknown_artist=d["unknownArtist"],
|
|
134
|
+
delimiter=d["delimiter"],
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
@dataclass(frozen=True)
|
|
139
|
+
class Junk:
|
|
140
|
+
"""Removed noise with its kind."""
|
|
141
|
+
|
|
142
|
+
raw: str
|
|
143
|
+
kind: JunkKind
|
|
144
|
+
|
|
145
|
+
def to_dict(self) -> dict[str, Any]:
|
|
146
|
+
return {"raw": self.raw, "kind": self.kind}
|
|
147
|
+
|
|
148
|
+
@classmethod
|
|
149
|
+
def from_dict(cls, d: dict[str, Any]) -> Junk:
|
|
150
|
+
return cls(raw=d["raw"], kind=d["kind"])
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
@dataclass(frozen=True)
|
|
154
|
+
class Flags:
|
|
155
|
+
explicit: bool = False
|
|
156
|
+
clean: bool = False
|
|
157
|
+
unknown_artist: bool = False
|
|
158
|
+
unknown_title: bool = False
|
|
159
|
+
|
|
160
|
+
def to_dict(self) -> dict[str, Any]:
|
|
161
|
+
return {
|
|
162
|
+
"explicit": self.explicit,
|
|
163
|
+
"clean": self.clean,
|
|
164
|
+
"unknownArtist": self.unknown_artist,
|
|
165
|
+
"unknownTitle": self.unknown_title,
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
@classmethod
|
|
169
|
+
def from_dict(cls, d: dict[str, Any]) -> Flags:
|
|
170
|
+
return cls(
|
|
171
|
+
explicit=d["explicit"],
|
|
172
|
+
clean=d["clean"],
|
|
173
|
+
unknown_artist=d["unknownArtist"],
|
|
174
|
+
unknown_title=d["unknownTitle"],
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
@dataclass(frozen=True)
|
|
179
|
+
class Position:
|
|
180
|
+
"""Track number prefix (R4.2)."""
|
|
181
|
+
|
|
182
|
+
raw: str
|
|
183
|
+
number: int
|
|
184
|
+
|
|
185
|
+
def to_dict(self) -> dict[str, Any]:
|
|
186
|
+
return {"raw": self.raw, "number": self.number}
|
|
187
|
+
|
|
188
|
+
@classmethod
|
|
189
|
+
def from_dict(cls, d: dict[str, Any]) -> Position:
|
|
190
|
+
return cls(raw=d["raw"], number=d["number"])
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
@dataclass(frozen=True)
|
|
194
|
+
class Timestamp:
|
|
195
|
+
"""Leading cue time (R4.1)."""
|
|
196
|
+
|
|
197
|
+
raw: str
|
|
198
|
+
seconds: int
|
|
199
|
+
|
|
200
|
+
def to_dict(self) -> dict[str, Any]:
|
|
201
|
+
return {"raw": self.raw, "seconds": self.seconds}
|
|
202
|
+
|
|
203
|
+
@classmethod
|
|
204
|
+
def from_dict(cls, d: dict[str, Any]) -> Timestamp:
|
|
205
|
+
return cls(raw=d["raw"], seconds=d["seconds"])
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
@dataclass(frozen=True)
|
|
209
|
+
class ParsedTrack:
|
|
210
|
+
"""Output of :func:`trackparse.parse`. See SPEC.md "Output model"."""
|
|
211
|
+
|
|
212
|
+
input: str
|
|
213
|
+
mode: Mode
|
|
214
|
+
position: Optional[Position]
|
|
215
|
+
timestamp: Optional[Timestamp]
|
|
216
|
+
artists: tuple[Artist, ...]
|
|
217
|
+
title: str
|
|
218
|
+
full_title: str
|
|
219
|
+
versions: tuple[Version, ...]
|
|
220
|
+
year: Optional[int]
|
|
221
|
+
flags: Flags
|
|
222
|
+
junk: tuple[Junk, ...]
|
|
223
|
+
warnings: tuple[Warning, ...]
|
|
224
|
+
|
|
225
|
+
def to_dict(self) -> dict[str, Any]:
|
|
226
|
+
"""The canonical JSON form (camelCase keys), identical to the JS reference output."""
|
|
227
|
+
return {
|
|
228
|
+
"input": self.input,
|
|
229
|
+
"mode": self.mode,
|
|
230
|
+
"position": None if self.position is None else self.position.to_dict(),
|
|
231
|
+
"timestamp": None if self.timestamp is None else self.timestamp.to_dict(),
|
|
232
|
+
"artists": [a.to_dict() for a in self.artists],
|
|
233
|
+
"title": self.title,
|
|
234
|
+
"fullTitle": self.full_title,
|
|
235
|
+
"versions": [v.to_dict() for v in self.versions],
|
|
236
|
+
"year": self.year,
|
|
237
|
+
"flags": self.flags.to_dict(),
|
|
238
|
+
"junk": [j.to_dict() for j in self.junk],
|
|
239
|
+
"warnings": list(self.warnings),
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
@classmethod
|
|
243
|
+
def from_dict(cls, d: dict[str, Any]) -> ParsedTrack:
|
|
244
|
+
position = d["position"]
|
|
245
|
+
timestamp = d["timestamp"]
|
|
246
|
+
return cls(
|
|
247
|
+
input=d["input"],
|
|
248
|
+
mode=d["mode"],
|
|
249
|
+
position=None if position is None else Position.from_dict(position),
|
|
250
|
+
timestamp=None if timestamp is None else Timestamp.from_dict(timestamp),
|
|
251
|
+
artists=tuple(Artist.from_dict(a) for a in d["artists"]),
|
|
252
|
+
title=d["title"],
|
|
253
|
+
full_title=d["fullTitle"],
|
|
254
|
+
versions=tuple(Version.from_dict(v) for v in d["versions"]),
|
|
255
|
+
year=d["year"],
|
|
256
|
+
flags=Flags.from_dict(d["flags"]),
|
|
257
|
+
junk=tuple(Junk.from_dict(j) for j in d["junk"]),
|
|
258
|
+
warnings=cast(tuple[Warning, ...], tuple(d["warnings"])),
|
|
259
|
+
)
|