trackparse 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- trackparse/__init__.py +119 -0
- trackparse/_artists.py +101 -0
- trackparse/_assemble.py +141 -0
- trackparse/_classify.py +468 -0
- trackparse/_context.py +74 -0
- trackparse/_credits.py +218 -0
- trackparse/_data.py +513 -0
- trackparse/_format.py +149 -0
- trackparse/_mode.py +204 -0
- trackparse/_normalize.py +107 -0
- trackparse/_parser.py +257 -0
- trackparse/_prefix.py +143 -0
- trackparse/_scanner.py +240 -0
- trackparse/_separator.py +209 -0
- trackparse/_tables.py +228 -0
- trackparse/_title.py +246 -0
- trackparse/_words.py +344 -0
- trackparse/options.py +62 -0
- trackparse/py.typed +0 -0
- trackparse/types.py +259 -0
- trackparse-0.2.0.dist-info/METADATA +389 -0
- trackparse-0.2.0.dist-info/RECORD +24 -0
- trackparse-0.2.0.dist-info/WHEEL +4 -0
- trackparse-0.2.0.dist-info/licenses/LICENSE +21 -0
trackparse/_classify.py
ADDED
|
@@ -0,0 +1,468 @@
|
|
|
1
|
+
"""R5: classification of a group's inner text (also used for dash suffixes and pipe segments)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import NamedTuple, Optional, Union
|
|
6
|
+
|
|
7
|
+
from ._context import Ctx
|
|
8
|
+
from ._credits import Credit, ListOptions, ListResult, PositionedYear, split_credits
|
|
9
|
+
from ._scanner import scan
|
|
10
|
+
from ._tables import RunInfo, is_unknown_token
|
|
11
|
+
from ._words import (
|
|
12
|
+
WordSpan,
|
|
13
|
+
ascii_lower,
|
|
14
|
+
dedup_key,
|
|
15
|
+
in_vocab,
|
|
16
|
+
is_all_digits,
|
|
17
|
+
split_words,
|
|
18
|
+
utf16_len,
|
|
19
|
+
word_key,
|
|
20
|
+
word_spans,
|
|
21
|
+
year_of,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class VersionDraft:
|
|
26
|
+
__slots__ = (
|
|
27
|
+
"ambiguous_mix",
|
|
28
|
+
"artists",
|
|
29
|
+
"delimiter",
|
|
30
|
+
"descriptor",
|
|
31
|
+
"modifiers",
|
|
32
|
+
"pos",
|
|
33
|
+
"prefix_form",
|
|
34
|
+
"raw",
|
|
35
|
+
"type",
|
|
36
|
+
"unknown_artist",
|
|
37
|
+
"year",
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
def __init__(self, raw: str, type: str, site: Site) -> None:
|
|
41
|
+
self.type = type
|
|
42
|
+
self.raw = raw
|
|
43
|
+
self.artists: list[Credit] = []
|
|
44
|
+
self.modifiers: list[str] = []
|
|
45
|
+
self.descriptor: Optional[str] = None
|
|
46
|
+
self.year: Optional[int] = None
|
|
47
|
+
self.unknown_artist = False
|
|
48
|
+
self.delimiter = site.delimiter
|
|
49
|
+
self.pos = site.pos
|
|
50
|
+
#: R5.7.2 ambiguousMixCredit; emitted only when the version is used.
|
|
51
|
+
self.ambiguous_mix = False
|
|
52
|
+
#: Recognised by the R5.7.3 prefix form (R6.2 condition (ii) does not apply).
|
|
53
|
+
self.prefix_form = False
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class Junk(NamedTuple):
|
|
57
|
+
junk_kind: str
|
|
58
|
+
kind: str = "junk"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class Feat(NamedTuple):
|
|
62
|
+
credits: list[Credit]
|
|
63
|
+
unknown: bool
|
|
64
|
+
years: list[PositionedYear]
|
|
65
|
+
kind: str = "feat"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class Producer(NamedTuple):
|
|
69
|
+
credits: list[Credit]
|
|
70
|
+
unknown: bool
|
|
71
|
+
years: list[PositionedYear]
|
|
72
|
+
kind: str = "producer"
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class Flag(NamedTuple):
|
|
76
|
+
flag: str # "explicit" | "clean"
|
|
77
|
+
kind: str = "flag"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class Year(NamedTuple):
|
|
81
|
+
year: int
|
|
82
|
+
kind: str = "year"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class Versions(NamedTuple):
|
|
86
|
+
versions: list[VersionDraft]
|
|
87
|
+
kind: str = "versions"
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class Unknown(NamedTuple):
|
|
91
|
+
kind: str = "unknown"
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
Classification = Union[Junk, Feat, Producer, Flag, Year, Versions, Unknown]
|
|
95
|
+
UNKNOWN = Unknown()
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class Site(NamedTuple):
|
|
99
|
+
"""Where the classified text sits: credits' source, versions' delimiter, absolute offset."""
|
|
100
|
+
|
|
101
|
+
source: str
|
|
102
|
+
delimiter: str
|
|
103
|
+
pos: int
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
_EXPLICIT_FLAGS = frozenset(["explicit", "explicit version", "dirty", "dirty version"])
|
|
107
|
+
_CLEAN_FLAGS = frozenset(["clean", "clean version", "radio clean"])
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def classify(inner: str, site: Site, ctx: Ctx) -> Classification:
|
|
111
|
+
"""R5: classify ``inner`` (already trimmed and whitespace-collapsed)."""
|
|
112
|
+
words = split_words(inner)
|
|
113
|
+
if not words:
|
|
114
|
+
return UNKNOWN
|
|
115
|
+
junk_kind = junk_kind_of(words, ctx)
|
|
116
|
+
if junk_kind is not None:
|
|
117
|
+
return Junk(junk_kind)
|
|
118
|
+
credit = _feat_or_producer(inner, site, ctx)
|
|
119
|
+
if credit is not None:
|
|
120
|
+
return credit
|
|
121
|
+
flag = _flag_of(words)
|
|
122
|
+
if flag is not None:
|
|
123
|
+
return Flag(flag)
|
|
124
|
+
if len(words) == 1 and is_all_digits(words[0]):
|
|
125
|
+
year = year_of(words[0])
|
|
126
|
+
if year is not None:
|
|
127
|
+
return Year(year)
|
|
128
|
+
versions = _multi_version(inner, site, ctx)
|
|
129
|
+
if versions is None:
|
|
130
|
+
versions = _single_version(inner, site, ctx)
|
|
131
|
+
if versions is not None:
|
|
132
|
+
return Versions(versions)
|
|
133
|
+
return UNKNOWN
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
# ---------------------------------------------------------------------------
|
|
137
|
+
# R5.1 Junk
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def junk_kind_of(words: list[str], ctx: Ctx) -> Optional[str]:
|
|
141
|
+
"""R5.1: the junk kind if the whole word list is junk vocabulary, else None."""
|
|
142
|
+
first_kind: Optional[str] = None
|
|
143
|
+
i = 0
|
|
144
|
+
n = len(words)
|
|
145
|
+
while i < n:
|
|
146
|
+
m = ctx.t.junk_or_genre.match_at(words, i)
|
|
147
|
+
if m is not None:
|
|
148
|
+
if first_kind is None:
|
|
149
|
+
first_kind = m.entry.value
|
|
150
|
+
i += m.length
|
|
151
|
+
continue
|
|
152
|
+
w = words[i]
|
|
153
|
+
# R5.1: connectors and year words may sit between junk phrases: `Official Video 2015`.
|
|
154
|
+
if (
|
|
155
|
+
w in ctx.t.junk_connectors
|
|
156
|
+
or word_key(w) in ctx.t.junk_connectors
|
|
157
|
+
or year_of(w) is not None
|
|
158
|
+
):
|
|
159
|
+
i += 1
|
|
160
|
+
continue
|
|
161
|
+
first_kind = None
|
|
162
|
+
break
|
|
163
|
+
if first_kind:
|
|
164
|
+
return first_kind
|
|
165
|
+
if n >= 2 and in_vocab(words[-1], ctx.t.label_suffixes):
|
|
166
|
+
return "label"
|
|
167
|
+
return None
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
# ---------------------------------------------------------------------------
|
|
171
|
+
# R5.2 Feat, R5.3 Producer
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _feat_or_producer(inner: str, site: Site, ctx: Ctx) -> Optional[Classification]:
|
|
175
|
+
spans = word_spans(inner)
|
|
176
|
+
words = [s.text for s in spans]
|
|
177
|
+
feat = ctx.t.feat_markers.match_at(words, 0)
|
|
178
|
+
bracket_only = None if feat is not None else ctx.t.bracket_only_feat.match_at(words, 0)
|
|
179
|
+
if feat is not None:
|
|
180
|
+
feat_len = feat.length
|
|
181
|
+
elif bracket_only is not None:
|
|
182
|
+
feat_len = bracket_only.length
|
|
183
|
+
else:
|
|
184
|
+
feat_len = 0
|
|
185
|
+
if feat_len > 0 and len(words) > feat_len:
|
|
186
|
+
nxt = words[feat_len]
|
|
187
|
+
if bracket_only is None or not in_vocab(nxt, ctx.t.stopwords):
|
|
188
|
+
r = _credits_after(inner, spans, feat_len, "featured", site, ctx)
|
|
189
|
+
return Feat(r.credits, r.unknown, r.years)
|
|
190
|
+
prod = ctx.t.producer_markers.match_at(words, 0)
|
|
191
|
+
if prod is not None:
|
|
192
|
+
length = prod.length
|
|
193
|
+
if length < len(words) and word_key(words[length]) == "by":
|
|
194
|
+
length += 1
|
|
195
|
+
if len(words) > length:
|
|
196
|
+
r = _credits_after(inner, spans, length, "producer", site, ctx)
|
|
197
|
+
return Producer(r.credits, r.unknown, r.years)
|
|
198
|
+
return None
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _credits_after(
|
|
202
|
+
inner: str, spans: list[WordSpan], marker_words: int, role: str, site: Site, ctx: Ctx
|
|
203
|
+
) -> ListResult:
|
|
204
|
+
last_marker = spans[marker_words - 1]
|
|
205
|
+
first_name = spans[marker_words]
|
|
206
|
+
lead = inner[spans[0].start : last_marker.end]
|
|
207
|
+
# Offsets are UTF-16 units (see _scanner).
|
|
208
|
+
base = site.pos + 1 + utf16_len(inner[: first_name.start])
|
|
209
|
+
return list_from_text(
|
|
210
|
+
inner[first_name.start :],
|
|
211
|
+
base,
|
|
212
|
+
ctx,
|
|
213
|
+
ListOptions(role, site.source, lead, False, role == "featured"),
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def list_from_text(text: str, base: int, ctx: Ctx, opts: ListOptions) -> ListResult:
|
|
218
|
+
"""Split a plain string (nested brackets protected by a fresh scan) as a credit list."""
|
|
219
|
+
return split_credits(scan(text, base).skel, opts, ctx)
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
# ---------------------------------------------------------------------------
|
|
223
|
+
# R5.4 Flag
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _flag_of(words: list[str]) -> Optional[str]:
|
|
227
|
+
key = " ".join(word_key(w) for w in words)
|
|
228
|
+
if key in _EXPLICIT_FLAGS:
|
|
229
|
+
return "explicit"
|
|
230
|
+
if key in _CLEAN_FLAGS:
|
|
231
|
+
return "clean"
|
|
232
|
+
return None
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
# ---------------------------------------------------------------------------
|
|
236
|
+
# R5.6 Multi-version
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _split_multi_parts(inner: str) -> list[str]:
|
|
240
|
+
"""Split at top-level `` / `` or ``; `` (outside nested brackets)."""
|
|
241
|
+
parts: list[str] = []
|
|
242
|
+
depth = 0
|
|
243
|
+
start = 0
|
|
244
|
+
n = len(inner)
|
|
245
|
+
for i, ch in enumerate(inner):
|
|
246
|
+
if ch == "(" or ch == "[" or ch == "{":
|
|
247
|
+
depth += 1
|
|
248
|
+
elif (ch == ")" or ch == "]" or ch == "}") and depth > 0:
|
|
249
|
+
depth -= 1
|
|
250
|
+
elif (
|
|
251
|
+
depth == 0
|
|
252
|
+
and ch == "/"
|
|
253
|
+
and i > 0
|
|
254
|
+
and inner[i - 1] == " "
|
|
255
|
+
and i + 1 < n
|
|
256
|
+
and inner[i + 1] == " "
|
|
257
|
+
):
|
|
258
|
+
parts.append(inner[start : i - 1])
|
|
259
|
+
start = i + 2
|
|
260
|
+
elif depth == 0 and ch == ";" and i + 1 < n and inner[i + 1] == " ":
|
|
261
|
+
parts.append(inner[start:i])
|
|
262
|
+
start = i + 2
|
|
263
|
+
parts.append(inner[start:])
|
|
264
|
+
# `inner` is whitespace-collapsed, so trimming U+0020 is the R0.2 trim here.
|
|
265
|
+
return [p.strip(" ") for p in parts]
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _multi_version(inner: str, site: Site, ctx: Ctx) -> Optional[list[VersionDraft]]:
|
|
269
|
+
parts = _split_multi_parts(inner)
|
|
270
|
+
if len(parts) < 2:
|
|
271
|
+
return None
|
|
272
|
+
out: list[VersionDraft] = []
|
|
273
|
+
for part in parts:
|
|
274
|
+
v = _single_version(part, site, ctx) if part else None
|
|
275
|
+
if v is None:
|
|
276
|
+
return None
|
|
277
|
+
out.extend(v)
|
|
278
|
+
return out
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
# ---------------------------------------------------------------------------
|
|
282
|
+
# R5.7 Version
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _single_version(raw: str, site: Site, ctx: Ctx) -> Optional[list[VersionDraft]]:
|
|
286
|
+
spans = word_spans(raw)
|
|
287
|
+
words = [s.text for s in spans]
|
|
288
|
+
v = _head_by(raw, spans, words, site, ctx)
|
|
289
|
+
if v is None:
|
|
290
|
+
v = _head_scan(raw, words, site, ctx)
|
|
291
|
+
if v is None:
|
|
292
|
+
v = _prefix_form(raw, words, site, ctx)
|
|
293
|
+
return [v] if v is not None else None
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def _remixer_opts() -> ListOptions:
|
|
297
|
+
return ListOptions("remixer", "version", None, False, False)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def _dedup_remixers(credits: list[Credit]) -> list[Credit]:
|
|
301
|
+
"""R9.2: remixers are deduped within their own version only."""
|
|
302
|
+
seen = set()
|
|
303
|
+
out: list[Credit] = []
|
|
304
|
+
for c in credits:
|
|
305
|
+
k = dedup_key(c.name)
|
|
306
|
+
if k in seen:
|
|
307
|
+
continue
|
|
308
|
+
seen.add(k)
|
|
309
|
+
out.append(c)
|
|
310
|
+
return out
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _set_credits(v: VersionDraft, text: str, site: Site, ctx: Ctx) -> None:
|
|
314
|
+
r = list_from_text(text, site.pos, ctx, _remixer_opts())
|
|
315
|
+
v.artists = _dedup_remixers(r.credits)
|
|
316
|
+
if r.unknown:
|
|
317
|
+
v.unknown_artist = True
|
|
318
|
+
# R7.5: a dropped year-only name sets the version's year if it is still null.
|
|
319
|
+
if r.years and v.year is None:
|
|
320
|
+
v.year = r.years[0].year
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _head_by(
|
|
324
|
+
raw: str, spans: list[WordSpan], words: list[str], site: Site, ctx: Ctx
|
|
325
|
+
) -> Optional[VersionDraft]:
|
|
326
|
+
"""R5.7.1: ``Remix by Skrillex``."""
|
|
327
|
+
head = ctx.t.heads.match_at(words, 0)
|
|
328
|
+
if head is None:
|
|
329
|
+
return None
|
|
330
|
+
if head.length + 1 >= len(words) or word_key(words[head.length]) != "by":
|
|
331
|
+
return None
|
|
332
|
+
rest = spans[head.length + 1]
|
|
333
|
+
v = VersionDraft(raw, head.entry.value, site)
|
|
334
|
+
_set_credits(v, raw[rest.start :], site, ctx)
|
|
335
|
+
return v
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
class _RunItem:
|
|
339
|
+
__slots__ = ("end", "info", "start", "year")
|
|
340
|
+
|
|
341
|
+
def __init__(self, start: int, end: int, info: RunInfo, year: Optional[int]) -> None:
|
|
342
|
+
self.start = start
|
|
343
|
+
self.end = end
|
|
344
|
+
self.info = info
|
|
345
|
+
self.year = year
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
_EMPTY_INFO = RunInfo()
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _collect_run(words: list[str], end: int, ctx: Ctx) -> list[_RunItem]:
|
|
352
|
+
"""Walk left from ``end`` collecting the modifier run (R5.7.2)."""
|
|
353
|
+
items: list[_RunItem] = []
|
|
354
|
+
j = end
|
|
355
|
+
while j > 0:
|
|
356
|
+
m = ctx.t.run_words.match_ending(words, j)
|
|
357
|
+
if m is not None:
|
|
358
|
+
items.append(_RunItem(j - m.length, j, m.entry.value, None))
|
|
359
|
+
j -= m.length
|
|
360
|
+
continue
|
|
361
|
+
year = year_of(words[j - 1])
|
|
362
|
+
if year is None:
|
|
363
|
+
break
|
|
364
|
+
items.append(_RunItem(j - 1, j, _EMPTY_INFO, year))
|
|
365
|
+
j -= 1
|
|
366
|
+
items.reverse()
|
|
367
|
+
return items
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def _head_scan(raw: str, words: list[str], site: Site, ctx: Ctx) -> Optional[VersionDraft]:
|
|
371
|
+
"""R5.7.2: ``Extended VIP Mix``, ``Skrillex Remix``, ``Taylor's Version``."""
|
|
372
|
+
head = ctx.t.heads.match_ending(words, len(words))
|
|
373
|
+
if head is None:
|
|
374
|
+
return None
|
|
375
|
+
head_type = head.entry.value
|
|
376
|
+
head_start = len(words) - head.length
|
|
377
|
+
run = _collect_run(words, head_start, ctx)
|
|
378
|
+
credit_end = run[0].start if run else head_start
|
|
379
|
+
credit_words = words[:credit_end]
|
|
380
|
+
left = run[-1] if run else None
|
|
381
|
+
|
|
382
|
+
# R5.7.2 type resolution. Generic = canonical type mix/edit/version, any spelling.
|
|
383
|
+
generic = head_type in ctx.t.generic_head_types
|
|
384
|
+
typ = head_type
|
|
385
|
+
type_item: Optional[_RunItem] = None
|
|
386
|
+
via_mix = False
|
|
387
|
+
nearest_head = _nearest_non_generic_head(run, ctx) if generic else None
|
|
388
|
+
if generic and left is not None and left.info.type_capable:
|
|
389
|
+
typ = left.info.type_capable
|
|
390
|
+
type_item = left
|
|
391
|
+
elif nearest_head is not None and nearest_head.info.head:
|
|
392
|
+
# `2011 Remastered Version` → remaster
|
|
393
|
+
typ = nearest_head.info.head
|
|
394
|
+
type_item = nearest_head
|
|
395
|
+
elif head_type == "mix" and len(credit_words) > 0:
|
|
396
|
+
typ = "remix"
|
|
397
|
+
via_mix = True
|
|
398
|
+
v = VersionDraft(raw, typ, site)
|
|
399
|
+
|
|
400
|
+
descriptors: list[str] = []
|
|
401
|
+
for item in run:
|
|
402
|
+
text = " ".join(words[item.start : item.end])
|
|
403
|
+
if item.year is not None:
|
|
404
|
+
if v.year is None:
|
|
405
|
+
v.year = item.year
|
|
406
|
+
continue
|
|
407
|
+
modifier = (
|
|
408
|
+
item.info.type_capable if item.info.type_capable is not None else item.info.descriptor
|
|
409
|
+
)
|
|
410
|
+
if modifier is not None:
|
|
411
|
+
if item is not type_item and modifier not in v.modifiers:
|
|
412
|
+
v.modifiers.append(modifier)
|
|
413
|
+
elif item.info.genre:
|
|
414
|
+
descriptors.append(text)
|
|
415
|
+
|
|
416
|
+
if credit_words:
|
|
417
|
+
credit = " ".join(credit_words)
|
|
418
|
+
last_word = ascii_lower(credit_words[-1])
|
|
419
|
+
if is_unknown_token(credit, ctx.t):
|
|
420
|
+
v.unknown_artist = True
|
|
421
|
+
elif typ == "version" or typ == "cover" or _is_all_junk_phrases(credit_words, ctx):
|
|
422
|
+
# R5.7.2: `Japanese Version`, `Taylor's Version`, `Adele Cover`, `Video Edit`.
|
|
423
|
+
descriptors.insert(0, credit)
|
|
424
|
+
else:
|
|
425
|
+
name = credit[:-2] if last_word.endswith("'s") else credit
|
|
426
|
+
_set_credits(v, name, site, ctx)
|
|
427
|
+
if via_mix and v.artists and len(split_words(v.artists[-1].name)) >= 3:
|
|
428
|
+
v.ambiguous_mix = True
|
|
429
|
+
v.descriptor = " ".join(descriptors) if descriptors else None
|
|
430
|
+
return v
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
def _nearest_non_generic_head(run: list[_RunItem], ctx: Ctx) -> Optional[_RunItem]:
|
|
434
|
+
"""R5.7.2 type rule 2: the non-generic head in the run nearest the (generic) head."""
|
|
435
|
+
for item in reversed(run):
|
|
436
|
+
head = item.info.head
|
|
437
|
+
if head is not None and head not in ctx.t.generic_head_types:
|
|
438
|
+
return item
|
|
439
|
+
return None
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
def _is_all_junk_phrases(words: list[str], ctx: Ctx) -> bool:
|
|
443
|
+
"""R5.7.2: the credit span is made entirely of junk phrases (``Video`` in ``Video Edit``)."""
|
|
444
|
+
i = 0
|
|
445
|
+
while i < len(words):
|
|
446
|
+
m = ctx.t.junk_phrases.match_at(words, i)
|
|
447
|
+
if m is None:
|
|
448
|
+
return False
|
|
449
|
+
i += m.length
|
|
450
|
+
return len(words) > 0
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _prefix_form(raw: str, words: list[str], site: Site, ctx: Ctx) -> Optional[VersionDraft]:
|
|
454
|
+
"""R5.7.3: ``Live at Wembley 1986``, ``Remastered 2015``, ``Sped Up``."""
|
|
455
|
+
m = ctx.t.prefix_forms.match_at(words, 0)
|
|
456
|
+
if m is None:
|
|
457
|
+
return None
|
|
458
|
+
v = VersionDraft(raw, m.entry.value, site)
|
|
459
|
+
v.prefix_form = True
|
|
460
|
+
rest: list[str] = []
|
|
461
|
+
for w in words[m.length :]:
|
|
462
|
+
year = year_of(w) if v.year is None else None
|
|
463
|
+
if year is not None:
|
|
464
|
+
v.year = year
|
|
465
|
+
else:
|
|
466
|
+
rest.append(w)
|
|
467
|
+
v.descriptor = " ".join(rest) if rest else None
|
|
468
|
+
return v
|
trackparse/_context.py
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""Per-call parse context: resolved options + vocabulary tables."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from ._tables import Tables
|
|
9
|
+
from ._words import char_at, dedup_key, is_whitespace, utf16_len
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Ctx:
|
|
13
|
+
__slots__ = ("known_keys", "known_max_len", "list_seq", "split_and", "t", "warnings")
|
|
14
|
+
|
|
15
|
+
def __init__(self, t: Tables, options: Mapping[str, Any]) -> None:
|
|
16
|
+
raw_known = options.get("known_artists")
|
|
17
|
+
known: list[str] = []
|
|
18
|
+
if isinstance(raw_known, (list, tuple)):
|
|
19
|
+
known = [k for k in raw_known if isinstance(k, str)]
|
|
20
|
+
self.t = t
|
|
21
|
+
self.known_keys: frozenset[str] = frozenset(
|
|
22
|
+
k for k in (dedup_key(s) for s in known) if len(k) > 0
|
|
23
|
+
)
|
|
24
|
+
#: Longest known_artists entry in UTF-16 units, like the JS reference (bounds R7.3a).
|
|
25
|
+
self.known_max_len: int = max((utf16_len(k) for k in known), default=0)
|
|
26
|
+
split_and = options.get("split_and")
|
|
27
|
+
self.split_and: Any = "auto" if split_and is None else split_and
|
|
28
|
+
self.warnings: list[str] = []
|
|
29
|
+
#: Counter giving every split credit list an id (R7.6 joiner inheritance across R9.2).
|
|
30
|
+
self.list_seq = 0
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def warn(ctx: Ctx, w: str) -> None:
|
|
34
|
+
if w not in ctx.warnings:
|
|
35
|
+
ctx.warnings.append(w)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def match_known_at(text: str, start: int, ctx: Ctx, extra_boundary: str = "") -> int:
|
|
39
|
+
"""R7.3a: length of the longest known_artists entry matching ``text`` at ``start``
|
|
40
|
+
(dedup-key comparison), followed by end, whitespace, ``,`` or ``;`` (or ``extra_boundary``:
|
|
41
|
+
R4.2 also accepts ``-``, so ``3 Doors Down-Kryptonite`` is protected). 0 when none matches.
|
|
42
|
+
|
|
43
|
+
The search window is ``2 * longest + 8`` UTF-16 units, exactly as in the JS reference.
|
|
44
|
+
"""
|
|
45
|
+
if not ctx.known_keys:
|
|
46
|
+
return 0
|
|
47
|
+
budget = ctx.known_max_len * 2 + 8
|
|
48
|
+
limit = start
|
|
49
|
+
used = 0
|
|
50
|
+
n = len(text)
|
|
51
|
+
while limit < n:
|
|
52
|
+
w = 2 if text[limit] > "" else 1
|
|
53
|
+
if used + w > budget:
|
|
54
|
+
break
|
|
55
|
+
used += w
|
|
56
|
+
limit += 1
|
|
57
|
+
for end in range(limit, start, -1):
|
|
58
|
+
nxt = char_at(text, end)
|
|
59
|
+
boundary = (
|
|
60
|
+
end == n
|
|
61
|
+
or nxt == ","
|
|
62
|
+
or nxt == ";"
|
|
63
|
+
or is_whitespace(nxt)
|
|
64
|
+
or (extra_boundary != "" and nxt == extra_boundary)
|
|
65
|
+
)
|
|
66
|
+
if not boundary:
|
|
67
|
+
continue
|
|
68
|
+
if dedup_key(text[start:end]) in ctx.known_keys:
|
|
69
|
+
return end - start
|
|
70
|
+
return 0
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def is_known_artist(text: str, ctx: Ctx) -> bool:
|
|
74
|
+
return len(ctx.known_keys) > 0 and dedup_key(text) in ctx.known_keys
|