trackparse 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,468 @@
1
+ """R5: classification of a group's inner text (also used for dash suffixes and pipe segments)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import NamedTuple, Optional, Union
6
+
7
+ from ._context import Ctx
8
+ from ._credits import Credit, ListOptions, ListResult, PositionedYear, split_credits
9
+ from ._scanner import scan
10
+ from ._tables import RunInfo, is_unknown_token
11
+ from ._words import (
12
+ WordSpan,
13
+ ascii_lower,
14
+ dedup_key,
15
+ in_vocab,
16
+ is_all_digits,
17
+ split_words,
18
+ utf16_len,
19
+ word_key,
20
+ word_spans,
21
+ year_of,
22
+ )
23
+
24
+
25
+ class VersionDraft:
26
+ __slots__ = (
27
+ "ambiguous_mix",
28
+ "artists",
29
+ "delimiter",
30
+ "descriptor",
31
+ "modifiers",
32
+ "pos",
33
+ "prefix_form",
34
+ "raw",
35
+ "type",
36
+ "unknown_artist",
37
+ "year",
38
+ )
39
+
40
+ def __init__(self, raw: str, type: str, site: Site) -> None:
41
+ self.type = type
42
+ self.raw = raw
43
+ self.artists: list[Credit] = []
44
+ self.modifiers: list[str] = []
45
+ self.descriptor: Optional[str] = None
46
+ self.year: Optional[int] = None
47
+ self.unknown_artist = False
48
+ self.delimiter = site.delimiter
49
+ self.pos = site.pos
50
+ #: R5.7.2 ambiguousMixCredit; emitted only when the version is used.
51
+ self.ambiguous_mix = False
52
+ #: Recognised by the R5.7.3 prefix form (R6.2 condition (ii) does not apply).
53
+ self.prefix_form = False
54
+
55
+
56
+ class Junk(NamedTuple):
57
+ junk_kind: str
58
+ kind: str = "junk"
59
+
60
+
61
+ class Feat(NamedTuple):
62
+ credits: list[Credit]
63
+ unknown: bool
64
+ years: list[PositionedYear]
65
+ kind: str = "feat"
66
+
67
+
68
+ class Producer(NamedTuple):
69
+ credits: list[Credit]
70
+ unknown: bool
71
+ years: list[PositionedYear]
72
+ kind: str = "producer"
73
+
74
+
75
+ class Flag(NamedTuple):
76
+ flag: str # "explicit" | "clean"
77
+ kind: str = "flag"
78
+
79
+
80
+ class Year(NamedTuple):
81
+ year: int
82
+ kind: str = "year"
83
+
84
+
85
+ class Versions(NamedTuple):
86
+ versions: list[VersionDraft]
87
+ kind: str = "versions"
88
+
89
+
90
+ class Unknown(NamedTuple):
91
+ kind: str = "unknown"
92
+
93
+
94
+ Classification = Union[Junk, Feat, Producer, Flag, Year, Versions, Unknown]
95
+ UNKNOWN = Unknown()
96
+
97
+
98
+ class Site(NamedTuple):
99
+ """Where the classified text sits: credits' source, versions' delimiter, absolute offset."""
100
+
101
+ source: str
102
+ delimiter: str
103
+ pos: int
104
+
105
+
106
+ _EXPLICIT_FLAGS = frozenset(["explicit", "explicit version", "dirty", "dirty version"])
107
+ _CLEAN_FLAGS = frozenset(["clean", "clean version", "radio clean"])
108
+
109
+
110
+ def classify(inner: str, site: Site, ctx: Ctx) -> Classification:
111
+ """R5: classify ``inner`` (already trimmed and whitespace-collapsed)."""
112
+ words = split_words(inner)
113
+ if not words:
114
+ return UNKNOWN
115
+ junk_kind = junk_kind_of(words, ctx)
116
+ if junk_kind is not None:
117
+ return Junk(junk_kind)
118
+ credit = _feat_or_producer(inner, site, ctx)
119
+ if credit is not None:
120
+ return credit
121
+ flag = _flag_of(words)
122
+ if flag is not None:
123
+ return Flag(flag)
124
+ if len(words) == 1 and is_all_digits(words[0]):
125
+ year = year_of(words[0])
126
+ if year is not None:
127
+ return Year(year)
128
+ versions = _multi_version(inner, site, ctx)
129
+ if versions is None:
130
+ versions = _single_version(inner, site, ctx)
131
+ if versions is not None:
132
+ return Versions(versions)
133
+ return UNKNOWN
134
+
135
+
136
+ # ---------------------------------------------------------------------------
137
+ # R5.1 Junk
138
+
139
+
140
+ def junk_kind_of(words: list[str], ctx: Ctx) -> Optional[str]:
141
+ """R5.1: the junk kind if the whole word list is junk vocabulary, else None."""
142
+ first_kind: Optional[str] = None
143
+ i = 0
144
+ n = len(words)
145
+ while i < n:
146
+ m = ctx.t.junk_or_genre.match_at(words, i)
147
+ if m is not None:
148
+ if first_kind is None:
149
+ first_kind = m.entry.value
150
+ i += m.length
151
+ continue
152
+ w = words[i]
153
+ # R5.1: connectors and year words may sit between junk phrases: `Official Video 2015`.
154
+ if (
155
+ w in ctx.t.junk_connectors
156
+ or word_key(w) in ctx.t.junk_connectors
157
+ or year_of(w) is not None
158
+ ):
159
+ i += 1
160
+ continue
161
+ first_kind = None
162
+ break
163
+ if first_kind:
164
+ return first_kind
165
+ if n >= 2 and in_vocab(words[-1], ctx.t.label_suffixes):
166
+ return "label"
167
+ return None
168
+
169
+
170
+ # ---------------------------------------------------------------------------
171
+ # R5.2 Feat, R5.3 Producer
172
+
173
+
174
+ def _feat_or_producer(inner: str, site: Site, ctx: Ctx) -> Optional[Classification]:
175
+ spans = word_spans(inner)
176
+ words = [s.text for s in spans]
177
+ feat = ctx.t.feat_markers.match_at(words, 0)
178
+ bracket_only = None if feat is not None else ctx.t.bracket_only_feat.match_at(words, 0)
179
+ if feat is not None:
180
+ feat_len = feat.length
181
+ elif bracket_only is not None:
182
+ feat_len = bracket_only.length
183
+ else:
184
+ feat_len = 0
185
+ if feat_len > 0 and len(words) > feat_len:
186
+ nxt = words[feat_len]
187
+ if bracket_only is None or not in_vocab(nxt, ctx.t.stopwords):
188
+ r = _credits_after(inner, spans, feat_len, "featured", site, ctx)
189
+ return Feat(r.credits, r.unknown, r.years)
190
+ prod = ctx.t.producer_markers.match_at(words, 0)
191
+ if prod is not None:
192
+ length = prod.length
193
+ if length < len(words) and word_key(words[length]) == "by":
194
+ length += 1
195
+ if len(words) > length:
196
+ r = _credits_after(inner, spans, length, "producer", site, ctx)
197
+ return Producer(r.credits, r.unknown, r.years)
198
+ return None
199
+
200
+
201
+ def _credits_after(
202
+ inner: str, spans: list[WordSpan], marker_words: int, role: str, site: Site, ctx: Ctx
203
+ ) -> ListResult:
204
+ last_marker = spans[marker_words - 1]
205
+ first_name = spans[marker_words]
206
+ lead = inner[spans[0].start : last_marker.end]
207
+ # Offsets are UTF-16 units (see _scanner).
208
+ base = site.pos + 1 + utf16_len(inner[: first_name.start])
209
+ return list_from_text(
210
+ inner[first_name.start :],
211
+ base,
212
+ ctx,
213
+ ListOptions(role, site.source, lead, False, role == "featured"),
214
+ )
215
+
216
+
217
+ def list_from_text(text: str, base: int, ctx: Ctx, opts: ListOptions) -> ListResult:
218
+ """Split a plain string (nested brackets protected by a fresh scan) as a credit list."""
219
+ return split_credits(scan(text, base).skel, opts, ctx)
220
+
221
+
222
+ # ---------------------------------------------------------------------------
223
+ # R5.4 Flag
224
+
225
+
226
+ def _flag_of(words: list[str]) -> Optional[str]:
227
+ key = " ".join(word_key(w) for w in words)
228
+ if key in _EXPLICIT_FLAGS:
229
+ return "explicit"
230
+ if key in _CLEAN_FLAGS:
231
+ return "clean"
232
+ return None
233
+
234
+
235
+ # ---------------------------------------------------------------------------
236
+ # R5.6 Multi-version
237
+
238
+
239
+ def _split_multi_parts(inner: str) -> list[str]:
240
+ """Split at top-level `` / `` or ``; `` (outside nested brackets)."""
241
+ parts: list[str] = []
242
+ depth = 0
243
+ start = 0
244
+ n = len(inner)
245
+ for i, ch in enumerate(inner):
246
+ if ch == "(" or ch == "[" or ch == "{":
247
+ depth += 1
248
+ elif (ch == ")" or ch == "]" or ch == "}") and depth > 0:
249
+ depth -= 1
250
+ elif (
251
+ depth == 0
252
+ and ch == "/"
253
+ and i > 0
254
+ and inner[i - 1] == " "
255
+ and i + 1 < n
256
+ and inner[i + 1] == " "
257
+ ):
258
+ parts.append(inner[start : i - 1])
259
+ start = i + 2
260
+ elif depth == 0 and ch == ";" and i + 1 < n and inner[i + 1] == " ":
261
+ parts.append(inner[start:i])
262
+ start = i + 2
263
+ parts.append(inner[start:])
264
+ # `inner` is whitespace-collapsed, so trimming U+0020 is the R0.2 trim here.
265
+ return [p.strip(" ") for p in parts]
266
+
267
+
268
+ def _multi_version(inner: str, site: Site, ctx: Ctx) -> Optional[list[VersionDraft]]:
269
+ parts = _split_multi_parts(inner)
270
+ if len(parts) < 2:
271
+ return None
272
+ out: list[VersionDraft] = []
273
+ for part in parts:
274
+ v = _single_version(part, site, ctx) if part else None
275
+ if v is None:
276
+ return None
277
+ out.extend(v)
278
+ return out
279
+
280
+
281
+ # ---------------------------------------------------------------------------
282
+ # R5.7 Version
283
+
284
+
285
+ def _single_version(raw: str, site: Site, ctx: Ctx) -> Optional[list[VersionDraft]]:
286
+ spans = word_spans(raw)
287
+ words = [s.text for s in spans]
288
+ v = _head_by(raw, spans, words, site, ctx)
289
+ if v is None:
290
+ v = _head_scan(raw, words, site, ctx)
291
+ if v is None:
292
+ v = _prefix_form(raw, words, site, ctx)
293
+ return [v] if v is not None else None
294
+
295
+
296
+ def _remixer_opts() -> ListOptions:
297
+ return ListOptions("remixer", "version", None, False, False)
298
+
299
+
300
+ def _dedup_remixers(credits: list[Credit]) -> list[Credit]:
301
+ """R9.2: remixers are deduped within their own version only."""
302
+ seen = set()
303
+ out: list[Credit] = []
304
+ for c in credits:
305
+ k = dedup_key(c.name)
306
+ if k in seen:
307
+ continue
308
+ seen.add(k)
309
+ out.append(c)
310
+ return out
311
+
312
+
313
+ def _set_credits(v: VersionDraft, text: str, site: Site, ctx: Ctx) -> None:
314
+ r = list_from_text(text, site.pos, ctx, _remixer_opts())
315
+ v.artists = _dedup_remixers(r.credits)
316
+ if r.unknown:
317
+ v.unknown_artist = True
318
+ # R7.5: a dropped year-only name sets the version's year if it is still null.
319
+ if r.years and v.year is None:
320
+ v.year = r.years[0].year
321
+
322
+
323
+ def _head_by(
324
+ raw: str, spans: list[WordSpan], words: list[str], site: Site, ctx: Ctx
325
+ ) -> Optional[VersionDraft]:
326
+ """R5.7.1: ``Remix by Skrillex``."""
327
+ head = ctx.t.heads.match_at(words, 0)
328
+ if head is None:
329
+ return None
330
+ if head.length + 1 >= len(words) or word_key(words[head.length]) != "by":
331
+ return None
332
+ rest = spans[head.length + 1]
333
+ v = VersionDraft(raw, head.entry.value, site)
334
+ _set_credits(v, raw[rest.start :], site, ctx)
335
+ return v
336
+
337
+
338
+ class _RunItem:
339
+ __slots__ = ("end", "info", "start", "year")
340
+
341
+ def __init__(self, start: int, end: int, info: RunInfo, year: Optional[int]) -> None:
342
+ self.start = start
343
+ self.end = end
344
+ self.info = info
345
+ self.year = year
346
+
347
+
348
+ _EMPTY_INFO = RunInfo()
349
+
350
+
351
+ def _collect_run(words: list[str], end: int, ctx: Ctx) -> list[_RunItem]:
352
+ """Walk left from ``end`` collecting the modifier run (R5.7.2)."""
353
+ items: list[_RunItem] = []
354
+ j = end
355
+ while j > 0:
356
+ m = ctx.t.run_words.match_ending(words, j)
357
+ if m is not None:
358
+ items.append(_RunItem(j - m.length, j, m.entry.value, None))
359
+ j -= m.length
360
+ continue
361
+ year = year_of(words[j - 1])
362
+ if year is None:
363
+ break
364
+ items.append(_RunItem(j - 1, j, _EMPTY_INFO, year))
365
+ j -= 1
366
+ items.reverse()
367
+ return items
368
+
369
+
370
+ def _head_scan(raw: str, words: list[str], site: Site, ctx: Ctx) -> Optional[VersionDraft]:
371
+ """R5.7.2: ``Extended VIP Mix``, ``Skrillex Remix``, ``Taylor's Version``."""
372
+ head = ctx.t.heads.match_ending(words, len(words))
373
+ if head is None:
374
+ return None
375
+ head_type = head.entry.value
376
+ head_start = len(words) - head.length
377
+ run = _collect_run(words, head_start, ctx)
378
+ credit_end = run[0].start if run else head_start
379
+ credit_words = words[:credit_end]
380
+ left = run[-1] if run else None
381
+
382
+ # R5.7.2 type resolution. Generic = canonical type mix/edit/version, any spelling.
383
+ generic = head_type in ctx.t.generic_head_types
384
+ typ = head_type
385
+ type_item: Optional[_RunItem] = None
386
+ via_mix = False
387
+ nearest_head = _nearest_non_generic_head(run, ctx) if generic else None
388
+ if generic and left is not None and left.info.type_capable:
389
+ typ = left.info.type_capable
390
+ type_item = left
391
+ elif nearest_head is not None and nearest_head.info.head:
392
+ # `2011 Remastered Version` → remaster
393
+ typ = nearest_head.info.head
394
+ type_item = nearest_head
395
+ elif head_type == "mix" and len(credit_words) > 0:
396
+ typ = "remix"
397
+ via_mix = True
398
+ v = VersionDraft(raw, typ, site)
399
+
400
+ descriptors: list[str] = []
401
+ for item in run:
402
+ text = " ".join(words[item.start : item.end])
403
+ if item.year is not None:
404
+ if v.year is None:
405
+ v.year = item.year
406
+ continue
407
+ modifier = (
408
+ item.info.type_capable if item.info.type_capable is not None else item.info.descriptor
409
+ )
410
+ if modifier is not None:
411
+ if item is not type_item and modifier not in v.modifiers:
412
+ v.modifiers.append(modifier)
413
+ elif item.info.genre:
414
+ descriptors.append(text)
415
+
416
+ if credit_words:
417
+ credit = " ".join(credit_words)
418
+ last_word = ascii_lower(credit_words[-1])
419
+ if is_unknown_token(credit, ctx.t):
420
+ v.unknown_artist = True
421
+ elif typ == "version" or typ == "cover" or _is_all_junk_phrases(credit_words, ctx):
422
+ # R5.7.2: `Japanese Version`, `Taylor's Version`, `Adele Cover`, `Video Edit`.
423
+ descriptors.insert(0, credit)
424
+ else:
425
+ name = credit[:-2] if last_word.endswith("'s") else credit
426
+ _set_credits(v, name, site, ctx)
427
+ if via_mix and v.artists and len(split_words(v.artists[-1].name)) >= 3:
428
+ v.ambiguous_mix = True
429
+ v.descriptor = " ".join(descriptors) if descriptors else None
430
+ return v
431
+
432
+
433
+ def _nearest_non_generic_head(run: list[_RunItem], ctx: Ctx) -> Optional[_RunItem]:
434
+ """R5.7.2 type rule 2: the non-generic head in the run nearest the (generic) head."""
435
+ for item in reversed(run):
436
+ head = item.info.head
437
+ if head is not None and head not in ctx.t.generic_head_types:
438
+ return item
439
+ return None
440
+
441
+
442
+ def _is_all_junk_phrases(words: list[str], ctx: Ctx) -> bool:
443
+ """R5.7.2: the credit span is made entirely of junk phrases (``Video`` in ``Video Edit``)."""
444
+ i = 0
445
+ while i < len(words):
446
+ m = ctx.t.junk_phrases.match_at(words, i)
447
+ if m is None:
448
+ return False
449
+ i += m.length
450
+ return len(words) > 0
451
+
452
+
453
+ def _prefix_form(raw: str, words: list[str], site: Site, ctx: Ctx) -> Optional[VersionDraft]:
454
+ """R5.7.3: ``Live at Wembley 1986``, ``Remastered 2015``, ``Sped Up``."""
455
+ m = ctx.t.prefix_forms.match_at(words, 0)
456
+ if m is None:
457
+ return None
458
+ v = VersionDraft(raw, m.entry.value, site)
459
+ v.prefix_form = True
460
+ rest: list[str] = []
461
+ for w in words[m.length :]:
462
+ year = year_of(w) if v.year is None else None
463
+ if year is not None:
464
+ v.year = year
465
+ else:
466
+ rest.append(w)
467
+ v.descriptor = " ".join(rest) if rest else None
468
+ return v
trackparse/_context.py ADDED
@@ -0,0 +1,74 @@
1
+ """Per-call parse context: resolved options + vocabulary tables."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from typing import Any
7
+
8
+ from ._tables import Tables
9
+ from ._words import char_at, dedup_key, is_whitespace, utf16_len
10
+
11
+
12
+ class Ctx:
13
+ __slots__ = ("known_keys", "known_max_len", "list_seq", "split_and", "t", "warnings")
14
+
15
+ def __init__(self, t: Tables, options: Mapping[str, Any]) -> None:
16
+ raw_known = options.get("known_artists")
17
+ known: list[str] = []
18
+ if isinstance(raw_known, (list, tuple)):
19
+ known = [k for k in raw_known if isinstance(k, str)]
20
+ self.t = t
21
+ self.known_keys: frozenset[str] = frozenset(
22
+ k for k in (dedup_key(s) for s in known) if len(k) > 0
23
+ )
24
+ #: Longest known_artists entry in UTF-16 units, like the JS reference (bounds R7.3a).
25
+ self.known_max_len: int = max((utf16_len(k) for k in known), default=0)
26
+ split_and = options.get("split_and")
27
+ self.split_and: Any = "auto" if split_and is None else split_and
28
+ self.warnings: list[str] = []
29
+ #: Counter giving every split credit list an id (R7.6 joiner inheritance across R9.2).
30
+ self.list_seq = 0
31
+
32
+
33
+ def warn(ctx: Ctx, w: str) -> None:
34
+ if w not in ctx.warnings:
35
+ ctx.warnings.append(w)
36
+
37
+
38
+ def match_known_at(text: str, start: int, ctx: Ctx, extra_boundary: str = "") -> int:
39
+ """R7.3a: length of the longest known_artists entry matching ``text`` at ``start``
40
+ (dedup-key comparison), followed by end, whitespace, ``,`` or ``;`` (or ``extra_boundary``:
41
+ R4.2 also accepts ``-``, so ``3 Doors Down-Kryptonite`` is protected). 0 when none matches.
42
+
43
+ The search window is ``2 * longest + 8`` UTF-16 units, exactly as in the JS reference.
44
+ """
45
+ if not ctx.known_keys:
46
+ return 0
47
+ budget = ctx.known_max_len * 2 + 8
48
+ limit = start
49
+ used = 0
50
+ n = len(text)
51
+ while limit < n:
52
+ w = 2 if text[limit] > "￿" else 1
53
+ if used + w > budget:
54
+ break
55
+ used += w
56
+ limit += 1
57
+ for end in range(limit, start, -1):
58
+ nxt = char_at(text, end)
59
+ boundary = (
60
+ end == n
61
+ or nxt == ","
62
+ or nxt == ";"
63
+ or is_whitespace(nxt)
64
+ or (extra_boundary != "" and nxt == extra_boundary)
65
+ )
66
+ if not boundary:
67
+ continue
68
+ if dedup_key(text[start:end]) in ctx.known_keys:
69
+ return end - start
70
+ return 0
71
+
72
+
73
+ def is_known_artist(text: str, ctx: Ctx) -> bool:
74
+ return len(ctx.known_keys) > 0 and dedup_key(text) in ctx.known_keys