trackparse 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
trackparse/_data.py ADDED
@@ -0,0 +1,513 @@
1
+ # GENERATED from spec/data — do not edit. Run `python scripts/gen_data.py` to regenerate.
2
+ """Word lists compiled from spec/data (see SPEC.md). Generated; never edit by hand."""
3
+
4
+ from __future__ import annotations
5
+
6
+ from typing import Any
7
+
8
+ SPEC_VERSION = "0.2.0" # x-release-please-version
9
+
10
+ # spec/data/descriptors.json
11
+ DESCRIPTORS_DATA: dict[str, Any] = {
12
+ "descriptors": [
13
+ "radio",
14
+ "extended",
15
+ "club",
16
+ "original",
17
+ "instrumental",
18
+ "dub",
19
+ "vip",
20
+ "acoustic",
21
+ "live",
22
+ "demo",
23
+ "vocal",
24
+ "clean",
25
+ "dirty",
26
+ "explicit",
27
+ "short",
28
+ "long",
29
+ "full",
30
+ "album",
31
+ "single",
32
+ "alternate",
33
+ "alt",
34
+ "alternative",
35
+ "main",
36
+ "special",
37
+ "bonus",
38
+ "deluxe",
39
+ "mono",
40
+ "stereo",
41
+ "new",
42
+ "official",
43
+ "orchestral",
44
+ "piano",
45
+ "unplugged",
46
+ "12\"",
47
+ "7\"",
48
+ "12''",
49
+ "7''",
50
+ "12 inch",
51
+ "7 inch",
52
+ "re",
53
+ "final",
54
+ "second",
55
+ "2nd",
56
+ "3rd",
57
+ ],
58
+ }
59
+
60
+ # spec/data/extensions.json
61
+ EXTENSIONS_DATA: dict[str, Any] = {
62
+ "extensions": [
63
+ "mp3",
64
+ "flac",
65
+ "wav",
66
+ "m4a",
67
+ "ogg",
68
+ "opus",
69
+ "aac",
70
+ "aiff",
71
+ "aif",
72
+ "wma",
73
+ "alac",
74
+ "mp4",
75
+ "webm",
76
+ "mkv",
77
+ ],
78
+ }
79
+
80
+ # spec/data/feat-markers.json
81
+ FEAT_MARKERS_DATA: dict[str, Any] = {
82
+ "markers": [
83
+ "feat.",
84
+ "feat",
85
+ "ft.",
86
+ "ft",
87
+ "featuring",
88
+ "feat:",
89
+ "ft:",
90
+ ],
91
+ "bracketOnly": [
92
+ "with",
93
+ "w/",
94
+ ],
95
+ "producer": [
96
+ "prod.",
97
+ "prod",
98
+ "produced",
99
+ "prod.by",
100
+ "prod:",
101
+ ],
102
+ "titleMarkers": [
103
+ "feat.",
104
+ "ft.",
105
+ "featuring",
106
+ ],
107
+ "titleProducer": [
108
+ "prod.",
109
+ "prod. by",
110
+ "produced by",
111
+ "prod.by",
112
+ ],
113
+ "producerBy": "by",
114
+ }
115
+
116
+ # spec/data/genres.json
117
+ GENRES_DATA: dict[str, Any] = {
118
+ "genres": [
119
+ "d'n'b",
120
+ "dnb",
121
+ "d&b",
122
+ "d & b",
123
+ "drum & bass",
124
+ "drum&bass",
125
+ "drum and bass",
126
+ "drum n bass",
127
+ "drum 'n' bass",
128
+ "drum'n'bass",
129
+ "neurofunk",
130
+ "neuro",
131
+ "liquid",
132
+ "liquid funk",
133
+ "jump up",
134
+ "jump-up",
135
+ "jungle",
136
+ "halftime",
137
+ "dancefloor",
138
+ "dubstep",
139
+ "riddim",
140
+ "brostep",
141
+ "trap",
142
+ "future bass",
143
+ "house",
144
+ "deep house",
145
+ "tech house",
146
+ "progressive house",
147
+ "electro house",
148
+ "techno",
149
+ "trance",
150
+ "psytrance",
151
+ "psy",
152
+ "hardstyle",
153
+ "hardcore",
154
+ "breaks",
155
+ "breakbeat",
156
+ "garage",
157
+ "uk garage",
158
+ "ukg",
159
+ "2step",
160
+ "2-step",
161
+ "grime",
162
+ "bass",
163
+ "bass house",
164
+ "edm",
165
+ "electronic",
166
+ "idm",
167
+ "synthwave",
168
+ "phonk",
169
+ "hip hop",
170
+ "hip-hop",
171
+ "rap",
172
+ "pop",
173
+ "rock",
174
+ "metal",
175
+ "ambient",
176
+ "chillout",
177
+ "lofi",
178
+ "lo-fi",
179
+ ],
180
+ }
181
+
182
+ # spec/data/joiners.json
183
+ JOINERS_DATA: dict[str, Any] = {
184
+ "spaced": [
185
+ {
186
+ "raw": "&",
187
+ "canonical": "&",
188
+ },
189
+ {
190
+ "raw": "and",
191
+ "canonical": "&",
192
+ "kind": "and",
193
+ },
194
+ {
195
+ "raw": "x",
196
+ "canonical": "x",
197
+ "caseSensitive": True,
198
+ },
199
+ {
200
+ "raw": "×",
201
+ "canonical": "x",
202
+ },
203
+ {
204
+ "raw": "vs.",
205
+ "canonical": "vs.",
206
+ },
207
+ {
208
+ "raw": "vs",
209
+ "canonical": "vs.",
210
+ },
211
+ {
212
+ "raw": "versus",
213
+ "canonical": "vs.",
214
+ },
215
+ {
216
+ "raw": "/",
217
+ "canonical": "/",
218
+ },
219
+ {
220
+ "raw": "+",
221
+ "canonical": "+",
222
+ },
223
+ {
224
+ "raw": "with",
225
+ "canonical": "with",
226
+ },
227
+ {
228
+ "raw": "meets",
229
+ "canonical": "meets",
230
+ },
231
+ {
232
+ "raw": "pres.",
233
+ "canonical": "pres.",
234
+ },
235
+ {
236
+ "raw": "presents",
237
+ "canonical": "pres.",
238
+ },
239
+ {
240
+ "raw": "b2b",
241
+ "canonical": "b2b",
242
+ },
243
+ ],
244
+ "tight": [
245
+ {
246
+ "raw": ",",
247
+ "canonical": ",",
248
+ },
249
+ {
250
+ "raw": ";",
251
+ "canonical": ",",
252
+ },
253
+ ],
254
+ "featCanonical": "feat.",
255
+ }
256
+
257
+ # spec/data/junk.json
258
+ JUNK_DATA: dict[str, Any] = {
259
+ "phrases": {
260
+ "official video": "video",
261
+ "official music video": "video",
262
+ "music video": "video",
263
+ "official video clip": "video",
264
+ "video clip": "video",
265
+ "official clip": "video",
266
+ "clip officiel": "video",
267
+ "official hd video": "video",
268
+ "official 4k video": "video",
269
+ "official mv": "video",
270
+ "mv": "video",
271
+ "m/v": "video",
272
+ "video": "video",
273
+ "official visualizer": "video",
274
+ "official visualiser": "video",
275
+ "visualizer": "video",
276
+ "visualiser": "video",
277
+ "audio visualizer": "video",
278
+ "official animated video": "video",
279
+ "animated video": "video",
280
+ "live video": "video",
281
+ "official live video": "video",
282
+ "performance video": "video",
283
+ "official performance video": "video",
284
+ "official audio": "audio",
285
+ "audio": "audio",
286
+ "audio only": "audio",
287
+ "full audio": "audio",
288
+ "official lyric video": "lyrics",
289
+ "official lyrics video": "lyrics",
290
+ "lyric video": "lyrics",
291
+ "lyrics video": "lyrics",
292
+ "lyrics": "lyrics",
293
+ "lyric": "lyrics",
294
+ "with lyrics": "lyrics",
295
+ "lyrics on screen": "lyrics",
296
+ "letra": "lyrics",
297
+ "hd": "quality",
298
+ "hq": "quality",
299
+ "uhd": "quality",
300
+ "4k": "quality",
301
+ "8k": "quality",
302
+ "1080p": "quality",
303
+ "720p": "quality",
304
+ "480p": "quality",
305
+ "high quality": "quality",
306
+ "full hd": "quality",
307
+ "free download": "promo",
308
+ "free dl": "promo",
309
+ "out now": "promo",
310
+ "premiere": "promo",
311
+ "exclusive": "promo",
312
+ "preview": "promo",
313
+ "teaser": "promo",
314
+ "snippet": "promo",
315
+ "new song": "promo",
316
+ "available now": "promo",
317
+ "ncs release": "label",
318
+ "monstercat release": "label",
319
+ "official": "other",
320
+ },
321
+ "connectors": [
322
+ "-",
323
+ "|",
324
+ "/",
325
+ "&",
326
+ "+",
327
+ ",",
328
+ ],
329
+ "labelSuffixes": [
330
+ "release",
331
+ "records",
332
+ "recordings",
333
+ ],
334
+ }
335
+
336
+ # spec/data/no-split-before.json
337
+ NO_SPLIT_BEFORE_DATA: dict[str, Any] = {
338
+ "words": [
339
+ "sons",
340
+ "daughters",
341
+ "friends",
342
+ "co",
343
+ "co.",
344
+ "company",
345
+ "his",
346
+ "her",
347
+ "their",
348
+ "orchestra",
349
+ ],
350
+ }
351
+
352
+ # spec/data/platform-suffixes.json
353
+ PLATFORM_SUFFIXES_DATA: dict[str, Any] = {
354
+ "suffixes": [
355
+ " - YouTube Music",
356
+ " - YouTube",
357
+ " | Free Listening on SoundCloud",
358
+ " - SoundCloud",
359
+ " | SoundCloud",
360
+ " - Topic",
361
+ " | Bandcamp",
362
+ " - Bandcamp",
363
+ ],
364
+ }
365
+
366
+ # spec/data/stopwords.json
367
+ STOPWORDS_DATA: dict[str, Any] = {
368
+ "words": [
369
+ "you",
370
+ "me",
371
+ "him",
372
+ "her",
373
+ "us",
374
+ "them",
375
+ "it",
376
+ "i",
377
+ "love",
378
+ "my",
379
+ "your",
380
+ "our",
381
+ "the",
382
+ "a",
383
+ "an",
384
+ "this",
385
+ "that",
386
+ "myself",
387
+ "yourself",
388
+ "everyone",
389
+ "everybody",
390
+ "somebody",
391
+ "someone",
392
+ "nobody",
393
+ "nothing",
394
+ "all",
395
+ "night",
396
+ "side",
397
+ "day",
398
+ ],
399
+ }
400
+
401
+ # spec/data/unknown-tokens.json
402
+ UNKNOWN_TOKENS_DATA: dict[str, Any] = {
403
+ "caseSensitive": [
404
+ "ID",
405
+ "IDs",
406
+ "ID?",
407
+ ],
408
+ "caseInsensitive": [
409
+ "unknown",
410
+ "unknown artist",
411
+ "unknown title",
412
+ "untitled",
413
+ "?",
414
+ "??",
415
+ "???",
416
+ ],
417
+ }
418
+
419
+ # spec/data/version-keywords.json
420
+ VERSION_KEYWORDS_DATA: dict[str, Any] = {
421
+ "heads": {
422
+ "remix": "remix",
423
+ "rmx": "remix",
424
+ "mix": "mix",
425
+ "bootleg": "bootleg",
426
+ "vip": "vip",
427
+ "edit": "edit",
428
+ "re-edit": "edit",
429
+ "reedit": "edit",
430
+ "flip": "flip",
431
+ "refix": "refix",
432
+ "rework": "rework",
433
+ "re-work": "rework",
434
+ "mashup": "mashup",
435
+ "mash-up": "mashup",
436
+ "mash up": "mashup",
437
+ "blend": "blend",
438
+ "dub": "dub",
439
+ "version": "version",
440
+ "ver": "version",
441
+ "remaster": "remaster",
442
+ "remastered": "remaster",
443
+ "cover": "cover",
444
+ "reprise": "reprise",
445
+ },
446
+ "genericHeads": [
447
+ "mix",
448
+ "edit",
449
+ "version",
450
+ ],
451
+ "typeCapableModifiers": {
452
+ "radio": "radio",
453
+ "extended": "extended",
454
+ "club": "club",
455
+ "original": "original",
456
+ "instrumental": "instrumental",
457
+ "dub": "dub",
458
+ "vip": "vip",
459
+ "acoustic": "acoustic",
460
+ "live": "live",
461
+ "acapella": "acapella",
462
+ "a cappella": "acapella",
463
+ "a capella": "acapella",
464
+ "acappella": "acapella",
465
+ "demo": "demo",
466
+ },
467
+ "prefixForms": {
468
+ "live": "live",
469
+ "remastered": "remaster",
470
+ "remaster": "remaster",
471
+ "acoustic": "acoustic",
472
+ "instrumental": "instrumental",
473
+ "acapella": "acapella",
474
+ "a cappella": "acapella",
475
+ "demo": "demo",
476
+ "sped up": "spedUp",
477
+ "speed up": "spedUp",
478
+ "slowed + reverb": "slowed",
479
+ "slowed & reverb": "slowed",
480
+ "slowed and reverb": "slowed",
481
+ "slowed": "slowed",
482
+ "nightcore": "nightcore",
483
+ },
484
+ "types": [
485
+ "remix",
486
+ "bootleg",
487
+ "vip",
488
+ "edit",
489
+ "flip",
490
+ "refix",
491
+ "rework",
492
+ "mashup",
493
+ "blend",
494
+ "dub",
495
+ "mix",
496
+ "extended",
497
+ "radio",
498
+ "club",
499
+ "original",
500
+ "instrumental",
501
+ "acapella",
502
+ "live",
503
+ "acoustic",
504
+ "remaster",
505
+ "demo",
506
+ "reprise",
507
+ "cover",
508
+ "version",
509
+ "spedUp",
510
+ "slowed",
511
+ "nightcore",
512
+ ],
513
+ }
trackparse/_format.py ADDED
@@ -0,0 +1,149 @@
1
+ """R11: render a ParsedTrack back to a string."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Sequence
6
+ from typing import NamedTuple, Optional
7
+
8
+ from ._assemble import render_version
9
+ from ._tables import Tables, get_default_tables
10
+ from ._words import dedup_key, word_matches
11
+ from .types import Artist, ParsedTrack
12
+
13
+
14
+ class _Resolved(NamedTuple):
15
+ feat: object
16
+ feat_marker: Optional[str]
17
+ canonical: bool
18
+ versions: bool
19
+ producers: object
20
+ position: object
21
+
22
+
23
+ def _is_feat_marker(raw: str, t: Tables) -> bool:
24
+ return (
25
+ t.feat_markers.match_at([raw], 0) is not None
26
+ or t.bracket_only_feat.match_at([raw], 0) is not None
27
+ )
28
+
29
+
30
+ def _canonical_joiner(raw: str, t: Tables, feat_lead: bool = False) -> str:
31
+ """R11.5: map a raw joiner to its canonical form through data/joiners.json; feat markers
32
+ become ``featCanonical``. A featured list's lead marker is always a feat marker (``with`` →
33
+ ``feat.``), while between names the joiner table wins (``Sigala with Ella Eyre`` keeps
34
+ ``with``)."""
35
+ if feat_lead and _is_feat_marker(raw, t):
36
+ return t.feat_canonical
37
+ tight = t.tight_joiners.get(raw)
38
+ if tight is not None:
39
+ return tight
40
+ for j in t.spaced_joiners:
41
+ if (raw == j.raw) if j.case_sensitive else word_matches(raw, j.raw):
42
+ return j.canonical
43
+ return t.feat_canonical if _is_feat_marker(raw, t) else raw
44
+
45
+
46
+ def _join_sep(joiner: Optional[str], o: _Resolved, t: Tables) -> str:
47
+ """R11.1: ``" " + joiner + " "``; tight joiners as the raw char + ``" "``; a missing joiner
48
+ as ``&``."""
49
+ if joiner is None:
50
+ j = "&"
51
+ elif o.canonical:
52
+ j = _canonical_joiner(joiner, t)
53
+ else:
54
+ j = joiner
55
+ return f"{j} " if j in t.tight_joiners else f" {j} "
56
+
57
+
58
+ def _join_names(lst: Sequence[Artist], o: _Resolved, t: Tables) -> str:
59
+ """Names of one credit list joined by their own joiners (the first joiner is not
60
+ rendered)."""
61
+ return "".join(
62
+ a.name if i == 0 else _join_sep(a.joiner, o, t) + a.name for i, a in enumerate(lst)
63
+ )
64
+
65
+
66
+ def _lead_marker(first: Artist, o: _Resolved, t: Tables, fallback: str) -> str:
67
+ """The marker in front of a featured list."""
68
+ if o.feat_marker is not None:
69
+ return o.feat_marker
70
+ raw = first.joiner if first.joiner is not None else fallback
71
+ return _canonical_joiner(raw, t, True) if o.canonical else raw
72
+
73
+
74
+ def format_track(
75
+ track: ParsedTrack,
76
+ *,
77
+ feat: str = "source",
78
+ feat_marker: Optional[str] = None,
79
+ joiners: str = "original",
80
+ versions: str = "all",
81
+ producers: bool = True,
82
+ position: bool = False,
83
+ ) -> str:
84
+ """R11: render ``track`` as ``Artists - Title (versions) (prod. ...)``."""
85
+ t = get_default_tables()
86
+ o = _Resolved(
87
+ feat="source" if feat is None else feat,
88
+ feat_marker=feat_marker,
89
+ canonical=joiners == "canonical",
90
+ versions=("all" if versions is None else versions) == "all",
91
+ producers=True if producers is None else producers,
92
+ position=False if position is None else position,
93
+ )
94
+ artists = list(track.artists)
95
+ primaries = [a for a in artists if a.role == "primary"]
96
+ featured = [a for a in artists if a.role == "featured"]
97
+ producer_list = [a for a in artists if a.role == "producer"]
98
+
99
+ artist_feat: list[Artist] = []
100
+ title_feat: list[Artist] = []
101
+ if o.feat == "source":
102
+ artist_feat = [a for a in featured if a.source == "artist"]
103
+ title_feat = [a for a in featured if a.source != "artist"]
104
+ elif o.feat == "artist":
105
+ artist_feat = featured
106
+ elif o.feat == "title":
107
+ title_feat = featured
108
+
109
+ # R11.1 / R11.2
110
+ artist_part = _join_names(primaries, o, t)
111
+ if artist_feat:
112
+ lead = _lead_marker(artist_feat[0], o, t, t.feat_canonical)
113
+ names = _join_names(artist_feat, o, t)
114
+ artist_part = f"{artist_part} {lead} {names}" if artist_part else f"{lead} {names}"
115
+
116
+ # R11.3
117
+ pieces: list[str] = []
118
+ if track.title:
119
+ pieces.append(track.title)
120
+ if title_feat:
121
+ lead = _lead_marker(title_feat[0], o, t, t.feat_canonical)
122
+ pieces.append(f"({lead} {_join_names(title_feat, o, t)})")
123
+ if o.versions:
124
+ for v in track.versions:
125
+ pieces.append(render_version(v.raw, v.delimiter))
126
+ if o.producers and producer_list:
127
+ first = producer_list[0]
128
+ marker = first.joiner if first.joiner is not None else "prod."
129
+ pieces.append(f"({marker} {_join_names(producer_list, o, t)})")
130
+ title_part = " ".join(pieces)
131
+
132
+ # R11.4
133
+ result = f"{artist_part} - {title_part}" if artist_part else title_part
134
+ if o.position and track.position is not None:
135
+ result = f"{track.position.raw}. {result}"
136
+ return result
137
+
138
+
139
+ def all_artists(track: ParsedTrack) -> list[Artist]:
140
+ """All credited names: track artists plus every version's remixers, deduped by R0.4 key."""
141
+ seen = set()
142
+ out: list[Artist] = []
143
+ for a in [*track.artists, *(x for v in track.versions for x in v.artists)]:
144
+ key = dedup_key(a.name)
145
+ if key in seen:
146
+ continue
147
+ seen.add(key)
148
+ out.append(a)
149
+ return out