trackparse 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
trackparse/_tables.py ADDED
@@ -0,0 +1,228 @@
1
+ """Vocabulary tables compiled from spec/data, optionally extended by ``keywords`` (R12)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from typing import Any, Optional
7
+
8
+ from ._data import (
9
+ DESCRIPTORS_DATA,
10
+ EXTENSIONS_DATA,
11
+ FEAT_MARKERS_DATA,
12
+ GENRES_DATA,
13
+ JOINERS_DATA,
14
+ JUNK_DATA,
15
+ NO_SPLIT_BEFORE_DATA,
16
+ PLATFORM_SUFFIXES_DATA,
17
+ STOPWORDS_DATA,
18
+ UNKNOWN_TOKENS_DATA,
19
+ VERSION_KEYWORDS_DATA,
20
+ )
21
+ from ._words import PhraseTable, ascii_lower, js_trim
22
+
23
+
24
+ class SpacedJoiner:
25
+ __slots__ = ("canonical", "case_sensitive", "is_and", "raw")
26
+
27
+ def __init__(self, raw: str, canonical: str, is_and: bool, case_sensitive: bool) -> None:
28
+ self.raw = raw
29
+ self.canonical = canonical
30
+ self.is_and = is_and
31
+ self.case_sensitive = case_sensitive
32
+
33
+
34
+ class RunInfo:
35
+ """Category of a word in the modifier run left of a version head (R5.7.2)."""
36
+
37
+ __slots__ = ("descriptor", "genre", "head", "type_capable")
38
+
39
+ def __init__(
40
+ self,
41
+ type_capable: Optional[str] = None,
42
+ descriptor: Optional[str] = None,
43
+ genre: bool = False,
44
+ head: Optional[str] = None,
45
+ ) -> None:
46
+ self.type_capable = type_capable
47
+ self.descriptor = descriptor
48
+ self.genre = genre
49
+ self.head = head
50
+
51
+
52
+ class Tables:
53
+ __slots__ = (
54
+ "bracket_only_feat",
55
+ "extensions",
56
+ "feat_canonical",
57
+ "feat_markers",
58
+ "generic_head_types",
59
+ "heads",
60
+ "junk_connectors",
61
+ "junk_or_genre",
62
+ "junk_phrases",
63
+ "label_suffixes",
64
+ "no_split_before",
65
+ "platform_suffixes",
66
+ "prefix_forms",
67
+ "producer_markers",
68
+ "run_words",
69
+ "spaced_joiners",
70
+ "stopwords",
71
+ "tight_joiners",
72
+ "title_feat_markers",
73
+ "title_producer_markers",
74
+ "unknown_case_insensitive",
75
+ "unknown_case_sensitive",
76
+ )
77
+
78
+ feat_markers: PhraseTable[bool]
79
+ bracket_only_feat: PhraseTable[bool]
80
+ producer_markers: PhraseTable[bool]
81
+ #: R8.3: the only unbracketed feat markers on the title side (+ keywords feat_markers).
82
+ title_feat_markers: PhraseTable[bool]
83
+ #: R8.3: the only unbracketed producer markers on the title side.
84
+ title_producer_markers: PhraseTable[bool]
85
+ #: Junk phrases plus genres (kind ``genre``): the R5.1 tokenizer vocabulary.
86
+ junk_or_genre: PhraseTable[str]
87
+ #: Junk phrases only (R8.4).
88
+ junk_phrases: PhraseTable[str]
89
+ junk_connectors: frozenset[str]
90
+ label_suffixes: frozenset[str]
91
+ heads: PhraseTable[str]
92
+ generic_head_types: frozenset[str]
93
+ #: Everything allowed in the R5.7.2 modifier run, keyed by phrase.
94
+ run_words: PhraseTable[RunInfo]
95
+ prefix_forms: PhraseTable[str]
96
+ spaced_joiners: list[SpacedJoiner]
97
+ tight_joiners: dict[str, str]
98
+ feat_canonical: str
99
+ no_split_before: frozenset[str]
100
+ stopwords: frozenset[str]
101
+ unknown_case_sensitive: frozenset[str]
102
+ unknown_case_insensitive: frozenset[str]
103
+ extensions: frozenset[str]
104
+ platform_suffixes: list[str]
105
+
106
+
107
+ def _is_array_index(key: str) -> bool:
108
+ if not key or len(key) > 10 or not all("0" <= c <= "9" for c in key):
109
+ return False
110
+ if len(key) > 1 and key[0] == "0":
111
+ return False
112
+ return int(key) <= 0xFFFFFFFE
113
+
114
+
115
+ def js_entries(d: Mapping[str, Any]) -> list[tuple[str, Any]]:
116
+ """``Object.entries`` order: integer-like keys first (ascending), then insertion order."""
117
+ items = list(d.items())
118
+ ints = sorted((kv for kv in items if _is_array_index(kv[0])), key=lambda kv: int(kv[0]))
119
+ return ints + [kv for kv in items if not _is_array_index(kv[0])]
120
+
121
+
122
+ def _str_list(value: Any) -> list[str]:
123
+ if not isinstance(value, (list, tuple)):
124
+ return []
125
+ return [s for s in value if isinstance(s, str)]
126
+
127
+
128
+ def _str_map(value: Any) -> list[tuple[str, str]]:
129
+ if not isinstance(value, Mapping):
130
+ return []
131
+ return [(k, v) for k, v in js_entries(value) if isinstance(k, str) and isinstance(v, str)]
132
+
133
+
134
+ def _lower_keys(value: Any) -> list[str]:
135
+ return [k for k in (js_trim(ascii_lower(s)) for s in _str_list(value)) if len(k) > 0]
136
+
137
+
138
+ def build_tables(keywords: Optional[Mapping[str, Any]] = None) -> Tables:
139
+ kw: Mapping[str, Any] = keywords if isinstance(keywords, Mapping) else {}
140
+ t = Tables()
141
+ extra_feat = _lower_keys(kw.get("feat_markers"))
142
+ t.feat_markers = PhraseTable((m, True) for m in [*FEAT_MARKERS_DATA["markers"], *extra_feat])
143
+ t.bracket_only_feat = PhraseTable((m, True) for m in FEAT_MARKERS_DATA["bracketOnly"])
144
+ t.producer_markers = PhraseTable((m, True) for m in FEAT_MARKERS_DATA["producer"])
145
+ t.title_feat_markers = PhraseTable(
146
+ (m, True) for m in [*FEAT_MARKERS_DATA["titleMarkers"], *extra_feat]
147
+ )
148
+ t.title_producer_markers = PhraseTable((m, True) for m in FEAT_MARKERS_DATA["titleProducer"])
149
+
150
+ junk_entries: list[tuple[str, str]] = [
151
+ *js_entries(JUNK_DATA["phrases"]),
152
+ *_str_map(kw.get("junk")),
153
+ ]
154
+ genres: list[str] = [*GENRES_DATA["genres"], *_lower_keys(kw.get("genres"))]
155
+ t.junk_phrases = PhraseTable(junk_entries)
156
+ # Junk phrases are added first, so they win over an identical genre key.
157
+ t.junk_or_genre = PhraseTable([*junk_entries, *((g, "genre") for g in genres)])
158
+
159
+ head_entries: list[tuple[str, str]] = [
160
+ *js_entries(VERSION_KEYWORDS_DATA["heads"]),
161
+ *_str_map(kw.get("version_heads")),
162
+ ]
163
+ t.heads = PhraseTable(head_entries)
164
+
165
+ run_info: dict[str, RunInfo] = {}
166
+
167
+ def note(key: str, **patch: Any) -> None:
168
+ k = js_trim(ascii_lower(key))
169
+ info = run_info.get(k)
170
+ if info is None:
171
+ info = RunInfo()
172
+ merged = RunInfo(info.type_capable, info.descriptor, info.genre, info.head)
173
+ for name, value in patch.items():
174
+ setattr(merged, name, value)
175
+ run_info[k] = merged
176
+
177
+ for k, typ in js_entries(VERSION_KEYWORDS_DATA["typeCapableModifiers"]):
178
+ note(k, type_capable=typ)
179
+ for d in [*DESCRIPTORS_DATA["descriptors"], *_lower_keys(kw.get("descriptors"))]:
180
+ note(d, descriptor=js_trim(ascii_lower(d)))
181
+ for g in genres:
182
+ note(g, genre=True)
183
+ for k, typ in head_entries:
184
+ note(k, head=typ)
185
+
186
+ t.junk_connectors = frozenset(JUNK_DATA["connectors"])
187
+ t.label_suffixes = frozenset(JUNK_DATA["labelSuffixes"])
188
+ t.generic_head_types = frozenset(VERSION_KEYWORDS_DATA["genericHeads"])
189
+ t.run_words = PhraseTable(run_info.items())
190
+ t.prefix_forms = PhraseTable(js_entries(VERSION_KEYWORDS_DATA["prefixForms"]))
191
+ t.spaced_joiners = [
192
+ SpacedJoiner(
193
+ j["raw"],
194
+ j["canonical"],
195
+ j.get("kind") == "and",
196
+ j.get("caseSensitive") is True,
197
+ )
198
+ for j in JOINERS_DATA["spaced"]
199
+ ]
200
+ t.tight_joiners = {j["raw"]: j["canonical"] for j in JOINERS_DATA["tight"]}
201
+ t.feat_canonical = JOINERS_DATA["featCanonical"]
202
+ t.no_split_before = frozenset(NO_SPLIT_BEFORE_DATA["words"])
203
+ t.stopwords = frozenset(STOPWORDS_DATA["words"])
204
+ t.unknown_case_sensitive = frozenset(UNKNOWN_TOKENS_DATA["caseSensitive"])
205
+ t.unknown_case_insensitive = frozenset(UNKNOWN_TOKENS_DATA["caseInsensitive"])
206
+ t.extensions = frozenset(EXTENSIONS_DATA["extensions"])
207
+ # JS sorts by UTF-16 length; the suffixes are ASCII, so code-point length is the same.
208
+ t.platform_suffixes = sorted(PLATFORM_SUFFIXES_DATA["suffixes"], key=lambda s: -len(s))
209
+ return t
210
+
211
+
212
+ _default_tables: Optional[Tables] = None
213
+
214
+
215
+ def get_default_tables() -> Tables:
216
+ global _default_tables
217
+ if _default_tables is None:
218
+ _default_tables = build_tables()
219
+ return _default_tables
220
+
221
+
222
+ def tables_for(keywords: Optional[Mapping[str, Any]]) -> Tables:
223
+ return build_tables(keywords) if keywords else get_default_tables()
224
+
225
+
226
+ def is_unknown_token(text: str, t: Tables) -> bool:
227
+ """R7.5 / R8.6: an unknown placeholder name (``ID``, ``???``, ``Untitled``…)."""
228
+ return text in t.unknown_case_sensitive or ascii_lower(text) in t.unknown_case_insensitive
trackparse/_title.py ADDED
@@ -0,0 +1,246 @@
1
+ """R8: the title side."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import NamedTuple
6
+
7
+ from ._classify import Feat, Flag, Junk, Producer, Site, VersionDraft, Versions, Year, classify
8
+ from ._context import Ctx
9
+ from ._credits import Credit, ListOptions, PositionedYear, split_credits
10
+ from ._mode import PositionedJunk
11
+ from ._scanner import (
12
+ PH,
13
+ Skel,
14
+ cut_skel,
15
+ groups_of,
16
+ remove_groups,
17
+ render_skel,
18
+ slice_skel,
19
+ split_at_spaced_dashes,
20
+ trim_skel,
21
+ )
22
+ from ._separator import Peeled, peel_suffixes
23
+ from ._tables import is_unknown_token
24
+ from ._words import char_at, collapse_spaces, word_spans
25
+
26
+
27
+ class Extracted:
28
+ """Everything the title side (and R6.2 peeling) can extract."""
29
+
30
+ __slots__ = ("clean", "credits", "explicit", "junk", "unknown_artist", "versions", "years")
31
+
32
+ def __init__(self) -> None:
33
+ self.credits: list[Credit] = []
34
+ self.versions: list[VersionDraft] = []
35
+ self.junk: list[PositionedJunk] = []
36
+ self.years: list[PositionedYear] = []
37
+ self.explicit = False
38
+ self.clean = False
39
+ self.unknown_artist = False
40
+
41
+ def set_flag(self, flag: str) -> None:
42
+ if flag == "explicit":
43
+ self.explicit = True
44
+ else:
45
+ self.clean = True
46
+
47
+
48
+ def add_peeled(out: Extracted, peeled: list[Peeled]) -> None:
49
+ """Record R6.2 / R8.2 peeled dash suffixes."""
50
+ for p in peeled:
51
+ c = p.cls
52
+ if isinstance(c, Versions):
53
+ out.versions.extend(c.versions)
54
+ elif isinstance(c, Junk):
55
+ out.junk.append(PositionedJunk(p.raw, c.junk_kind, p.pos))
56
+ elif isinstance(c, Flag):
57
+ out.set_flag(c.flag)
58
+ else:
59
+ out.years.append(PositionedYear(c.year, p.pos))
60
+
61
+
62
+ class TitleSideResult(NamedTuple):
63
+ title: str
64
+ unknown_title: bool
65
+
66
+
67
+ def parse_title_side(
68
+ skel: Skel, relaxed: bool, youtube: bool, out: Extracted, ctx: Ctx
69
+ ) -> TitleSideResult:
70
+ """``relaxed``: an artist side was split off (R8.2 peels without conditions)."""
71
+ removed = _extract_groups(skel, out, ctx) # R8.1
72
+
73
+ # R8.2
74
+ segments = split_at_spaced_dashes(skel)
75
+ remaining, peeled = peel_suffixes(segments, relaxed, ctx)
76
+ add_peeled(out, peeled)
77
+ skel = slice_skel(skel, 0, remaining[-1].end)
78
+
79
+ skel = _extract_unbracketed_credits(skel, out, ctx) # R8.3
80
+ skel = remove_groups(skel, removed)
81
+ if youtube:
82
+ skel = _strip_trailing_junk(skel, out, ctx) # R8.4
83
+
84
+ title = cleanup_title(render_skel(skel)) # R8.5
85
+ return TitleSideResult(title, len(title) > 0 and is_unknown_token(title, ctx.t)) # R8.6
86
+
87
+
88
+ def _extract_groups(skel: Skel, out: Extracted, ctx: Ctx) -> set[int]:
89
+ """R8.1: classify every group; returns the positions of the groups to remove."""
90
+ removed: set[int] = set()
91
+ for _, group in groups_of(skel):
92
+ c = classify(group.inner, Site("title", group.open, group.pos), ctx)
93
+ if isinstance(c, Versions):
94
+ out.versions.extend(c.versions)
95
+ elif isinstance(c, Junk):
96
+ out.junk.append(PositionedJunk(group.inner, c.junk_kind, group.pos))
97
+ elif isinstance(c, Flag):
98
+ out.set_flag(c.flag)
99
+ elif isinstance(c, Year):
100
+ out.years.append(PositionedYear(c.year, group.pos))
101
+ elif isinstance(c, (Feat, Producer)):
102
+ out.credits.extend(c.credits)
103
+ out.years.extend(c.years)
104
+ if c.unknown:
105
+ out.unknown_artist = True
106
+ else:
107
+ continue
108
+ removed.add(group.pos)
109
+ return removed
110
+
111
+
112
+ class _MarkerHit(NamedTuple):
113
+ start: int
114
+ end: int
115
+ marker: str
116
+ role: str # "featured" | "producer"
117
+
118
+
119
+ def _title_markers(s: Skel, ctx: Ctx) -> list[_MarkerHit]:
120
+ """R8.3 markers, left to right, each with a space on both sides: feat markers from
121
+ ``#titleMarkers`` (``feat.``, ``ft.``, ``featuring``; undotted ``feat``/``ft`` stay text)
122
+ and producer markers from ``#titleProducer``."""
123
+ spans = word_spans(s.text)
124
+ words = [w.text for w in spans]
125
+ hits: list[_MarkerHit] = []
126
+ i = 0
127
+ while i < len(spans):
128
+ span = spans[i]
129
+ if char_at(s.text, span.start - 1) != " ":
130
+ i += 1
131
+ continue
132
+ role = "featured"
133
+ feat = ctx.t.title_feat_markers.match_at(words, i)
134
+ if feat is not None:
135
+ last_index = i + feat.length - 1
136
+ else:
137
+ prod = ctx.t.title_producer_markers.match_at(words, i)
138
+ if prod is None:
139
+ i += 1
140
+ continue
141
+ role = "producer"
142
+ last_index = i + prod.length - 1
143
+ last = spans[last_index]
144
+ if char_at(s.text, last.end) != " ":
145
+ i += 1
146
+ continue
147
+ hits.append(_MarkerHit(span.start, last.end, s.text[span.start : last.end], role))
148
+ i = last_index + 1
149
+ return hits
150
+
151
+
152
+ def _extract_unbracketed_credits(inp: Skel, out: Extracted, ctx: Ctx) -> Skel:
153
+ """R8.3: ``Title feat. X``, ``Title prod. by Y`` → credits, removed from the title. A list
154
+ runs to the next placeholder, the next marker of the other kind, or the end; later feat
155
+ markers inside a featured list are joiners (R7.2)."""
156
+ markers = _title_markers(inp, ctx)
157
+ cuts: list[tuple[int, int]] = []
158
+ i = 0
159
+ while i < len(markers):
160
+ hit = markers[i]
161
+ list_end = inp.text.find(PH, hit.end)
162
+ if list_end < 0:
163
+ list_end = len(inp.text)
164
+ other = next(
165
+ (
166
+ m
167
+ for k, m in enumerate(markers)
168
+ if k > i and m.role != hit.role and m.start >= hit.end
169
+ ),
170
+ None,
171
+ )
172
+ if other is not None and other.start < list_end:
173
+ list_end = other.start
174
+ lst = trim_skel(slice_skel(inp, hit.end, list_end))
175
+ if len(lst.text) > 0:
176
+ r = split_credits(
177
+ lst,
178
+ ListOptions(hit.role, "title", hit.marker, False, hit.role == "featured"),
179
+ ctx,
180
+ )
181
+ out.credits.extend(r.credits)
182
+ out.years.extend(r.years)
183
+ if r.unknown:
184
+ out.unknown_artist = True
185
+ cuts.append((hit.start, list_end))
186
+ else:
187
+ list_end = hit.end
188
+ while i < len(markers) and markers[i].start < list_end:
189
+ i += 1
190
+ skel = inp
191
+ for start, end in reversed(cuts):
192
+ skel = cut_skel(skel, start, end)
193
+ return skel
194
+
195
+
196
+ def _strip_trailing_junk(inp: Skel, out: Extracted, ctx: Ctx) -> Skel:
197
+ """R8.4: strip trailing unbracketed junk phrases (optionally after a lone ``-`` / ``|``).
198
+
199
+ Word spans are computed once and walked backwards, so repeated junk stays linear (R0.8).
200
+ """
201
+ skel = trim_skel(inp)
202
+ spans = word_spans(skel.text)
203
+ words = [span.text for span in spans]
204
+ end = len(spans) # words still in the title
205
+ while True:
206
+ m = ctx.t.junk_phrases.match_ending(words, end)
207
+ if m is None:
208
+ break
209
+ first = end - m.length
210
+ keep = first
211
+ if keep > 0 and _is_lone_edge(words[keep - 1]):
212
+ keep -= 1
213
+ if keep == 0:
214
+ break # never strip the whole title
215
+ start = spans[first].start
216
+ raw = collapse_spaces(skel.text[start : spans[end - 1].end])
217
+ pos = skel.pos[start] if start < len(skel.pos) else 0
218
+ out.junk.append(PositionedJunk(raw, m.entry.value, pos))
219
+ end = keep
220
+ if end == len(spans):
221
+ return skel
222
+ return trim_skel(slice_skel(skel, 0, spans[end - 1].end))
223
+
224
+
225
+ def _is_lone_edge(word: str) -> bool:
226
+ return word == "-" or word == "|"
227
+
228
+
229
+ def _strip_lone_edges(s: str) -> str:
230
+ words = s.split(" ")
231
+ a = 0
232
+ b = len(words)
233
+ while a < b and _is_lone_edge(words[a]):
234
+ a += 1
235
+ while b > a and _is_lone_edge(words[b - 1]):
236
+ b -= 1
237
+ return " ".join(words[a:b])
238
+
239
+
240
+ def cleanup_title(raw: str) -> str:
241
+ """R8.5"""
242
+ title = _strip_lone_edges(collapse_spaces(raw))
243
+ q = title[:1]
244
+ if len(title) >= 2 and (q == '"' or q == "'") and title[-1] == q:
245
+ title = _strip_lone_edges(collapse_spaces(title[1:-1]))
246
+ return title