trackparse 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
trackparse/_mode.py ADDED
@@ -0,0 +1,204 @@
1
+ """R2: mode resolution, platform suffixes, filename handling and youtube pipes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import NamedTuple, Optional
6
+
7
+ from ._classify import junk_kind_of
8
+ from ._context import Ctx, warn
9
+ from ._scanner import (
10
+ Segment,
11
+ Skel,
12
+ groups_of,
13
+ has_spaced_dash,
14
+ render_skel,
15
+ scan,
16
+ slice_skel,
17
+ start_pos,
18
+ )
19
+ from ._words import ascii_lower, collapse_spaces, r0_trim, split_words, utf16_len, word_spans
20
+
21
+
22
+ class PositionedJunk(NamedTuple):
23
+ raw: str
24
+ kind: str
25
+ pos: int
26
+
27
+
28
+ class Piece(NamedTuple):
29
+ """A contiguous piece of the normalized string and its absolute offset."""
30
+
31
+ text: str
32
+ base: int
33
+
34
+
35
+ class Prelude(NamedTuple):
36
+ mode: str
37
+ #: The working string (R2.4 step 2/4), or the artist piece in the `artist | title` case.
38
+ working: Piece
39
+ #: R2.4 step 3: the title piece when two dash-less pipe segments remain.
40
+ pipe_title: Optional[Piece]
41
+ junk: list[PositionedJunk]
42
+
43
+
44
+ def _strip_platform_suffixes(s: str, ctx: Ctx, junk: list[PositionedJunk]) -> str:
45
+ """R2.2: strip trailing platform suffixes (longest first, repeatedly)."""
46
+ stripped = True
47
+ while stripped:
48
+ stripped = False
49
+ lower = ascii_lower(s)
50
+ for suffix in ctx.t.platform_suffixes:
51
+ if not lower.endswith(ascii_lower(suffix)):
52
+ continue
53
+ start = len(s) - len(suffix)
54
+ junk.append(PositionedJunk(s[start + 3 :], "platform", utf16_len(s[:start]) + 3))
55
+ s = s[:start]
56
+ stripped = True
57
+ break
58
+ return s
59
+
60
+
61
+ def _extension_of(s: str, ctx: Ctx) -> Optional[str]:
62
+ dot = s.rfind(".")
63
+ if dot < 0:
64
+ return None
65
+ ext = ascii_lower(s[dot + 1 :])
66
+ return ext if ext in ctx.t.extensions else None
67
+
68
+
69
+ def split_pipes(skel: Skel) -> list[Segment]:
70
+ """Top-level `` | `` / `` || `` separators on a skeleton → trimmed segment ranges."""
71
+ t = skel.text
72
+ n = len(t)
73
+ cuts: list[tuple[int, int]] = []
74
+ i = 1
75
+ while i < n:
76
+ if t[i - 1] == " " and t[i] == "|":
77
+ bars = 2 if i + 1 < n and t[i + 1] == "|" else 1
78
+ if i + bars < n and t[i + bars] == " ":
79
+ cuts.append((i - 1, i + bars + 1))
80
+ i += bars
81
+ i += 1
82
+ segments: list[Segment] = []
83
+ start = 0
84
+ for a, b in [*cuts, (n, n)]:
85
+ s = start
86
+ e = a
87
+ while s < e and t[s] == " ":
88
+ s += 1
89
+ while e > s and t[e - 1] == " ":
90
+ e -= 1
91
+ segments.append(Segment(slice_skel(skel, s, e), s, e))
92
+ start = b
93
+ return segments
94
+
95
+
96
+ class TrailingJunk(NamedTuple):
97
+ start: int
98
+ kind: str
99
+
100
+
101
+ def trailing_junk_phrase(text: str, ctx: Ctx) -> Optional[TrailingJunk]:
102
+ """R8.4 test: the skeleton ends with an unbracketed junk phrase (genres excluded)."""
103
+ spans = word_spans(text)
104
+ m = ctx.t.junk_phrases.match_ending([s.text for s in spans], len(spans))
105
+ if m is None:
106
+ return None
107
+ return TrailingJunk(spans[len(spans) - m.length].start, m.entry.value)
108
+
109
+
110
+ def _is_junk_text(text: str, ctx: Ctx) -> Optional[str]:
111
+ words = split_words(text)
112
+ return junk_kind_of(words, ctx) if words else None
113
+
114
+
115
+ def _looks_like_youtube(skel: Skel, ctx: Ctx) -> bool:
116
+ """R2.1 youtube signals (besides platform suffixes)."""
117
+ if len(split_pipes(skel)) > 1:
118
+ return True
119
+ for _, group in groups_of(skel):
120
+ kind = _is_junk_text(group.inner, ctx)
121
+ if kind and kind != "genre":
122
+ return True
123
+ return trailing_junk_phrase(skel.text, ctx) is not None
124
+
125
+
126
+ def prelude(normalized: str, requested: object, ctx: Ctx) -> Prelude:
127
+ junk: list[PositionedJunk] = []
128
+ s = _strip_platform_suffixes(normalized, ctx, junk)
129
+ had_platform = len(junk) > 0
130
+
131
+ if requested in ("clean", "youtube", "filename"):
132
+ mode = str(requested)
133
+ elif _extension_of(s, ctx):
134
+ mode = "filename"
135
+ else:
136
+ mode = "clean"
137
+
138
+ if mode == "filename":
139
+ # R2.3: strip the extension, trim, then `_` → space when there is no space.
140
+ if _extension_of(s, ctx):
141
+ s = s[: s.rfind(".")]
142
+ s = r0_trim(s)
143
+ if " " not in s and "_" in s:
144
+ s = r0_trim(s.replace("_", " "))
145
+
146
+ skel, unbalanced = scan(s, 0)
147
+ if unbalanced:
148
+ warn(ctx, "unbalancedBrackets")
149
+ if (
150
+ (requested == "auto" or requested is None)
151
+ and mode == "clean"
152
+ and (had_platform or _looks_like_youtube(skel, ctx))
153
+ ):
154
+ mode = "youtube"
155
+
156
+ whole = Piece(s, 0)
157
+ if mode != "youtube":
158
+ return Prelude(mode, whole, None, junk)
159
+ working, pipe_title = _split_youtube_pipes(skel, ctx, junk)
160
+ return Prelude(mode, working, pipe_title, junk)
161
+
162
+
163
+ def _piece_of(seg: Segment) -> Piece:
164
+ return Piece(render_skel(seg.skel), start_pos(seg.skel))
165
+
166
+
167
+ def _split_youtube_pipes(
168
+ skel: Skel, ctx: Ctx, junk: list[PositionedJunk]
169
+ ) -> tuple[Piece, Optional[Piece]]:
170
+ """R2.4"""
171
+ segments = split_pipes(skel)
172
+ if len(segments) <= 1:
173
+ return Piece(render_skel(skel), 0), None
174
+
175
+ remaining: list[Segment] = []
176
+ for seg in segments:
177
+ text = collapse_spaces(render_skel(seg.skel))
178
+ # R2.4 step 1: a segment with a spaced dash is an artist/title candidate, never junk.
179
+ kind = None if has_spaced_dash(seg.skel.text) else _is_junk_text(text, ctx)
180
+ if kind:
181
+ junk.append(PositionedJunk(text, kind, start_pos(seg.skel)))
182
+ elif len(text) > 0:
183
+ remaining.append(seg)
184
+
185
+ def push_other(keep: Optional[Segment], title: Optional[Segment] = None) -> None:
186
+ for seg in remaining:
187
+ if seg is keep or seg is title:
188
+ continue
189
+ junk.append(
190
+ PositionedJunk(collapse_spaces(render_skel(seg.skel)), "other", start_pos(seg.skel))
191
+ )
192
+
193
+ with_dash = next((seg for seg in remaining if has_spaced_dash(seg.skel.text)), None)
194
+ if with_dash is not None:
195
+ push_other(with_dash)
196
+ return _piece_of(with_dash), None
197
+ if len(remaining) == 2:
198
+ warn(ctx, "ambiguousSeparator")
199
+ return _piece_of(remaining[0]), _piece_of(remaining[1])
200
+ if remaining:
201
+ push_other(remaining[0])
202
+ return _piece_of(remaining[0]), None
203
+ # Every segment was junk: nothing left to parse.
204
+ return Piece("", 0), None
@@ -0,0 +1,107 @@
1
+ """R1: normalization."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import unicodedata
6
+
7
+ from ._words import is_whitespace
8
+
9
+ # R1.2 invisible characters removed outright. ZWNJ/ZWJ (U+200C/U+200D) are kept (ZWJ emoji).
10
+ # R1.2: U+E000 is reserved as the R3.4 group placeholder, so it can never occur in input.
11
+ _REMOVED = frozenset(["​", "⁠", "", "­", ""])
12
+
13
+ # R1.3 brackets, R1.4 quotes, R1.5 dashes → ASCII.
14
+ _CHAR_MAP: dict[str, str] = {
15
+ "(": "(",
16
+ ")": ")",
17
+ "[": "[",
18
+ "【": "[",
19
+ "〔": "[",
20
+ "]": "]",
21
+ "】": "]",
22
+ "〕": "]",
23
+ "{": "{",
24
+ "}": "}",
25
+ "“": '"',
26
+ "”": '"',
27
+ "„": '"',
28
+ "«": '"',
29
+ "»": '"',
30
+ "「": '"',
31
+ "」": '"',
32
+ "『": '"',
33
+ "』": '"',
34
+ """: '"',
35
+ "‘": "'",
36
+ "’": "'",
37
+ "‚": "'",
38
+ "`": "'",
39
+ "´": "'",
40
+ "−": "-",
41
+ "﹘": "-",
42
+ "﹣": "-",
43
+ "-": "-",
44
+ }
45
+ # R1.5: U+2010–U+2015 → "-".
46
+ for _cp in range(0x2010, 0x2016):
47
+ _CHAR_MAP[chr(_cp)] = "-"
48
+
49
+
50
+ def _collapse_spaced_dash_runs(chars: list[str]) -> list[str]:
51
+ """R1.5 (second half): a run of 2+ ``-`` with whitespace on both sides becomes one ``-``."""
52
+ out: list[str] = []
53
+ i = 0
54
+ n = len(chars)
55
+ while i < n:
56
+ if chars[i] != "-":
57
+ out.append(chars[i])
58
+ i += 1
59
+ continue
60
+ j = i
61
+ while j < n and chars[j] == "-":
62
+ j += 1
63
+ spaced = (
64
+ j - i >= 2
65
+ and i > 0
66
+ and j < n
67
+ and is_whitespace(chars[i - 1])
68
+ and is_whitespace(chars[j])
69
+ )
70
+ if spaced:
71
+ out.append("-")
72
+ else:
73
+ out.extend("-" * (j - i))
74
+ i = j
75
+ return out
76
+
77
+
78
+ def _collapse_whitespace(chars: list[str]) -> str:
79
+ """R1.6: every whitespace code point → one space, runs collapsed, trimmed."""
80
+ out: list[str] = []
81
+ pending = False
82
+ for ch in chars:
83
+ if is_whitespace(ch):
84
+ pending = len(out) > 0
85
+ continue
86
+ if pending:
87
+ out.append(" ")
88
+ pending = False
89
+ out.append(ch)
90
+ return "".join(out)
91
+
92
+
93
+ def normalize(input: str) -> str:
94
+ """R1: normalize a track string. Idempotent; total on any ``str``.
95
+
96
+ Raises ``TypeError`` when ``input`` is not a ``str``.
97
+ """
98
+ if not isinstance(input, str):
99
+ raise TypeError(f"normalize() expects str, got {type(input).__name__}")
100
+ chars: list[str] = []
101
+ for ch in unicodedata.normalize("NFC", input):
102
+ if ch in _REMOVED:
103
+ continue
104
+ chars.append(_CHAR_MAP.get(ch, ch))
105
+ # R1.8: removing invisibles can leave a decomposed sequence behind; re-compose so that
106
+ # normalize() stays idempotent.
107
+ return unicodedata.normalize("NFC", _collapse_whitespace(_collapse_spaced_dash_runs(chars)))
trackparse/_parser.py ADDED
@@ -0,0 +1,257 @@
1
+ """Pipeline: R1 → R2 → R4 → R3/R6 → R7 → R8 → R9, plus R10 parse_artists."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping, Sequence
6
+ from dataclasses import replace
7
+ from typing import Any, Optional, cast
8
+
9
+ from ._artists import parse_artist_side
10
+ from ._assemble import by_pos, dedup_artists, full_title_of, to_version
11
+ from ._context import Ctx, warn
12
+ from ._mode import Piece, prelude
13
+ from ._normalize import normalize
14
+ from ._prefix import strip_prefixes
15
+ from ._scanner import Skel, scan, trim_skel
16
+ from ._separator import find_separator
17
+ from ._tables import Tables, js_entries, tables_for
18
+ from ._title import Extracted, add_peeled, parse_title_side
19
+ from ._words import r0_trim_end
20
+ from .options import KeywordOptions
21
+ from .types import (
22
+ Artist,
23
+ Flags,
24
+ Junk,
25
+ JunkKind,
26
+ Mode,
27
+ ParsedTrack,
28
+ Position,
29
+ Timestamp,
30
+ Warning,
31
+ )
32
+
33
+ #: Offset used for uploader text so it sorts before everything from the input.
34
+ _UPLOADER_BASE = -1_000_000_000
35
+
36
+ _OPTION_KEYS = ("mode", "uploader", "known_artists", "split_and", "keywords")
37
+
38
+
39
+ def _skel_of(piece: Piece, ctx: Ctx) -> Skel:
40
+ r = scan(piece.text, piece.base)
41
+ if r.unbalanced:
42
+ warn(ctx, "unbalancedBrackets")
43
+ return trim_skel(r.skel)
44
+
45
+
46
+ def _uploader_skel(uploader: object) -> Optional[Skel]:
47
+ """R6.8: uploader minus a trailing `` - Topic`` and ``VEVO``."""
48
+ if not isinstance(uploader, str):
49
+ return None
50
+ u = normalize(uploader)
51
+ if u.endswith(" - Topic"):
52
+ u = u[: -len(" - Topic")]
53
+ if u.endswith("VEVO"):
54
+ u = r0_trim_end(u[: -len("VEVO")])
55
+ if len(u) == 0:
56
+ return None
57
+ return scan(u, _UPLOADER_BASE).skel
58
+
59
+
60
+ def _empty_track(inp: str, mode: str) -> ParsedTrack:
61
+ return ParsedTrack(
62
+ input=inp,
63
+ mode=cast(Mode, mode),
64
+ position=None,
65
+ timestamp=None,
66
+ artists=(),
67
+ title="",
68
+ full_title="",
69
+ versions=(),
70
+ year=None,
71
+ flags=Flags(),
72
+ junk=(),
73
+ warnings=(),
74
+ )
75
+
76
+
77
+ def _run(inp: str, options: Mapping[str, Any], t: Tables) -> ParsedTrack:
78
+ ctx = Ctx(t, options)
79
+ norm = normalize(inp)
80
+ requested = options.get("mode")
81
+ if requested is None:
82
+ requested = "auto"
83
+ pre = prelude(norm, requested, ctx)
84
+ if len(norm) == 0:
85
+ return _empty_track(inp, pre.mode) # R9.5
86
+
87
+ # R4
88
+ prefix = strip_prefixes(pre.working.text, pre.working.base, pre.mode, ctx)
89
+ working = _skel_of(Piece(prefix.rest, prefix.base), ctx)
90
+
91
+ out = Extracted()
92
+ artist_skel: Optional[Skel]
93
+ if pre.pipe_title is not None:
94
+ # R2.4 step 3: `artist | title`, R6 skipped.
95
+ artist_skel = working if len(working.text) > 0 else None
96
+ title_skel = _skel_of(pre.pipe_title, ctx)
97
+ relaxed = artist_skel is not None
98
+ else:
99
+ sep = find_separator(working, pre.mode, _uploader_skel(options.get("uploader")), ctx)
100
+ add_peeled(out, sep.peeled)
101
+ artist_skel = sep.artist
102
+ title_skel = sep.title
103
+ relaxed = sep.split
104
+
105
+ artist_side = parse_artist_side(artist_skel, ctx) if artist_skel is not None else None
106
+ title_side = parse_title_side(title_skel, relaxed, pre.mode == "youtube", out, ctx)
107
+
108
+ # R9
109
+ versions = by_pos(out.versions)
110
+ for v in versions:
111
+ if v.ambiguous_mix:
112
+ warn(ctx, "ambiguousMixCredit")
113
+ artist_credits = artist_side.credits if artist_side is not None else []
114
+ artists = dedup_artists([*artist_credits, *by_pos(out.credits)])
115
+ final_versions = tuple(to_version(v) for v in versions)
116
+ # R9.6: the first year in text order.
117
+ years = by_pos([*(artist_side.years if artist_side is not None else []), *out.years])
118
+ junk = by_pos([*pre.junk, *(artist_side.junk if artist_side is not None else []), *out.junk])
119
+ flags = Flags(
120
+ explicit=out.explicit or (artist_side is not None and artist_side.explicit),
121
+ clean=out.clean or (artist_side is not None and artist_side.clean),
122
+ unknown_artist=(artist_side is not None and artist_side.unknown_artist)
123
+ or out.unknown_artist,
124
+ unknown_title=title_side.unknown_title,
125
+ )
126
+ ts = prefix.timestamp
127
+ pos = prefix.position
128
+ return ParsedTrack(
129
+ input=inp,
130
+ mode=cast(Mode, pre.mode),
131
+ position=None if pos is None else Position(raw=pos[0], number=pos[1]),
132
+ timestamp=None if ts is None else Timestamp(raw=ts[0], seconds=ts[1]),
133
+ artists=tuple(artists),
134
+ title=title_side.title,
135
+ full_title=full_title_of(title_side.title, final_versions),
136
+ versions=final_versions,
137
+ year=years[0].year if years else None,
138
+ flags=flags,
139
+ junk=tuple(Junk(raw=j.raw, kind=cast(JunkKind, j.kind)) for j in junk),
140
+ warnings=cast(tuple[Warning, ...], tuple(ctx.warnings)),
141
+ )
142
+
143
+
144
+ def _run_artists(inp: str, options: Mapping[str, Any], t: Tables) -> list[Artist]:
145
+ ctx = Ctx(t, options)
146
+ norm = normalize(inp)
147
+ if len(norm) == 0:
148
+ return []
149
+ side = parse_artist_side(scan(norm, 0).skel, ctx)
150
+ return [replace(a, source="artist") for a in dedup_artists(side.credits)]
151
+
152
+
153
+ def _check_input(inp: object, fn: str) -> str:
154
+ if not isinstance(inp, str):
155
+ raise TypeError(f"{fn}() expects str input, got {type(inp).__name__}")
156
+ return inp
157
+
158
+
159
+ def _safe_options(options: Mapping[str, Any]) -> dict[str, Any]:
160
+ # A `None` per-call value must not override a create_parser() base option.
161
+ return {k: v for k, v in options.items() if k in _OPTION_KEYS and v is not None}
162
+
163
+
164
+ def _merge_keywords(a: Mapping[str, Any], b: Mapping[str, Any]) -> dict[str, Any]:
165
+ """Like the JS mergeKeywords: maps merged (later wins), lists concatenated."""
166
+
167
+ def kw(o: Mapping[str, Any]) -> Mapping[str, Any]:
168
+ k = o.get("keywords")
169
+ return k if isinstance(k, Mapping) else {}
170
+
171
+ def as_map(v: Any) -> dict[str, Any]:
172
+ return dict(js_entries(v)) if isinstance(v, Mapping) else {}
173
+
174
+ def as_list(v: Any) -> list[Any]:
175
+ return list(v) if isinstance(v, (list, tuple)) else []
176
+
177
+ x = kw(a)
178
+ y = kw(b)
179
+ return {
180
+ "version_heads": {**as_map(x.get("version_heads")), **as_map(y.get("version_heads"))},
181
+ "descriptors": [*as_list(x.get("descriptors")), *as_list(y.get("descriptors"))],
182
+ "genres": [*as_list(x.get("genres")), *as_list(y.get("genres"))],
183
+ "junk": {**as_map(x.get("junk")), **as_map(y.get("junk"))},
184
+ "feat_markers": [*as_list(x.get("feat_markers")), *as_list(y.get("feat_markers"))],
185
+ }
186
+
187
+
188
+ class Parser:
189
+ """A parser with ``keywords`` compiled once. Per-call options are merged over the base
190
+ options; a per-call ``None`` does not override a base value."""
191
+
192
+ def __init__(self, **options: Any) -> None:
193
+ self._base = _safe_options(options)
194
+ self._tables = tables_for(self._base.get("keywords"))
195
+
196
+ def _resolve(self, call: Mapping[str, Any]) -> tuple[dict[str, Any], Tables]:
197
+ per_call = _safe_options(call)
198
+ merged = {**self._base, **per_call}
199
+ if per_call.get("keywords"):
200
+ return merged, tables_for(_merge_keywords(self._base, per_call))
201
+ return merged, self._tables
202
+
203
+ def parse(
204
+ self,
205
+ input: str,
206
+ *,
207
+ mode: Optional[str] = None,
208
+ uploader: Optional[str] = None,
209
+ known_artists: Optional[Sequence[str]] = None,
210
+ split_and: Optional[str] = None,
211
+ keywords: Optional[KeywordOptions] = None,
212
+ ) -> ParsedTrack:
213
+ """Parse one track string (see :func:`trackparse.parse`)."""
214
+ inp = _check_input(input, "parse")
215
+ merged, tables = self._resolve(
216
+ {
217
+ "mode": mode,
218
+ "uploader": uploader,
219
+ "known_artists": known_artists,
220
+ "split_and": split_and,
221
+ "keywords": keywords,
222
+ }
223
+ )
224
+ return _run(inp, merged, tables)
225
+
226
+ def parse_artists(
227
+ self,
228
+ input: str,
229
+ *,
230
+ mode: Optional[str] = None,
231
+ uploader: Optional[str] = None,
232
+ known_artists: Optional[Sequence[str]] = None,
233
+ split_and: Optional[str] = None,
234
+ keywords: Optional[KeywordOptions] = None,
235
+ ) -> list[Artist]:
236
+ """Split an artist string into credits (R10)."""
237
+ inp = _check_input(input, "parse_artists")
238
+ merged, tables = self._resolve(
239
+ {
240
+ "mode": mode,
241
+ "uploader": uploader,
242
+ "known_artists": known_artists,
243
+ "split_and": split_and,
244
+ "keywords": keywords,
245
+ }
246
+ )
247
+ return _run_artists(inp, merged, tables)
248
+
249
+
250
+ _default_parser: Optional[Parser] = None
251
+
252
+
253
+ def default_parser() -> Parser:
254
+ global _default_parser
255
+ if _default_parser is None:
256
+ _default_parser = Parser()
257
+ return _default_parser