trackparse 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
trackparse/_prefix.py ADDED
@@ -0,0 +1,143 @@
1
+ """R4: leading timestamp and track-position prefixes."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import NamedTuple, Optional
6
+
7
+ from ._context import Ctx, match_known_at
8
+ from ._scanner import has_spaced_dash, scan
9
+ from ._words import char_at, is_digit
10
+
11
+
12
+ class PrefixResult(NamedTuple):
13
+ rest: str
14
+ #: Absolute offset of ``rest`` (prefixes are ASCII, so code points == UTF-16 units).
15
+ base: int
16
+ timestamp: Optional[tuple[str, int]]
17
+ position: Optional[tuple[str, int]]
18
+
19
+
20
+ def _digit_run(s: str, frm: int) -> int:
21
+ i = frm
22
+ while is_digit(char_at(s, i)):
23
+ i += 1
24
+ return i - frm
25
+
26
+
27
+ def _two_digits_up_to_59(s: str, at: int) -> Optional[int]:
28
+ if (
29
+ not is_digit(char_at(s, at))
30
+ or not is_digit(char_at(s, at + 1))
31
+ or is_digit(char_at(s, at + 2))
32
+ ):
33
+ return None
34
+ n = int(s[at : at + 2])
35
+ return n if n <= 59 else None
36
+
37
+
38
+ class _TimeMatch(NamedTuple):
39
+ raw: str
40
+ seconds: int
41
+ end: int
42
+
43
+
44
+ def _match_time(s: str, at: int) -> Optional[_TimeMatch]:
45
+ """``H:MM:SS``, ``HH:MM:SS``, ``M:SS``, ``MM:SS`` starting at ``at``."""
46
+ lead = _digit_run(s, at)
47
+ if lead < 1 or lead > 2 or char_at(s, at + lead) != ":":
48
+ return None
49
+ first = int(s[at : at + lead])
50
+ mid = _two_digits_up_to_59(s, at + lead + 1)
51
+ if mid is None:
52
+ return None
53
+ after_mid = at + lead + 3
54
+ if char_at(s, after_mid) == ":":
55
+ sec = _two_digits_up_to_59(s, after_mid + 1)
56
+ if sec is not None:
57
+ end = after_mid + 3
58
+ return _TimeMatch(s[at:end], first * 3600 + mid * 60 + sec, end)
59
+ if lead == 2 and first > 59:
60
+ return None # MM is 00–59
61
+ return _TimeMatch(s[at:after_mid], first * 60 + mid, after_mid)
62
+
63
+
64
+ def _skip_space_and_dash(s: str, at: int) -> int:
65
+ """After a prefix: skip one space, then an optional ``- `` (spaced dash)."""
66
+ i = at
67
+ if char_at(s, i) == " ":
68
+ i += 1
69
+ if char_at(s, i) == "-" and char_at(s, i + 1) == " ":
70
+ i += 2
71
+ return i
72
+
73
+
74
+ def _match_timestamp(s: str) -> Optional[tuple[tuple[str, int], int]]:
75
+ """R4.1"""
76
+ first = char_at(s, 0)
77
+ bracket = "]" if first == "[" else ")" if first == "(" else None
78
+ t = _match_time(s, 1 if bracket else 0)
79
+ if t is None:
80
+ return None
81
+ end = t.end
82
+ if bracket:
83
+ if char_at(s, end) != bracket:
84
+ return None
85
+ end += 1
86
+ if end < len(s) and s[end] != " ":
87
+ return None
88
+ return (t.raw, t.seconds), _skip_space_and_dash(s, end)
89
+
90
+
91
+ def _contains_spaced_dash(s: str) -> bool:
92
+ return has_spaced_dash(scan(s).skel.text)
93
+
94
+
95
+ def _match_position(s: str, mode: str) -> Optional[int]:
96
+ """R4.2: end offset of the position prefix, or None. Forms (a), (b), (c), (d)/(e)."""
97
+ n = _digit_run(s, 0)
98
+ if n < 1 or n > 3:
99
+ return None
100
+ c = char_at(s, n)
101
+ zero_padded = s[0] == "0" and n >= 2
102
+ # (a) `01. `, `1) `
103
+ if (c == "." or c == ")") and (char_at(s, n + 1) == " " or n + 1 == len(s)):
104
+ return _skip_space_and_dash(s, n + 1)
105
+ # (b) `01 `: a position in every mode.
106
+ if zero_padded and c == " ":
107
+ return _skip_space_and_dash(s, n)
108
+ # (c) `3 - Artist - Title`; in filename mode `N - ` is always a position (`311 - Amber.mp3`).
109
+ if c == " " and char_at(s, n + 1) == "-" and char_at(s, n + 2) == " ":
110
+ if mode == "filename" or _contains_spaced_dash(s[n + 3 :]):
111
+ return n + 3
112
+ return None
113
+ if mode == "filename":
114
+ # (e) `01-Track`, `03_name`, `01.Track`
115
+ nxt = char_at(s, n + 1)
116
+ if (c == "-" or c == "_" or c == ".") and nxt is not None and nxt != " ":
117
+ return n + 1
118
+ # (e) `3 My Song.mp3`: only when the remainder has no spaced dash.
119
+ if c == " " and not _contains_spaced_dash(s[n + 1 :]):
120
+ return _skip_space_and_dash(s, n)
121
+ # (d) `2 Unlimited`, `50 Cent`: never a position outside filename mode.
122
+ return None
123
+
124
+
125
+ def strip_prefixes(s: str, base: int, mode: str, ctx: Ctx) -> PrefixResult:
126
+ rest = s
127
+ offset = base
128
+ timestamp: Optional[tuple[str, int]] = None
129
+ position: Optional[tuple[str, int]] = None
130
+
131
+ ts = _match_timestamp(rest)
132
+ if ts is not None:
133
+ timestamp = ts[0]
134
+ rest = rest[ts[1] :]
135
+ offset += ts[1]
136
+ if match_known_at(rest, 0, ctx, "-") == 0:
137
+ end = _match_position(rest, mode)
138
+ if end is not None and end < len(rest):
139
+ raw = rest[: _digit_run(rest, 0)]
140
+ position = (raw, int(raw))
141
+ rest = rest[end:]
142
+ offset += end
143
+ return PrefixResult(rest, offset, timestamp, position)
trackparse/_scanner.py ADDED
@@ -0,0 +1,240 @@
1
+ """R3: top-level group scanner and skeleton.
2
+
3
+ A ``Skel`` is a skeleton string plus, for every code point, its absolute offset in the
4
+ normalized input. Groups are looked up by the offset of their placeholder, so any slice of a
5
+ skeleton still knows which groups it contains. Offsets also give every extracted item (junk,
6
+ versions, credits) its position for the "order of appearance" rules.
7
+
8
+ Offsets are counted in UTF-16 units (an astral code point advances by 2) so that every
9
+ ordering comparison is identical to the JS reference; indices into ``text`` are code points.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from typing import NamedTuple, Optional
15
+
16
+ from ._words import char_at, collapse_spaces, is_whitespace
17
+
18
+ #: R3.4 placeholder code point.
19
+ PH = ""
20
+
21
+ _CLOSER = {"(": ")", "[": "]", "{": "}"}
22
+
23
+
24
+ class Group:
25
+ __slots__ = ("closed", "end", "inner", "open", "pos", "raw")
26
+
27
+ def __init__(self, open: str, inner: str, raw: str, closed: bool, pos: int, end: int) -> None:
28
+ self.open = open
29
+ #: R3.5: inner text, trimmed and whitespace-collapsed.
30
+ self.inner = inner
31
+ #: The group exactly as written, brackets included.
32
+ self.raw = raw
33
+ self.closed = closed
34
+ #: Absolute offset of the opener.
35
+ self.pos = pos
36
+ #: Absolute offset just past the group.
37
+ self.end = end
38
+
39
+
40
+ class Skel:
41
+ __slots__ = ("groups", "pos", "text")
42
+
43
+ def __init__(self, text: str, pos: list[int], groups: dict[int, Group]) -> None:
44
+ self.text = text
45
+ self.pos = pos
46
+ self.groups = groups
47
+
48
+
49
+ class ScanResult(NamedTuple):
50
+ skel: Skel
51
+ unbalanced: bool
52
+
53
+
54
+ def scan(s: str, base: int = 0) -> ScanResult:
55
+ """R3.1–R3.5: scan ``s`` (whose first unit sits at absolute offset ``base``)."""
56
+ text: list[str] = []
57
+ pos: list[int] = []
58
+ groups: dict[int, Group] = {}
59
+ stack: list[str] = []
60
+ group_start = -1
61
+ group_off = 0
62
+ # Absolute UTF-16 offset of each code point of `s` (plus the end), for group bookkeeping.
63
+ offs: list[int] = []
64
+ off = base
65
+ for ch in s:
66
+ offs.append(off)
67
+ off += 2 if ch > "￿" else 1
68
+ offs.append(off)
69
+ for i, ch in enumerate(s):
70
+ if not stack:
71
+ if ch in _CLOSER:
72
+ stack.append(_CLOSER[ch])
73
+ group_start = i
74
+ group_off = offs[i]
75
+ else:
76
+ text.append(ch)
77
+ pos.append(offs[i])
78
+ continue
79
+ if ch in _CLOSER:
80
+ stack.append(_CLOSER[ch])
81
+ elif ch == stack[-1]:
82
+ stack.pop()
83
+ if not stack:
84
+ _add_group(s, group_start, i + 1, True, offs, groups)
85
+ text.append(PH)
86
+ pos.append(group_off)
87
+ # R3.2: a non-matching closer is literal text inside the group.
88
+ unbalanced = len(stack) > 0
89
+ if unbalanced:
90
+ # R3.3: the group runs to the end of the string.
91
+ _add_group(s, group_start, len(s), False, offs, groups)
92
+ text.append(PH)
93
+ pos.append(group_off)
94
+ return ScanResult(Skel("".join(text), pos, groups), unbalanced)
95
+
96
+
97
+ def _add_group(
98
+ s: str, start: int, end: int, closed: bool, offs: list[int], groups: dict[int, Group]
99
+ ) -> None:
100
+ inner = s[start + 1 : end - 1 if closed else end]
101
+ groups[offs[start]] = Group(
102
+ s[start], collapse_spaces(inner), s[start:end], closed, offs[start], offs[end]
103
+ )
104
+
105
+
106
+ # ---------------------------------------------------------------------------
107
+ # Skeleton operations
108
+
109
+
110
+ def slice_skel(s: Skel, start: int, end: Optional[int] = None) -> Skel:
111
+ n = len(s.text)
112
+ if end is None:
113
+ end = n
114
+ a = max(0, min(start, n))
115
+ b = max(a, min(end, n))
116
+ return Skel(s.text[a:b], s.pos[a:b], s.groups)
117
+
118
+
119
+ def trim_skel(s: Skel) -> Skel:
120
+ t = s.text
121
+ a = 0
122
+ b = len(t)
123
+ while a < b and is_whitespace(t[a]):
124
+ a += 1
125
+ while b > a and is_whitespace(t[b - 1]):
126
+ b -= 1
127
+ return slice_skel(s, a, b)
128
+
129
+
130
+ def concat_skel(a: Skel, b: Skel) -> Skel:
131
+ return Skel(a.text + b.text, a.pos + b.pos, a.groups)
132
+
133
+
134
+ def cut_skel(s: Skel, start: int, end: int) -> Skel:
135
+ """Remove the code points in [start, end)."""
136
+ return concat_skel(slice_skel(s, 0, start), slice_skel(s, end))
137
+
138
+
139
+ def group_at(s: Skel, index: int) -> Optional[Group]:
140
+ """The group behind the placeholder at ``index``, if it is one."""
141
+ if char_at(s.text, index) != PH:
142
+ return None
143
+ return s.groups.get(s.pos[index])
144
+
145
+
146
+ def groups_of(s: Skel) -> list[tuple[int, Group]]:
147
+ """Groups in the skeleton, left to right, with their index."""
148
+ out: list[tuple[int, Group]] = []
149
+ if PH not in s.text:
150
+ return out
151
+ for i, ch in enumerate(s.text):
152
+ if ch == PH:
153
+ g = s.groups.get(s.pos[i])
154
+ if g is not None:
155
+ out.append((i, g))
156
+ return out
157
+
158
+
159
+ def has_group(s: Skel) -> bool:
160
+ return PH in s.text
161
+
162
+
163
+ def render_skel(s: Skel) -> str:
164
+ """Substitute groups back (R3.4 reassembly). Whitespace is not touched."""
165
+ if PH not in s.text:
166
+ return s.text
167
+ out: list[str] = []
168
+ for i, ch in enumerate(s.text):
169
+ g = s.groups.get(s.pos[i]) if ch == PH else None
170
+ out.append(g.raw if g is not None else ch)
171
+ return "".join(out)
172
+
173
+
174
+ def start_pos(s: Skel, fallback: int = 0) -> int:
175
+ """Absolute offset of the first code point (or ``fallback`` for an empty skeleton)."""
176
+ return s.pos[0] if s.pos else fallback
177
+
178
+
179
+ def remove_groups(s: Skel, remove: set[int]) -> Skel:
180
+ """Remove the placeholders of the given groups."""
181
+ if not remove:
182
+ return s
183
+ text: list[str] = []
184
+ pos: list[int] = []
185
+ for ch, p in zip(s.text, s.pos):
186
+ if ch == PH and p in remove:
187
+ continue
188
+ text.append(ch)
189
+ pos.append(p)
190
+ return Skel("".join(text), pos, s.groups)
191
+
192
+
193
+ def is_spaced_dash_at(text: str, i: int) -> bool:
194
+ """R6.1: ``-`` with a space immediately on both sides."""
195
+ return 0 < i < len(text) - 1 and text[i] == "-" and text[i - 1] == " " and text[i + 1] == " "
196
+
197
+
198
+ def has_spaced_dash(text: str) -> bool:
199
+ return " - " in text
200
+
201
+
202
+ class Segment:
203
+ __slots__ = ("end", "skel", "start")
204
+
205
+ def __init__(self, skel: Skel, start: int, end: int) -> None:
206
+ self.skel = skel
207
+ #: Trimmed range inside the parent skeleton.
208
+ self.start = start
209
+ self.end = end
210
+
211
+
212
+ def _trimmed_segment(s: Skel, start: int, end: int) -> Segment:
213
+ t = s.text
214
+ a = start
215
+ b = max(start, end)
216
+ while a < b and is_whitespace(char_at(t, a)):
217
+ a += 1
218
+ while b > a and is_whitespace(char_at(t, b - 1)):
219
+ b -= 1
220
+ return Segment(slice_skel(s, a, b), a, b)
221
+
222
+
223
+ def split_at_spaced_dashes(s: Skel, allow_leading: bool = False) -> list[Segment]:
224
+ """R6.1 / R8.2: split a skeleton at every spaced dash into trimmed segments. With
225
+ ``allow_leading``, a string starting with ``- `` yields an empty first segment (R6.3)."""
226
+ segments: list[Segment] = []
227
+ t = s.text
228
+ start = 0
229
+ if allow_leading and t[:2] == "- ":
230
+ segments.append(Segment(slice_skel(s, 0, 0), 0, 0))
231
+ start = 2
232
+ i = max(start, 1)
233
+ n = len(t)
234
+ while i < n - 1:
235
+ if t[i] == "-" and t[i - 1] == " " and t[i + 1] == " ":
236
+ segments.append(_trimmed_segment(s, start, i - 1))
237
+ start = i + 2
238
+ i += 1
239
+ segments.append(_trimmed_segment(s, start, n))
240
+ return segments
@@ -0,0 +1,209 @@
1
+ """R6: artist/title separator, including R6.2 suffix peeling (shared with R8.2)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import NamedTuple, Optional, Union
6
+
7
+ from ._classify import (
8
+ UNKNOWN,
9
+ Classification,
10
+ Flag,
11
+ Junk,
12
+ Site,
13
+ Versions,
14
+ Year,
15
+ classify,
16
+ )
17
+ from ._context import Ctx, is_known_artist, warn
18
+ from ._scanner import (
19
+ PH,
20
+ Segment,
21
+ Skel,
22
+ concat_skel,
23
+ has_group,
24
+ render_skel,
25
+ slice_skel,
26
+ split_at_spaced_dashes,
27
+ start_pos,
28
+ trim_skel,
29
+ )
30
+ from ._words import (
31
+ char_at,
32
+ collapse_spaces,
33
+ in_vocab,
34
+ is_letter,
35
+ is_whitespace,
36
+ split_words,
37
+ year_of,
38
+ )
39
+
40
+ Peelable = Union[Junk, Flag, Year, Versions]
41
+
42
+
43
+ class Peeled(NamedTuple):
44
+ cls: Peelable
45
+ raw: str
46
+ pos: int
47
+
48
+
49
+ def _classify_segment(seg: Skel, ctx: Ctx) -> Classification:
50
+ """R6.2 classification of a bare dash segment: placeholders make it Unknown."""
51
+ if has_group(seg):
52
+ return UNKNOWN
53
+ text = collapse_spaces(seg.text)
54
+ return classify(text, Site("title", "-", start_pos(seg)), ctx)
55
+
56
+
57
+ def peel_suffixes(
58
+ segments: list[Segment], relaxed: bool, ctx: Ctx
59
+ ) -> tuple[list[Segment], list[Peeled]]:
60
+ """R6.2 / R8.2: peel classifiable suffixes off the right end. ``relaxed`` drops conditions
61
+ (i)–(iv) (R8.2 with an artist side). Returns the remaining segments and the peeled
62
+ suffixes, left to right."""
63
+ remaining = list(segments)
64
+ peeled: list[Peeled] = []
65
+ while len(remaining) >= 2:
66
+ last = remaining[-1]
67
+ cls = _classify_segment(last.skel, ctx)
68
+ if not isinstance(cls, (Junk, Flag, Year, Versions)):
69
+ break
70
+ words = split_words(last.skel.text)
71
+ # (ii) does not apply to prefix-form versions (`Andy C b2b Hedex - Live at Printworks`).
72
+ prefix_form = isinstance(cls, Versions) and all(v.prefix_form for v in cls.versions)
73
+ ok = (
74
+ relaxed
75
+ or len(remaining) - 1 >= 2
76
+ or (len(words) >= 2 and not prefix_form)
77
+ or any(year_of(w) is not None for w in words)
78
+ or isinstance(cls, Junk)
79
+ )
80
+ if not ok:
81
+ break
82
+ peeled.insert(0, Peeled(cls, collapse_spaces(last.skel.text), start_pos(last.skel)))
83
+ remaining.pop()
84
+ return remaining, peeled
85
+
86
+
87
+ class SeparatorResult(NamedTuple):
88
+ artist: Optional[Skel]
89
+ title: Skel
90
+ #: A separator split the string (R6.3–R6.7): R8.2 peels without conditions.
91
+ split: bool
92
+ peeled: list[Peeled]
93
+
94
+
95
+ def _non_empty(s: Skel) -> Optional[Skel]:
96
+ t = trim_skel(s)
97
+ return t if len(t.text) > 0 else None
98
+
99
+
100
+ def _split_at(s: Skel, cut_start: int, cut_end: int) -> tuple[Optional[Skel], Skel]:
101
+ return _non_empty(slice_skel(s, 0, cut_start)), trim_skel(slice_skel(s, cut_end))
102
+
103
+
104
+ def find_separator(skel: Skel, mode: str, uploader: Optional[Skel], ctx: Ctx) -> SeparatorResult:
105
+ segments = split_at_spaced_dashes(skel, True)
106
+ remaining, peeled = peel_suffixes(segments, False, ctx)
107
+
108
+ # R6.3
109
+ if len(remaining) >= 2:
110
+ first = remaining[0]
111
+ second = remaining[1]
112
+ last_seg = remaining[-1]
113
+ artist = first.skel if len(first.skel.text) > 0 else None
114
+ title = slice_skel(skel, second.start, last_seg.end)
115
+ return SeparatorResult(artist, title, artist is not None, peeled)
116
+
117
+ rest = remaining[0].skel
118
+ other = _split_without_spaced_dash(rest, mode, ctx)
119
+ if other is not None:
120
+ return SeparatorResult(other[0], other[1], other[0] is not None, peeled)
121
+
122
+ # R6.8: the uploader fallback does not warn noSeparator.
123
+ artist = uploader if mode == "youtube" else None
124
+ if len(rest.text) > 0 and artist is None:
125
+ warn(ctx, "noSeparator")
126
+ return SeparatorResult(artist, rest, False, peeled)
127
+
128
+
129
+ def _split_without_spaced_dash(
130
+ s: Skel, mode: str, ctx: Ctx
131
+ ) -> Optional[tuple[Optional[Skel], Skel]]:
132
+ asym = _asymmetric_dash(s.text)
133
+ if asym >= 0:
134
+ warn(ctx, "asymmetricDashSplit")
135
+ return _split_at(s, asym, asym + 1)
136
+ if mode == "youtube":
137
+ quoted = _quoted_title(s)
138
+ if quoted is not None:
139
+ warn(ctx, "quotedTitleSplit")
140
+ return quoted
141
+ by = _by_split(s, ctx)
142
+ if by >= 0:
143
+ warn(ctx, "bySplit")
144
+ left, right = _split_at(s, by, by + 4)
145
+ return (right if len(right.text) > 0 else None, left if left is not None else s)
146
+ if mode == "youtube" or mode == "filename":
147
+ dash = _unspaced_dash(s, ctx)
148
+ if dash >= 0:
149
+ warn(ctx, "unspacedDashSplit")
150
+ return _split_at(s, dash, dash + 1)
151
+ return None
152
+
153
+
154
+ def _asymmetric_dash(t: str) -> int:
155
+ """R6.4: the first ``-`` with a space on exactly one side."""
156
+ for i in range(1, len(t) - 1):
157
+ if t[i] != "-":
158
+ continue
159
+ if is_whitespace(t[i - 1]) != is_whitespace(t[i + 1]):
160
+ return i
161
+ return -1
162
+
163
+
164
+ def _quoted_title(s: Skel) -> Optional[tuple[Optional[Skel], Skel]]:
165
+ """R6.5: ``Artist "Title" (extra)``."""
166
+ t = s.text
167
+ open_ = t.find('"')
168
+ if open_ <= 0:
169
+ return None
170
+ close = t.find('"', open_ + 1)
171
+ if close < 0:
172
+ return None
173
+ artist = trim_skel(slice_skel(s, 0, open_))
174
+ last = artist.text[-1:]
175
+ if last == ":" or last == "-":
176
+ artist = trim_skel(slice_skel(artist, 0, len(artist.text) - 1))
177
+ title = trim_skel(concat_skel(slice_skel(s, open_ + 1, close), slice_skel(s, close + 1)))
178
+ return (artist if len(artist.text) > 0 else None, title)
179
+
180
+
181
+ def _by_split(s: Skel, ctx: Ctx) -> int:
182
+ """R6.6: index of the space before the last usable `` by ``. The right side must be
183
+ non-empty and not a single stopword (``Stand by Me``)."""
184
+ t = s.text
185
+ i = t.rfind(" by ")
186
+ while i > 0:
187
+ # Placeholder-only words (groups such as `(Official Video)`) don't count here.
188
+ right = [w for w in split_words(t[i + 4 :]) if w != PH]
189
+ if right and not (len(right) == 1 and in_vocab(right[0], ctx.t.stopwords)):
190
+ return i
191
+ # JS lastIndexOf(" by ", i - 1): the last occurrence starting at or before i - 1.
192
+ i = t.rfind(" by ", 0, i - 1 + 4) if i - 1 >= 0 else -1
193
+ return -1
194
+
195
+
196
+ def _unspaced_dash(s: Skel, ctx: Ctx) -> int:
197
+ """R6.7: ``Jay-Z-Numb`` → the chosen unspaced dash index."""
198
+ t = s.text
199
+ last_candidate = -1
200
+ for i in range(1, len(t) - 1):
201
+ if t[i] != "-" or is_whitespace(t[i - 1]) or is_whitespace(t[i + 1]):
202
+ continue
203
+ nxt = char_at(t, i + 1)
204
+ if not (nxt == '"' or is_letter(nxt)):
205
+ continue
206
+ if ctx.known_keys and is_known_artist(render_skel(slice_skel(s, 0, i)), ctx):
207
+ return i
208
+ last_candidate = i
209
+ return last_candidate