trackparse 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
trackparse/_credits.py ADDED
@@ -0,0 +1,218 @@
1
+ """R7.3–R7.5: splitting a credit list into names."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import NamedTuple, Optional
6
+
7
+ from ._context import Ctx, match_known_at
8
+ from ._scanner import Skel, render_skel, slice_skel, trim_skel
9
+ from ._tables import is_unknown_token
10
+ from ._words import (
11
+ char_at,
12
+ collapse_spaces,
13
+ in_vocab,
14
+ is_all_digits,
15
+ is_digit,
16
+ is_whitespace,
17
+ r0_trim_end,
18
+ word_key,
19
+ word_matches,
20
+ year_of,
21
+ )
22
+
23
+
24
+ class Credit:
25
+ """An artist credit plus its absolute offset, used for ordering (R9.1)."""
26
+
27
+ __slots__ = ("joiner", "list", "name", "pos", "role", "source")
28
+
29
+ def __init__(
30
+ self, name: str, role: str, joiner: Optional[str], source: str, pos: int, list: int
31
+ ) -> None:
32
+ self.name = name
33
+ self.role = role
34
+ self.joiner = joiner
35
+ self.source = source
36
+ self.pos = pos
37
+ #: Id of the credit list it came from (R7.6 joiner inheritance in R9.2).
38
+ self.list = list
39
+
40
+
41
+ class PositionedYear(NamedTuple):
42
+ """A year with its absolute offset (R9.6: the track year is the first in text order)."""
43
+
44
+ year: int
45
+ pos: int
46
+
47
+
48
+ class ListOptions(NamedTuple):
49
+ role: str
50
+ source: str
51
+ #: Joiner of the first name: ``None``, or the feat / producer marker as written.
52
+ lead: Optional[str]
53
+ #: R7.3: ``with`` splits only in the main list of the artist side.
54
+ split_with: bool
55
+ #: R7.2: later feat markers act as joiners inside a featured list.
56
+ feat_markers_join: bool
57
+
58
+
59
+ class ListResult(NamedTuple):
60
+ credits: list[Credit]
61
+ #: R7.5: an unknown token was dropped.
62
+ unknown: bool
63
+ #: R7.5: dropped year-only names.
64
+ years: list[PositionedYear]
65
+
66
+
67
+ class _JoinerHit(NamedTuple):
68
+ start: int
69
+ end: int
70
+ raw: str
71
+
72
+
73
+ _AND_LIKE = frozenset(["&", "and", "+"])
74
+
75
+
76
+ def _word_at(text: str, start: int) -> int:
77
+ end = start
78
+ n = len(text)
79
+ while end < n and not is_whitespace(text[end]):
80
+ end += 1
81
+ return end
82
+
83
+
84
+ def _next_word(text: str, frm: int) -> str:
85
+ """The word after a joiner; a tight joiner (``,`` ``;``) ends it, since the name ends there."""
86
+ n = len(text)
87
+ a = frm
88
+ while a < n and is_whitespace(text[a]):
89
+ a += 1
90
+ b = a
91
+ while b < n and not is_whitespace(text[b]) and text[b] != "," and text[b] != ";":
92
+ b += 1
93
+ return text[a:b]
94
+
95
+
96
+ def _find_joiner(
97
+ text: str, frm: int, split_at_comma: bool, opts: ListOptions, ctx: Ctx
98
+ ) -> Optional[_JoinerHit]:
99
+ """R7.3b: the leftmost joiner occurrence at or after ``frm``, honouring the guards."""
100
+ n = len(text)
101
+ for p in range(max(frm, 0), n):
102
+ ch = text[p]
103
+ if ch == "," or ch == ";":
104
+ if ch == "," and is_digit(char_at(text, p - 1)) and is_digit(char_at(text, p + 1)):
105
+ continue # `1,000`
106
+ return _JoinerHit(p, p + 1, ch)
107
+ if p == 0 or text[p - 1] != " " or is_whitespace(ch):
108
+ continue
109
+ end = _word_at(text, p)
110
+ if end >= n or text[end] != " ":
111
+ continue # spaced joiners need a space after
112
+ word = text[p:end]
113
+ if _spaced_joiner(word, text, end, split_at_comma, opts, ctx):
114
+ return _JoinerHit(p, end, word)
115
+ return None
116
+
117
+
118
+ def _spaced_joiner(
119
+ word: str, text: str, end: int, split_at_comma: bool, opts: ListOptions, ctx: Ctx
120
+ ) -> bool:
121
+ if opts.feat_markers_join and ctx.t.feat_markers.match_at([word], 0) is not None:
122
+ return True
123
+ for j in ctx.t.spaced_joiners:
124
+ matches = word == j.raw if j.case_sensitive else word_matches(word, j.raw)
125
+ if not matches:
126
+ continue
127
+ if j.raw == "with" and not opts.split_with:
128
+ return False
129
+ if j.is_and:
130
+ if ctx.split_and == "never":
131
+ return False
132
+ if ctx.split_and == "auto" and not split_at_comma:
133
+ return False
134
+ return not (j.raw in _AND_LIKE and in_vocab(_next_word(text, end), ctx.t.no_split_before))
135
+ return False
136
+
137
+
138
+ def clean_name(raw: str) -> str:
139
+ """R7.4: clean one raw name."""
140
+ name = collapse_spaces(raw)
141
+ first = name[:1]
142
+ if len(name) >= 2 and (first == '"' or first == "'") and name[-1] == first:
143
+ name = collapse_spaces(name[1:-1])
144
+ while name.endswith(",") or name.endswith(";"):
145
+ name = r0_trim_end(name[:-1])
146
+ space = name.find(" ")
147
+ if space > 0 and word_key(name[:space]) == "by":
148
+ name = name[space + 1 :]
149
+ return collapse_spaces(name)
150
+
151
+
152
+ class _RawName(NamedTuple):
153
+ name: str
154
+ joiner: Optional[str]
155
+ pos: int
156
+
157
+
158
+ def _unwrap_list(lst: Skel) -> Skel:
159
+ """R7.4: strip one pair of wrapping quotes from the whole list (no other such quote inside)."""
160
+ t = trim_skel(lst)
161
+ text = t.text
162
+ q = text[:1]
163
+ if len(text) < 2 or (q != '"' and q != "'") or text[-1] != q:
164
+ return t
165
+ if text.find(q, 1) != len(text) - 1:
166
+ return t
167
+ return trim_skel(slice_skel(t, 1, len(text) - 1))
168
+
169
+
170
+ def split_credits(inp: Skel, opts: ListOptions, ctx: Ctx) -> ListResult:
171
+ """R7.3: split ``inp`` into credits."""
172
+ lst = _unwrap_list(inp) # R7.4: `"Camo & Krooked"`
173
+ text = lst.text
174
+ n = len(text)
175
+ raw: list[_RawName] = []
176
+ joiner = opts.lead
177
+ split_at_comma = False
178
+ start = 0
179
+ while start <= n:
180
+ while start < n and is_whitespace(text[start]):
181
+ start += 1
182
+ known = match_known_at(text, start, ctx, "-") # R7.3a
183
+ hit = _find_joiner(text, start + known, split_at_comma, opts, ctx)
184
+ end = hit.start if hit is not None else n
185
+ pos = lst.pos[start] if start < len(lst.pos) else lst.pos[-1] if lst.pos else 0
186
+ raw.append(_RawName(clean_name(render_skel(slice_skel(lst, start, end))), joiner, pos))
187
+ if hit is None:
188
+ break
189
+ if hit.raw == "," or hit.raw == ";":
190
+ split_at_comma = True # R7.3: `and` after a tight joiner
191
+ joiner = hit.raw
192
+ start = hit.end
193
+ return _finish_names(raw, opts, ctx)
194
+
195
+
196
+ def _finish_names(raw: list[_RawName], opts: ListOptions, ctx: Ctx) -> ListResult:
197
+ """R7.4 empty drops, R7.5 year/unknown drops. R7.6: the first survivor takes the lead
198
+ joiner (``None``, or the feat/producer marker)."""
199
+ named = [r for r in raw if len(r.name) > 0]
200
+ unknown = False
201
+ years: list[PositionedYear] = []
202
+ kept: list[_RawName] = []
203
+ for r in named:
204
+ if is_unknown_token(r.name, ctx.t):
205
+ unknown = True
206
+ continue
207
+ year = year_of(r.name) if is_all_digits(r.name) else None
208
+ if year is not None and len(named) > 1:
209
+ years.append(PositionedYear(year, r.pos))
210
+ continue
211
+ kept.append(r)
212
+ list_id = ctx.list_seq
213
+ ctx.list_seq += 1
214
+ credits = [
215
+ Credit(r.name, opts.role, opts.lead if i == 0 else r.joiner, opts.source, r.pos, list_id)
216
+ for i, r in enumerate(kept)
217
+ ]
218
+ return ListResult(credits, unknown, years)