trackparse 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- trackparse/__init__.py +119 -0
- trackparse/_artists.py +101 -0
- trackparse/_assemble.py +141 -0
- trackparse/_classify.py +468 -0
- trackparse/_context.py +74 -0
- trackparse/_credits.py +218 -0
- trackparse/_data.py +513 -0
- trackparse/_format.py +149 -0
- trackparse/_mode.py +204 -0
- trackparse/_normalize.py +107 -0
- trackparse/_parser.py +257 -0
- trackparse/_prefix.py +143 -0
- trackparse/_scanner.py +240 -0
- trackparse/_separator.py +209 -0
- trackparse/_tables.py +228 -0
- trackparse/_title.py +246 -0
- trackparse/_words.py +344 -0
- trackparse/options.py +62 -0
- trackparse/py.typed +0 -0
- trackparse/types.py +259 -0
- trackparse-0.2.0.dist-info/METADATA +389 -0
- trackparse-0.2.0.dist-info/RECORD +24 -0
- trackparse-0.2.0.dist-info/WHEEL +4 -0
- trackparse-0.2.0.dist-info/licenses/LICENSE +21 -0
trackparse/_credits.py
ADDED
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
"""R7.3–R7.5: splitting a credit list into names."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import NamedTuple, Optional
|
|
6
|
+
|
|
7
|
+
from ._context import Ctx, match_known_at
|
|
8
|
+
from ._scanner import Skel, render_skel, slice_skel, trim_skel
|
|
9
|
+
from ._tables import is_unknown_token
|
|
10
|
+
from ._words import (
|
|
11
|
+
char_at,
|
|
12
|
+
collapse_spaces,
|
|
13
|
+
in_vocab,
|
|
14
|
+
is_all_digits,
|
|
15
|
+
is_digit,
|
|
16
|
+
is_whitespace,
|
|
17
|
+
r0_trim_end,
|
|
18
|
+
word_key,
|
|
19
|
+
word_matches,
|
|
20
|
+
year_of,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Credit:
|
|
25
|
+
"""An artist credit plus its absolute offset, used for ordering (R9.1)."""
|
|
26
|
+
|
|
27
|
+
__slots__ = ("joiner", "list", "name", "pos", "role", "source")
|
|
28
|
+
|
|
29
|
+
def __init__(
|
|
30
|
+
self, name: str, role: str, joiner: Optional[str], source: str, pos: int, list: int
|
|
31
|
+
) -> None:
|
|
32
|
+
self.name = name
|
|
33
|
+
self.role = role
|
|
34
|
+
self.joiner = joiner
|
|
35
|
+
self.source = source
|
|
36
|
+
self.pos = pos
|
|
37
|
+
#: Id of the credit list it came from (R7.6 joiner inheritance in R9.2).
|
|
38
|
+
self.list = list
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class PositionedYear(NamedTuple):
|
|
42
|
+
"""A year with its absolute offset (R9.6: the track year is the first in text order)."""
|
|
43
|
+
|
|
44
|
+
year: int
|
|
45
|
+
pos: int
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class ListOptions(NamedTuple):
|
|
49
|
+
role: str
|
|
50
|
+
source: str
|
|
51
|
+
#: Joiner of the first name: ``None``, or the feat / producer marker as written.
|
|
52
|
+
lead: Optional[str]
|
|
53
|
+
#: R7.3: ``with`` splits only in the main list of the artist side.
|
|
54
|
+
split_with: bool
|
|
55
|
+
#: R7.2: later feat markers act as joiners inside a featured list.
|
|
56
|
+
feat_markers_join: bool
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class ListResult(NamedTuple):
|
|
60
|
+
credits: list[Credit]
|
|
61
|
+
#: R7.5: an unknown token was dropped.
|
|
62
|
+
unknown: bool
|
|
63
|
+
#: R7.5: dropped year-only names.
|
|
64
|
+
years: list[PositionedYear]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
class _JoinerHit(NamedTuple):
|
|
68
|
+
start: int
|
|
69
|
+
end: int
|
|
70
|
+
raw: str
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
_AND_LIKE = frozenset(["&", "and", "+"])
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _word_at(text: str, start: int) -> int:
|
|
77
|
+
end = start
|
|
78
|
+
n = len(text)
|
|
79
|
+
while end < n and not is_whitespace(text[end]):
|
|
80
|
+
end += 1
|
|
81
|
+
return end
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _next_word(text: str, frm: int) -> str:
|
|
85
|
+
"""The word after a joiner; a tight joiner (``,`` ``;``) ends it, since the name ends there."""
|
|
86
|
+
n = len(text)
|
|
87
|
+
a = frm
|
|
88
|
+
while a < n and is_whitespace(text[a]):
|
|
89
|
+
a += 1
|
|
90
|
+
b = a
|
|
91
|
+
while b < n and not is_whitespace(text[b]) and text[b] != "," and text[b] != ";":
|
|
92
|
+
b += 1
|
|
93
|
+
return text[a:b]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _find_joiner(
|
|
97
|
+
text: str, frm: int, split_at_comma: bool, opts: ListOptions, ctx: Ctx
|
|
98
|
+
) -> Optional[_JoinerHit]:
|
|
99
|
+
"""R7.3b: the leftmost joiner occurrence at or after ``frm``, honouring the guards."""
|
|
100
|
+
n = len(text)
|
|
101
|
+
for p in range(max(frm, 0), n):
|
|
102
|
+
ch = text[p]
|
|
103
|
+
if ch == "," or ch == ";":
|
|
104
|
+
if ch == "," and is_digit(char_at(text, p - 1)) and is_digit(char_at(text, p + 1)):
|
|
105
|
+
continue # `1,000`
|
|
106
|
+
return _JoinerHit(p, p + 1, ch)
|
|
107
|
+
if p == 0 or text[p - 1] != " " or is_whitespace(ch):
|
|
108
|
+
continue
|
|
109
|
+
end = _word_at(text, p)
|
|
110
|
+
if end >= n or text[end] != " ":
|
|
111
|
+
continue # spaced joiners need a space after
|
|
112
|
+
word = text[p:end]
|
|
113
|
+
if _spaced_joiner(word, text, end, split_at_comma, opts, ctx):
|
|
114
|
+
return _JoinerHit(p, end, word)
|
|
115
|
+
return None
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _spaced_joiner(
|
|
119
|
+
word: str, text: str, end: int, split_at_comma: bool, opts: ListOptions, ctx: Ctx
|
|
120
|
+
) -> bool:
|
|
121
|
+
if opts.feat_markers_join and ctx.t.feat_markers.match_at([word], 0) is not None:
|
|
122
|
+
return True
|
|
123
|
+
for j in ctx.t.spaced_joiners:
|
|
124
|
+
matches = word == j.raw if j.case_sensitive else word_matches(word, j.raw)
|
|
125
|
+
if not matches:
|
|
126
|
+
continue
|
|
127
|
+
if j.raw == "with" and not opts.split_with:
|
|
128
|
+
return False
|
|
129
|
+
if j.is_and:
|
|
130
|
+
if ctx.split_and == "never":
|
|
131
|
+
return False
|
|
132
|
+
if ctx.split_and == "auto" and not split_at_comma:
|
|
133
|
+
return False
|
|
134
|
+
return not (j.raw in _AND_LIKE and in_vocab(_next_word(text, end), ctx.t.no_split_before))
|
|
135
|
+
return False
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def clean_name(raw: str) -> str:
|
|
139
|
+
"""R7.4: clean one raw name."""
|
|
140
|
+
name = collapse_spaces(raw)
|
|
141
|
+
first = name[:1]
|
|
142
|
+
if len(name) >= 2 and (first == '"' or first == "'") and name[-1] == first:
|
|
143
|
+
name = collapse_spaces(name[1:-1])
|
|
144
|
+
while name.endswith(",") or name.endswith(";"):
|
|
145
|
+
name = r0_trim_end(name[:-1])
|
|
146
|
+
space = name.find(" ")
|
|
147
|
+
if space > 0 and word_key(name[:space]) == "by":
|
|
148
|
+
name = name[space + 1 :]
|
|
149
|
+
return collapse_spaces(name)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class _RawName(NamedTuple):
|
|
153
|
+
name: str
|
|
154
|
+
joiner: Optional[str]
|
|
155
|
+
pos: int
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _unwrap_list(lst: Skel) -> Skel:
|
|
159
|
+
"""R7.4: strip one pair of wrapping quotes from the whole list (no other such quote inside)."""
|
|
160
|
+
t = trim_skel(lst)
|
|
161
|
+
text = t.text
|
|
162
|
+
q = text[:1]
|
|
163
|
+
if len(text) < 2 or (q != '"' and q != "'") or text[-1] != q:
|
|
164
|
+
return t
|
|
165
|
+
if text.find(q, 1) != len(text) - 1:
|
|
166
|
+
return t
|
|
167
|
+
return trim_skel(slice_skel(t, 1, len(text) - 1))
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def split_credits(inp: Skel, opts: ListOptions, ctx: Ctx) -> ListResult:
|
|
171
|
+
"""R7.3: split ``inp`` into credits."""
|
|
172
|
+
lst = _unwrap_list(inp) # R7.4: `"Camo & Krooked"`
|
|
173
|
+
text = lst.text
|
|
174
|
+
n = len(text)
|
|
175
|
+
raw: list[_RawName] = []
|
|
176
|
+
joiner = opts.lead
|
|
177
|
+
split_at_comma = False
|
|
178
|
+
start = 0
|
|
179
|
+
while start <= n:
|
|
180
|
+
while start < n and is_whitespace(text[start]):
|
|
181
|
+
start += 1
|
|
182
|
+
known = match_known_at(text, start, ctx, "-") # R7.3a
|
|
183
|
+
hit = _find_joiner(text, start + known, split_at_comma, opts, ctx)
|
|
184
|
+
end = hit.start if hit is not None else n
|
|
185
|
+
pos = lst.pos[start] if start < len(lst.pos) else lst.pos[-1] if lst.pos else 0
|
|
186
|
+
raw.append(_RawName(clean_name(render_skel(slice_skel(lst, start, end))), joiner, pos))
|
|
187
|
+
if hit is None:
|
|
188
|
+
break
|
|
189
|
+
if hit.raw == "," or hit.raw == ";":
|
|
190
|
+
split_at_comma = True # R7.3: `and` after a tight joiner
|
|
191
|
+
joiner = hit.raw
|
|
192
|
+
start = hit.end
|
|
193
|
+
return _finish_names(raw, opts, ctx)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _finish_names(raw: list[_RawName], opts: ListOptions, ctx: Ctx) -> ListResult:
|
|
197
|
+
"""R7.4 empty drops, R7.5 year/unknown drops. R7.6: the first survivor takes the lead
|
|
198
|
+
joiner (``None``, or the feat/producer marker)."""
|
|
199
|
+
named = [r for r in raw if len(r.name) > 0]
|
|
200
|
+
unknown = False
|
|
201
|
+
years: list[PositionedYear] = []
|
|
202
|
+
kept: list[_RawName] = []
|
|
203
|
+
for r in named:
|
|
204
|
+
if is_unknown_token(r.name, ctx.t):
|
|
205
|
+
unknown = True
|
|
206
|
+
continue
|
|
207
|
+
year = year_of(r.name) if is_all_digits(r.name) else None
|
|
208
|
+
if year is not None and len(named) > 1:
|
|
209
|
+
years.append(PositionedYear(year, r.pos))
|
|
210
|
+
continue
|
|
211
|
+
kept.append(r)
|
|
212
|
+
list_id = ctx.list_seq
|
|
213
|
+
ctx.list_seq += 1
|
|
214
|
+
credits = [
|
|
215
|
+
Credit(r.name, opts.role, opts.lead if i == 0 else r.joiner, opts.source, r.pos, list_id)
|
|
216
|
+
for i, r in enumerate(kept)
|
|
217
|
+
]
|
|
218
|
+
return ListResult(credits, unknown, years)
|