trackparse 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- trackparse/__init__.py +119 -0
- trackparse/_artists.py +101 -0
- trackparse/_assemble.py +141 -0
- trackparse/_classify.py +468 -0
- trackparse/_context.py +74 -0
- trackparse/_credits.py +218 -0
- trackparse/_data.py +513 -0
- trackparse/_format.py +149 -0
- trackparse/_mode.py +204 -0
- trackparse/_normalize.py +107 -0
- trackparse/_parser.py +257 -0
- trackparse/_prefix.py +143 -0
- trackparse/_scanner.py +240 -0
- trackparse/_separator.py +209 -0
- trackparse/_tables.py +228 -0
- trackparse/_title.py +246 -0
- trackparse/_words.py +344 -0
- trackparse/options.py +62 -0
- trackparse/py.typed +0 -0
- trackparse/types.py +259 -0
- trackparse-0.2.0.dist-info/METADATA +389 -0
- trackparse-0.2.0.dist-info/RECORD +24 -0
- trackparse-0.2.0.dist-info/WHEEL +4 -0
- trackparse-0.2.0.dist-info/licenses/LICENSE +21 -0
trackparse/_tables.py
ADDED
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
"""Vocabulary tables compiled from spec/data, optionally extended by ``keywords`` (R12)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from typing import Any, Optional
|
|
7
|
+
|
|
8
|
+
from ._data import (
|
|
9
|
+
DESCRIPTORS_DATA,
|
|
10
|
+
EXTENSIONS_DATA,
|
|
11
|
+
FEAT_MARKERS_DATA,
|
|
12
|
+
GENRES_DATA,
|
|
13
|
+
JOINERS_DATA,
|
|
14
|
+
JUNK_DATA,
|
|
15
|
+
NO_SPLIT_BEFORE_DATA,
|
|
16
|
+
PLATFORM_SUFFIXES_DATA,
|
|
17
|
+
STOPWORDS_DATA,
|
|
18
|
+
UNKNOWN_TOKENS_DATA,
|
|
19
|
+
VERSION_KEYWORDS_DATA,
|
|
20
|
+
)
|
|
21
|
+
from ._words import PhraseTable, ascii_lower, js_trim
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class SpacedJoiner:
|
|
25
|
+
__slots__ = ("canonical", "case_sensitive", "is_and", "raw")
|
|
26
|
+
|
|
27
|
+
def __init__(self, raw: str, canonical: str, is_and: bool, case_sensitive: bool) -> None:
|
|
28
|
+
self.raw = raw
|
|
29
|
+
self.canonical = canonical
|
|
30
|
+
self.is_and = is_and
|
|
31
|
+
self.case_sensitive = case_sensitive
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class RunInfo:
|
|
35
|
+
"""Category of a word in the modifier run left of a version head (R5.7.2)."""
|
|
36
|
+
|
|
37
|
+
__slots__ = ("descriptor", "genre", "head", "type_capable")
|
|
38
|
+
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
type_capable: Optional[str] = None,
|
|
42
|
+
descriptor: Optional[str] = None,
|
|
43
|
+
genre: bool = False,
|
|
44
|
+
head: Optional[str] = None,
|
|
45
|
+
) -> None:
|
|
46
|
+
self.type_capable = type_capable
|
|
47
|
+
self.descriptor = descriptor
|
|
48
|
+
self.genre = genre
|
|
49
|
+
self.head = head
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class Tables:
|
|
53
|
+
__slots__ = (
|
|
54
|
+
"bracket_only_feat",
|
|
55
|
+
"extensions",
|
|
56
|
+
"feat_canonical",
|
|
57
|
+
"feat_markers",
|
|
58
|
+
"generic_head_types",
|
|
59
|
+
"heads",
|
|
60
|
+
"junk_connectors",
|
|
61
|
+
"junk_or_genre",
|
|
62
|
+
"junk_phrases",
|
|
63
|
+
"label_suffixes",
|
|
64
|
+
"no_split_before",
|
|
65
|
+
"platform_suffixes",
|
|
66
|
+
"prefix_forms",
|
|
67
|
+
"producer_markers",
|
|
68
|
+
"run_words",
|
|
69
|
+
"spaced_joiners",
|
|
70
|
+
"stopwords",
|
|
71
|
+
"tight_joiners",
|
|
72
|
+
"title_feat_markers",
|
|
73
|
+
"title_producer_markers",
|
|
74
|
+
"unknown_case_insensitive",
|
|
75
|
+
"unknown_case_sensitive",
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
feat_markers: PhraseTable[bool]
|
|
79
|
+
bracket_only_feat: PhraseTable[bool]
|
|
80
|
+
producer_markers: PhraseTable[bool]
|
|
81
|
+
#: R8.3: the only unbracketed feat markers on the title side (+ keywords feat_markers).
|
|
82
|
+
title_feat_markers: PhraseTable[bool]
|
|
83
|
+
#: R8.3: the only unbracketed producer markers on the title side.
|
|
84
|
+
title_producer_markers: PhraseTable[bool]
|
|
85
|
+
#: Junk phrases plus genres (kind ``genre``): the R5.1 tokenizer vocabulary.
|
|
86
|
+
junk_or_genre: PhraseTable[str]
|
|
87
|
+
#: Junk phrases only (R8.4).
|
|
88
|
+
junk_phrases: PhraseTable[str]
|
|
89
|
+
junk_connectors: frozenset[str]
|
|
90
|
+
label_suffixes: frozenset[str]
|
|
91
|
+
heads: PhraseTable[str]
|
|
92
|
+
generic_head_types: frozenset[str]
|
|
93
|
+
#: Everything allowed in the R5.7.2 modifier run, keyed by phrase.
|
|
94
|
+
run_words: PhraseTable[RunInfo]
|
|
95
|
+
prefix_forms: PhraseTable[str]
|
|
96
|
+
spaced_joiners: list[SpacedJoiner]
|
|
97
|
+
tight_joiners: dict[str, str]
|
|
98
|
+
feat_canonical: str
|
|
99
|
+
no_split_before: frozenset[str]
|
|
100
|
+
stopwords: frozenset[str]
|
|
101
|
+
unknown_case_sensitive: frozenset[str]
|
|
102
|
+
unknown_case_insensitive: frozenset[str]
|
|
103
|
+
extensions: frozenset[str]
|
|
104
|
+
platform_suffixes: list[str]
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _is_array_index(key: str) -> bool:
|
|
108
|
+
if not key or len(key) > 10 or not all("0" <= c <= "9" for c in key):
|
|
109
|
+
return False
|
|
110
|
+
if len(key) > 1 and key[0] == "0":
|
|
111
|
+
return False
|
|
112
|
+
return int(key) <= 0xFFFFFFFE
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def js_entries(d: Mapping[str, Any]) -> list[tuple[str, Any]]:
|
|
116
|
+
"""``Object.entries`` order: integer-like keys first (ascending), then insertion order."""
|
|
117
|
+
items = list(d.items())
|
|
118
|
+
ints = sorted((kv for kv in items if _is_array_index(kv[0])), key=lambda kv: int(kv[0]))
|
|
119
|
+
return ints + [kv for kv in items if not _is_array_index(kv[0])]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _str_list(value: Any) -> list[str]:
|
|
123
|
+
if not isinstance(value, (list, tuple)):
|
|
124
|
+
return []
|
|
125
|
+
return [s for s in value if isinstance(s, str)]
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _str_map(value: Any) -> list[tuple[str, str]]:
|
|
129
|
+
if not isinstance(value, Mapping):
|
|
130
|
+
return []
|
|
131
|
+
return [(k, v) for k, v in js_entries(value) if isinstance(k, str) and isinstance(v, str)]
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _lower_keys(value: Any) -> list[str]:
|
|
135
|
+
return [k for k in (js_trim(ascii_lower(s)) for s in _str_list(value)) if len(k) > 0]
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def build_tables(keywords: Optional[Mapping[str, Any]] = None) -> Tables:
|
|
139
|
+
kw: Mapping[str, Any] = keywords if isinstance(keywords, Mapping) else {}
|
|
140
|
+
t = Tables()
|
|
141
|
+
extra_feat = _lower_keys(kw.get("feat_markers"))
|
|
142
|
+
t.feat_markers = PhraseTable((m, True) for m in [*FEAT_MARKERS_DATA["markers"], *extra_feat])
|
|
143
|
+
t.bracket_only_feat = PhraseTable((m, True) for m in FEAT_MARKERS_DATA["bracketOnly"])
|
|
144
|
+
t.producer_markers = PhraseTable((m, True) for m in FEAT_MARKERS_DATA["producer"])
|
|
145
|
+
t.title_feat_markers = PhraseTable(
|
|
146
|
+
(m, True) for m in [*FEAT_MARKERS_DATA["titleMarkers"], *extra_feat]
|
|
147
|
+
)
|
|
148
|
+
t.title_producer_markers = PhraseTable((m, True) for m in FEAT_MARKERS_DATA["titleProducer"])
|
|
149
|
+
|
|
150
|
+
junk_entries: list[tuple[str, str]] = [
|
|
151
|
+
*js_entries(JUNK_DATA["phrases"]),
|
|
152
|
+
*_str_map(kw.get("junk")),
|
|
153
|
+
]
|
|
154
|
+
genres: list[str] = [*GENRES_DATA["genres"], *_lower_keys(kw.get("genres"))]
|
|
155
|
+
t.junk_phrases = PhraseTable(junk_entries)
|
|
156
|
+
# Junk phrases are added first, so they win over an identical genre key.
|
|
157
|
+
t.junk_or_genre = PhraseTable([*junk_entries, *((g, "genre") for g in genres)])
|
|
158
|
+
|
|
159
|
+
head_entries: list[tuple[str, str]] = [
|
|
160
|
+
*js_entries(VERSION_KEYWORDS_DATA["heads"]),
|
|
161
|
+
*_str_map(kw.get("version_heads")),
|
|
162
|
+
]
|
|
163
|
+
t.heads = PhraseTable(head_entries)
|
|
164
|
+
|
|
165
|
+
run_info: dict[str, RunInfo] = {}
|
|
166
|
+
|
|
167
|
+
def note(key: str, **patch: Any) -> None:
|
|
168
|
+
k = js_trim(ascii_lower(key))
|
|
169
|
+
info = run_info.get(k)
|
|
170
|
+
if info is None:
|
|
171
|
+
info = RunInfo()
|
|
172
|
+
merged = RunInfo(info.type_capable, info.descriptor, info.genre, info.head)
|
|
173
|
+
for name, value in patch.items():
|
|
174
|
+
setattr(merged, name, value)
|
|
175
|
+
run_info[k] = merged
|
|
176
|
+
|
|
177
|
+
for k, typ in js_entries(VERSION_KEYWORDS_DATA["typeCapableModifiers"]):
|
|
178
|
+
note(k, type_capable=typ)
|
|
179
|
+
for d in [*DESCRIPTORS_DATA["descriptors"], *_lower_keys(kw.get("descriptors"))]:
|
|
180
|
+
note(d, descriptor=js_trim(ascii_lower(d)))
|
|
181
|
+
for g in genres:
|
|
182
|
+
note(g, genre=True)
|
|
183
|
+
for k, typ in head_entries:
|
|
184
|
+
note(k, head=typ)
|
|
185
|
+
|
|
186
|
+
t.junk_connectors = frozenset(JUNK_DATA["connectors"])
|
|
187
|
+
t.label_suffixes = frozenset(JUNK_DATA["labelSuffixes"])
|
|
188
|
+
t.generic_head_types = frozenset(VERSION_KEYWORDS_DATA["genericHeads"])
|
|
189
|
+
t.run_words = PhraseTable(run_info.items())
|
|
190
|
+
t.prefix_forms = PhraseTable(js_entries(VERSION_KEYWORDS_DATA["prefixForms"]))
|
|
191
|
+
t.spaced_joiners = [
|
|
192
|
+
SpacedJoiner(
|
|
193
|
+
j["raw"],
|
|
194
|
+
j["canonical"],
|
|
195
|
+
j.get("kind") == "and",
|
|
196
|
+
j.get("caseSensitive") is True,
|
|
197
|
+
)
|
|
198
|
+
for j in JOINERS_DATA["spaced"]
|
|
199
|
+
]
|
|
200
|
+
t.tight_joiners = {j["raw"]: j["canonical"] for j in JOINERS_DATA["tight"]}
|
|
201
|
+
t.feat_canonical = JOINERS_DATA["featCanonical"]
|
|
202
|
+
t.no_split_before = frozenset(NO_SPLIT_BEFORE_DATA["words"])
|
|
203
|
+
t.stopwords = frozenset(STOPWORDS_DATA["words"])
|
|
204
|
+
t.unknown_case_sensitive = frozenset(UNKNOWN_TOKENS_DATA["caseSensitive"])
|
|
205
|
+
t.unknown_case_insensitive = frozenset(UNKNOWN_TOKENS_DATA["caseInsensitive"])
|
|
206
|
+
t.extensions = frozenset(EXTENSIONS_DATA["extensions"])
|
|
207
|
+
# JS sorts by UTF-16 length; the suffixes are ASCII, so code-point length is the same.
|
|
208
|
+
t.platform_suffixes = sorted(PLATFORM_SUFFIXES_DATA["suffixes"], key=lambda s: -len(s))
|
|
209
|
+
return t
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
_default_tables: Optional[Tables] = None
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def get_default_tables() -> Tables:
|
|
216
|
+
global _default_tables
|
|
217
|
+
if _default_tables is None:
|
|
218
|
+
_default_tables = build_tables()
|
|
219
|
+
return _default_tables
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def tables_for(keywords: Optional[Mapping[str, Any]]) -> Tables:
|
|
223
|
+
return build_tables(keywords) if keywords else get_default_tables()
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def is_unknown_token(text: str, t: Tables) -> bool:
|
|
227
|
+
"""R7.5 / R8.6: an unknown placeholder name (``ID``, ``???``, ``Untitled``…)."""
|
|
228
|
+
return text in t.unknown_case_sensitive or ascii_lower(text) in t.unknown_case_insensitive
|
trackparse/_title.py
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
"""R8: the title side."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import NamedTuple
|
|
6
|
+
|
|
7
|
+
from ._classify import Feat, Flag, Junk, Producer, Site, VersionDraft, Versions, Year, classify
|
|
8
|
+
from ._context import Ctx
|
|
9
|
+
from ._credits import Credit, ListOptions, PositionedYear, split_credits
|
|
10
|
+
from ._mode import PositionedJunk
|
|
11
|
+
from ._scanner import (
|
|
12
|
+
PH,
|
|
13
|
+
Skel,
|
|
14
|
+
cut_skel,
|
|
15
|
+
groups_of,
|
|
16
|
+
remove_groups,
|
|
17
|
+
render_skel,
|
|
18
|
+
slice_skel,
|
|
19
|
+
split_at_spaced_dashes,
|
|
20
|
+
trim_skel,
|
|
21
|
+
)
|
|
22
|
+
from ._separator import Peeled, peel_suffixes
|
|
23
|
+
from ._tables import is_unknown_token
|
|
24
|
+
from ._words import char_at, collapse_spaces, word_spans
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class Extracted:
|
|
28
|
+
"""Everything the title side (and R6.2 peeling) can extract."""
|
|
29
|
+
|
|
30
|
+
__slots__ = ("clean", "credits", "explicit", "junk", "unknown_artist", "versions", "years")
|
|
31
|
+
|
|
32
|
+
def __init__(self) -> None:
|
|
33
|
+
self.credits: list[Credit] = []
|
|
34
|
+
self.versions: list[VersionDraft] = []
|
|
35
|
+
self.junk: list[PositionedJunk] = []
|
|
36
|
+
self.years: list[PositionedYear] = []
|
|
37
|
+
self.explicit = False
|
|
38
|
+
self.clean = False
|
|
39
|
+
self.unknown_artist = False
|
|
40
|
+
|
|
41
|
+
def set_flag(self, flag: str) -> None:
|
|
42
|
+
if flag == "explicit":
|
|
43
|
+
self.explicit = True
|
|
44
|
+
else:
|
|
45
|
+
self.clean = True
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def add_peeled(out: Extracted, peeled: list[Peeled]) -> None:
|
|
49
|
+
"""Record R6.2 / R8.2 peeled dash suffixes."""
|
|
50
|
+
for p in peeled:
|
|
51
|
+
c = p.cls
|
|
52
|
+
if isinstance(c, Versions):
|
|
53
|
+
out.versions.extend(c.versions)
|
|
54
|
+
elif isinstance(c, Junk):
|
|
55
|
+
out.junk.append(PositionedJunk(p.raw, c.junk_kind, p.pos))
|
|
56
|
+
elif isinstance(c, Flag):
|
|
57
|
+
out.set_flag(c.flag)
|
|
58
|
+
else:
|
|
59
|
+
out.years.append(PositionedYear(c.year, p.pos))
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class TitleSideResult(NamedTuple):
|
|
63
|
+
title: str
|
|
64
|
+
unknown_title: bool
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def parse_title_side(
|
|
68
|
+
skel: Skel, relaxed: bool, youtube: bool, out: Extracted, ctx: Ctx
|
|
69
|
+
) -> TitleSideResult:
|
|
70
|
+
"""``relaxed``: an artist side was split off (R8.2 peels without conditions)."""
|
|
71
|
+
removed = _extract_groups(skel, out, ctx) # R8.1
|
|
72
|
+
|
|
73
|
+
# R8.2
|
|
74
|
+
segments = split_at_spaced_dashes(skel)
|
|
75
|
+
remaining, peeled = peel_suffixes(segments, relaxed, ctx)
|
|
76
|
+
add_peeled(out, peeled)
|
|
77
|
+
skel = slice_skel(skel, 0, remaining[-1].end)
|
|
78
|
+
|
|
79
|
+
skel = _extract_unbracketed_credits(skel, out, ctx) # R8.3
|
|
80
|
+
skel = remove_groups(skel, removed)
|
|
81
|
+
if youtube:
|
|
82
|
+
skel = _strip_trailing_junk(skel, out, ctx) # R8.4
|
|
83
|
+
|
|
84
|
+
title = cleanup_title(render_skel(skel)) # R8.5
|
|
85
|
+
return TitleSideResult(title, len(title) > 0 and is_unknown_token(title, ctx.t)) # R8.6
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _extract_groups(skel: Skel, out: Extracted, ctx: Ctx) -> set[int]:
|
|
89
|
+
"""R8.1: classify every group; returns the positions of the groups to remove."""
|
|
90
|
+
removed: set[int] = set()
|
|
91
|
+
for _, group in groups_of(skel):
|
|
92
|
+
c = classify(group.inner, Site("title", group.open, group.pos), ctx)
|
|
93
|
+
if isinstance(c, Versions):
|
|
94
|
+
out.versions.extend(c.versions)
|
|
95
|
+
elif isinstance(c, Junk):
|
|
96
|
+
out.junk.append(PositionedJunk(group.inner, c.junk_kind, group.pos))
|
|
97
|
+
elif isinstance(c, Flag):
|
|
98
|
+
out.set_flag(c.flag)
|
|
99
|
+
elif isinstance(c, Year):
|
|
100
|
+
out.years.append(PositionedYear(c.year, group.pos))
|
|
101
|
+
elif isinstance(c, (Feat, Producer)):
|
|
102
|
+
out.credits.extend(c.credits)
|
|
103
|
+
out.years.extend(c.years)
|
|
104
|
+
if c.unknown:
|
|
105
|
+
out.unknown_artist = True
|
|
106
|
+
else:
|
|
107
|
+
continue
|
|
108
|
+
removed.add(group.pos)
|
|
109
|
+
return removed
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class _MarkerHit(NamedTuple):
|
|
113
|
+
start: int
|
|
114
|
+
end: int
|
|
115
|
+
marker: str
|
|
116
|
+
role: str # "featured" | "producer"
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _title_markers(s: Skel, ctx: Ctx) -> list[_MarkerHit]:
|
|
120
|
+
"""R8.3 markers, left to right, each with a space on both sides: feat markers from
|
|
121
|
+
``#titleMarkers`` (``feat.``, ``ft.``, ``featuring``; undotted ``feat``/``ft`` stay text)
|
|
122
|
+
and producer markers from ``#titleProducer``."""
|
|
123
|
+
spans = word_spans(s.text)
|
|
124
|
+
words = [w.text for w in spans]
|
|
125
|
+
hits: list[_MarkerHit] = []
|
|
126
|
+
i = 0
|
|
127
|
+
while i < len(spans):
|
|
128
|
+
span = spans[i]
|
|
129
|
+
if char_at(s.text, span.start - 1) != " ":
|
|
130
|
+
i += 1
|
|
131
|
+
continue
|
|
132
|
+
role = "featured"
|
|
133
|
+
feat = ctx.t.title_feat_markers.match_at(words, i)
|
|
134
|
+
if feat is not None:
|
|
135
|
+
last_index = i + feat.length - 1
|
|
136
|
+
else:
|
|
137
|
+
prod = ctx.t.title_producer_markers.match_at(words, i)
|
|
138
|
+
if prod is None:
|
|
139
|
+
i += 1
|
|
140
|
+
continue
|
|
141
|
+
role = "producer"
|
|
142
|
+
last_index = i + prod.length - 1
|
|
143
|
+
last = spans[last_index]
|
|
144
|
+
if char_at(s.text, last.end) != " ":
|
|
145
|
+
i += 1
|
|
146
|
+
continue
|
|
147
|
+
hits.append(_MarkerHit(span.start, last.end, s.text[span.start : last.end], role))
|
|
148
|
+
i = last_index + 1
|
|
149
|
+
return hits
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _extract_unbracketed_credits(inp: Skel, out: Extracted, ctx: Ctx) -> Skel:
|
|
153
|
+
"""R8.3: ``Title feat. X``, ``Title prod. by Y`` → credits, removed from the title. A list
|
|
154
|
+
runs to the next placeholder, the next marker of the other kind, or the end; later feat
|
|
155
|
+
markers inside a featured list are joiners (R7.2)."""
|
|
156
|
+
markers = _title_markers(inp, ctx)
|
|
157
|
+
cuts: list[tuple[int, int]] = []
|
|
158
|
+
i = 0
|
|
159
|
+
while i < len(markers):
|
|
160
|
+
hit = markers[i]
|
|
161
|
+
list_end = inp.text.find(PH, hit.end)
|
|
162
|
+
if list_end < 0:
|
|
163
|
+
list_end = len(inp.text)
|
|
164
|
+
other = next(
|
|
165
|
+
(
|
|
166
|
+
m
|
|
167
|
+
for k, m in enumerate(markers)
|
|
168
|
+
if k > i and m.role != hit.role and m.start >= hit.end
|
|
169
|
+
),
|
|
170
|
+
None,
|
|
171
|
+
)
|
|
172
|
+
if other is not None and other.start < list_end:
|
|
173
|
+
list_end = other.start
|
|
174
|
+
lst = trim_skel(slice_skel(inp, hit.end, list_end))
|
|
175
|
+
if len(lst.text) > 0:
|
|
176
|
+
r = split_credits(
|
|
177
|
+
lst,
|
|
178
|
+
ListOptions(hit.role, "title", hit.marker, False, hit.role == "featured"),
|
|
179
|
+
ctx,
|
|
180
|
+
)
|
|
181
|
+
out.credits.extend(r.credits)
|
|
182
|
+
out.years.extend(r.years)
|
|
183
|
+
if r.unknown:
|
|
184
|
+
out.unknown_artist = True
|
|
185
|
+
cuts.append((hit.start, list_end))
|
|
186
|
+
else:
|
|
187
|
+
list_end = hit.end
|
|
188
|
+
while i < len(markers) and markers[i].start < list_end:
|
|
189
|
+
i += 1
|
|
190
|
+
skel = inp
|
|
191
|
+
for start, end in reversed(cuts):
|
|
192
|
+
skel = cut_skel(skel, start, end)
|
|
193
|
+
return skel
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _strip_trailing_junk(inp: Skel, out: Extracted, ctx: Ctx) -> Skel:
|
|
197
|
+
"""R8.4: strip trailing unbracketed junk phrases (optionally after a lone ``-`` / ``|``).
|
|
198
|
+
|
|
199
|
+
Word spans are computed once and walked backwards, so repeated junk stays linear (R0.8).
|
|
200
|
+
"""
|
|
201
|
+
skel = trim_skel(inp)
|
|
202
|
+
spans = word_spans(skel.text)
|
|
203
|
+
words = [span.text for span in spans]
|
|
204
|
+
end = len(spans) # words still in the title
|
|
205
|
+
while True:
|
|
206
|
+
m = ctx.t.junk_phrases.match_ending(words, end)
|
|
207
|
+
if m is None:
|
|
208
|
+
break
|
|
209
|
+
first = end - m.length
|
|
210
|
+
keep = first
|
|
211
|
+
if keep > 0 and _is_lone_edge(words[keep - 1]):
|
|
212
|
+
keep -= 1
|
|
213
|
+
if keep == 0:
|
|
214
|
+
break # never strip the whole title
|
|
215
|
+
start = spans[first].start
|
|
216
|
+
raw = collapse_spaces(skel.text[start : spans[end - 1].end])
|
|
217
|
+
pos = skel.pos[start] if start < len(skel.pos) else 0
|
|
218
|
+
out.junk.append(PositionedJunk(raw, m.entry.value, pos))
|
|
219
|
+
end = keep
|
|
220
|
+
if end == len(spans):
|
|
221
|
+
return skel
|
|
222
|
+
return trim_skel(slice_skel(skel, 0, spans[end - 1].end))
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _is_lone_edge(word: str) -> bool:
|
|
226
|
+
return word == "-" or word == "|"
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _strip_lone_edges(s: str) -> str:
|
|
230
|
+
words = s.split(" ")
|
|
231
|
+
a = 0
|
|
232
|
+
b = len(words)
|
|
233
|
+
while a < b and _is_lone_edge(words[a]):
|
|
234
|
+
a += 1
|
|
235
|
+
while b > a and _is_lone_edge(words[b - 1]):
|
|
236
|
+
b -= 1
|
|
237
|
+
return " ".join(words[a:b])
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def cleanup_title(raw: str) -> str:
|
|
241
|
+
"""R8.5"""
|
|
242
|
+
title = _strip_lone_edges(collapse_spaces(raw))
|
|
243
|
+
q = title[:1]
|
|
244
|
+
if len(title) >= 2 and (q == '"' or q == "'") and title[-1] == q:
|
|
245
|
+
title = _strip_lone_edges(collapse_spaces(title[1:-1]))
|
|
246
|
+
return title
|