trackparse 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
trackparse/__init__.py ADDED
@@ -0,0 +1,119 @@
1
+ """trackparse: spec-driven parser for music track strings.
2
+
3
+ Behaviour is defined by spec/SPEC.md (https://github.com/4matic/trackparse/blob/main/spec/SPEC.md);
4
+ rule IDs (R6.2…) in comments refer to it. Results are frozen dataclasses with snake_case
5
+ attributes; ``to_dict()`` gives the spec's canonical camelCase JSON.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from collections.abc import Sequence
11
+ from typing import Any, Optional
12
+
13
+ from ._data import SPEC_VERSION
14
+ from ._format import all_artists, format_track
15
+ from ._normalize import normalize
16
+ from ._parser import Parser, default_parser
17
+ from .options import FormatOptions, KeywordOptions, ParseOptions
18
+ from .types import (
19
+ Artist,
20
+ ArtistRole,
21
+ ArtistSource,
22
+ Flags,
23
+ Junk,
24
+ JunkKind,
25
+ Mode,
26
+ ModeOption,
27
+ ParsedTrack,
28
+ Position,
29
+ Timestamp,
30
+ Version,
31
+ VersionDelimiter,
32
+ VersionType,
33
+ Warning,
34
+ )
35
+
36
+ __version__ = "0.2.0" # x-release-please-version
37
+
38
+ __all__ = [
39
+ "SPEC_VERSION",
40
+ "Artist",
41
+ "ArtistRole",
42
+ "ArtistSource",
43
+ "Flags",
44
+ "FormatOptions",
45
+ "Junk",
46
+ "JunkKind",
47
+ "KeywordOptions",
48
+ "Mode",
49
+ "ModeOption",
50
+ "ParseOptions",
51
+ "ParsedTrack",
52
+ "Parser",
53
+ "Position",
54
+ "Timestamp",
55
+ "Version",
56
+ "VersionDelimiter",
57
+ "VersionType",
58
+ "Warning",
59
+ "__version__",
60
+ "all_artists",
61
+ "create_parser",
62
+ "format_track",
63
+ "normalize",
64
+ "parse",
65
+ "parse_artists",
66
+ ]
67
+
68
+
69
+ def parse(
70
+ input: str,
71
+ *,
72
+ mode: ModeOption = "auto",
73
+ uploader: Optional[str] = None,
74
+ known_artists: Sequence[str] = (),
75
+ split_and: str = "auto",
76
+ keywords: Optional[KeywordOptions] = None,
77
+ ) -> ParsedTrack:
78
+ """Parse a track string into a :class:`ParsedTrack`.
79
+
80
+ Total for any ``str`` (never raises on string input); raises ``TypeError`` for non-``str``.
81
+ """
82
+ return default_parser().parse(
83
+ input,
84
+ mode=mode,
85
+ uploader=uploader,
86
+ known_artists=known_artists,
87
+ split_and=split_and,
88
+ keywords=keywords,
89
+ )
90
+
91
+
92
+ def parse_artists(
93
+ input: str,
94
+ *,
95
+ mode: ModeOption = "auto",
96
+ uploader: Optional[str] = None,
97
+ known_artists: Sequence[str] = (),
98
+ split_and: str = "auto",
99
+ keywords: Optional[KeywordOptions] = None,
100
+ ) -> list[Artist]:
101
+ """Split an artist string into credits (R10). Every result has ``source == "artist"``."""
102
+ return default_parser().parse_artists(
103
+ input,
104
+ mode=mode,
105
+ uploader=uploader,
106
+ known_artists=known_artists,
107
+ split_and=split_and,
108
+ keywords=keywords,
109
+ )
110
+
111
+
112
+ def create_parser(**options: Any) -> Parser:
113
+ """A :class:`Parser` with ``keywords`` compiled once.
114
+
115
+ Accepts the same keyword arguments as :func:`parse`. Per-call options passed to
116
+ ``Parser.parse`` / ``Parser.parse_artists`` are merged over these; ``None`` does not
117
+ override, and per-call ``keywords`` are merged with the base ones.
118
+ """
119
+ return Parser(**options)
trackparse/_artists.py ADDED
@@ -0,0 +1,101 @@
1
+ """R7: the artist side."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import NamedTuple, Optional
6
+
7
+ from ._classify import Feat, Flag, Junk, Producer, Site, Year, classify
8
+ from ._context import Ctx
9
+ from ._credits import Credit, ListOptions, PositionedYear, split_credits
10
+ from ._mode import PositionedJunk
11
+ from ._scanner import PH, Skel, groups_of, remove_groups, slice_skel, trim_skel
12
+ from ._words import char_at, word_spans
13
+
14
+
15
+ class ArtistSideResult:
16
+ __slots__ = ("clean", "credits", "explicit", "junk", "unknown_artist", "years")
17
+
18
+ def __init__(self) -> None:
19
+ self.credits: list[Credit] = []
20
+ self.junk: list[PositionedJunk] = []
21
+ self.years: list[PositionedYear] = []
22
+ self.unknown_artist = False
23
+ #: R7.1: artist-side Flag groups.
24
+ self.explicit = False
25
+ self.clean = False
26
+
27
+
28
+ class FeatMarker(NamedTuple):
29
+ start: int
30
+ end: int
31
+ marker: str
32
+
33
+
34
+ def find_feat_marker(s: Skel, ctx: Ctx, frm: int = 0) -> Optional[FeatMarker]:
35
+ """R7.2: leftmost unbracketed feat marker with a space (or placeholder) left and a space
36
+ right."""
37
+ spans = word_spans(s.text)
38
+ words = [w.text for w in spans]
39
+ for i, span in enumerate(spans):
40
+ if span.start < frm:
41
+ continue
42
+ before = char_at(s.text, span.start - 1)
43
+ if before != " " and before != PH:
44
+ continue
45
+ m = ctx.t.feat_markers.match_at(words, i)
46
+ if m is None:
47
+ continue
48
+ last = spans[i + m.length - 1]
49
+ if char_at(s.text, last.end) != " ":
50
+ continue
51
+ return FeatMarker(span.start, last.end, s.text[span.start : last.end])
52
+ return None
53
+
54
+
55
+ def parse_artist_side(side: Skel, ctx: Ctx) -> ArtistSideResult:
56
+ """R7.1–R7.5 over an artist-side skeleton."""
57
+ result = ArtistSideResult()
58
+ removed: set[int] = set()
59
+
60
+ # R7.1
61
+ for _, group in groups_of(side):
62
+ c = classify(group.inner, Site("artist", group.open, group.pos), ctx)
63
+ if isinstance(c, (Feat, Producer)):
64
+ result.credits.extend(c.credits)
65
+ result.years.extend(c.years)
66
+ if c.unknown:
67
+ result.unknown_artist = True
68
+ elif isinstance(c, Flag):
69
+ if c.flag == "explicit":
70
+ result.explicit = True
71
+ else:
72
+ result.clean = True
73
+ elif isinstance(c, Year):
74
+ result.years.append(PositionedYear(c.year, group.pos))
75
+ elif isinstance(c, Junk):
76
+ result.junk.append(PositionedJunk(group.inner, c.junk_kind, group.pos))
77
+ else:
78
+ continue
79
+ removed.add(group.pos)
80
+ text = remove_groups(side, removed)
81
+
82
+ # R7.2
83
+ feat = find_feat_marker(text, ctx)
84
+ main = trim_skel(slice_skel(text, 0, feat.start)) if feat is not None else trim_skel(text)
85
+ main_list = split_credits(main, ListOptions("primary", "artist", None, True, False), ctx)
86
+ result.credits.extend(main_list.credits)
87
+ result.years.extend(main_list.years)
88
+ if main_list.unknown:
89
+ result.unknown_artist = True
90
+ if feat is not None:
91
+ featured = split_credits(
92
+ trim_skel(slice_skel(text, feat.end)),
93
+ ListOptions("featured", "artist", feat.marker, False, True),
94
+ ctx,
95
+ )
96
+ result.credits.extend(featured.credits)
97
+ result.years.extend(featured.years)
98
+ if featured.unknown:
99
+ result.unknown_artist = True
100
+ result.credits.sort(key=lambda c: c.pos)
101
+ return result
@@ -0,0 +1,141 @@
1
+ """R9: assembly and dedup."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Sequence
6
+ from typing import Optional, Protocol, TypeVar, cast
7
+
8
+ from ._classify import VersionDraft
9
+ from ._credits import Credit
10
+ from ._words import dedup_key
11
+ from .types import Artist, ArtistRole, ArtistSource, Version, VersionDelimiter, VersionType
12
+
13
+ _PRECEDENCE = {"primary": 2, "featured": 1, "producer": 0, "remixer": 0}
14
+
15
+
16
+ def to_artist(c: Credit) -> Artist:
17
+ return Artist(
18
+ name=c.name,
19
+ role=cast(ArtistRole, c.role),
20
+ joiner=c.joiner,
21
+ source=cast(ArtistSource, c.source),
22
+ )
23
+
24
+
25
+ class _Entry:
26
+ __slots__ = ("alive", "joiner", "list", "name", "role", "source")
27
+
28
+ def __init__(self, c: Credit) -> None:
29
+ self.name = c.name
30
+ self.role = c.role
31
+ self.joiner = c.joiner
32
+ self.source = c.source
33
+ self.list = c.list
34
+ self.alive = True
35
+
36
+
37
+ class _Lists:
38
+ """Alive members of every credit list, in entry order (R7.6 bookkeeping)."""
39
+
40
+ def __init__(self, entries: list[_Entry]) -> None:
41
+ self._members: dict[int, list[_Entry]] = {}
42
+ for e in entries:
43
+ self._members.setdefault(e.list, []).append(e)
44
+
45
+ def pass_joiner(self, e: _Entry, joiner: Optional[str]) -> None:
46
+ """R7.6: ``e`` leaves its list. If it was the list's first remaining name, the next
47
+ remaining name of that list takes ``joiner``."""
48
+ members = [x for x in self._members.get(e.list, []) if x.alive]
49
+ if not members or members[0] is not e:
50
+ return
51
+ if len(members) > 1:
52
+ members[1].joiner = joiner
53
+
54
+ def move(self, e: _Entry, new_list: int, entries: list[_Entry]) -> None:
55
+ """``e`` joins ``new_list`` (kept in global entry order, like the JS filter)."""
56
+ old = self._members.get(e.list, [])
57
+ if e in old:
58
+ old.remove(e)
59
+ e.list = new_list
60
+ order = {id(x): i for i, x in enumerate(entries)}
61
+ lst = self._members.setdefault(new_list, [])
62
+ lst.append(e)
63
+ lst.sort(key=lambda x: order[id(x)])
64
+
65
+
66
+ def dedup_artists(credits: Sequence[Credit]) -> list[Artist]:
67
+ """R9.2: keep the first occurrence. A later lower/equal-precedence duplicate is removed
68
+ (R7.6 joiner inheritance); a later higher-precedence duplicate (role upgrade) gives the kept
69
+ entry its role, joiner and source, and the kept entry moves into that occurrence's list."""
70
+ entries = [_Entry(c) for c in credits]
71
+ lists = _Lists(entries)
72
+ index: dict[str, _Entry] = {}
73
+ for e in entries:
74
+ # R9.2: producers are a separate credit class, deduped only among themselves.
75
+ key = ("p:" if e.role == "producer" else "c:") + dedup_key(e.name)
76
+ kept = index.get(key)
77
+ if kept is None:
78
+ index[key] = e
79
+ continue
80
+ if _PRECEDENCE.get(e.role, 0) > _PRECEDENCE.get(kept.role, 0):
81
+ lists.pass_joiner(kept, kept.joiner)
82
+ kept.role = e.role
83
+ kept.joiner = e.joiner
84
+ kept.source = e.source
85
+ e.alive = False
86
+ lists.move(kept, e.list, entries)
87
+ else:
88
+ lists.pass_joiner(e, e.joiner)
89
+ e.alive = False
90
+ return [
91
+ Artist(
92
+ name=e.name,
93
+ role=cast(ArtistRole, e.role),
94
+ joiner=e.joiner,
95
+ source=cast(ArtistSource, e.source),
96
+ )
97
+ for e in entries
98
+ if e.alive
99
+ ]
100
+
101
+
102
+ def to_version(v: VersionDraft) -> Version:
103
+ return Version(
104
+ type=cast(VersionType, v.type),
105
+ raw=v.raw,
106
+ artists=tuple(to_artist(a) for a in v.artists),
107
+ modifiers=tuple(v.modifiers),
108
+ descriptor=v.descriptor,
109
+ year=v.year,
110
+ unknown_artist=v.unknown_artist,
111
+ delimiter=cast(VersionDelimiter, v.delimiter),
112
+ )
113
+
114
+
115
+ def render_version(raw: str, delimiter: str) -> str:
116
+ """R9.4: a version rendered with its delimiter."""
117
+ if delimiter == "(":
118
+ return f"({raw})"
119
+ if delimiter == "[":
120
+ return f"[{raw}]"
121
+ if delimiter == "{":
122
+ return f"{{{raw}}}"
123
+ return f"- {raw}"
124
+
125
+
126
+ def full_title_of(title: str, versions: Sequence[Version]) -> str:
127
+ parts = [title, *(render_version(v.raw, v.delimiter) for v in versions)]
128
+ return " ".join(p for p in parts if len(p) > 0)
129
+
130
+
131
+ class _HasPos(Protocol):
132
+ @property
133
+ def pos(self) -> int: ...
134
+
135
+
136
+ T = TypeVar("T", bound=_HasPos)
137
+
138
+
139
+ def by_pos(items: Sequence[T]) -> list[T]:
140
+ """Stable sort by the ``pos`` attribute (like the JS sort)."""
141
+ return sorted(items, key=lambda x: x.pos)