tocparser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tocparser/__init__.py +88 -0
- tocparser/errors.py +80 -0
- tocparser/grammar.lark +89 -0
- tocparser/models.py +739 -0
- tocparser/parser.py +695 -0
- tocparser/py.typed +0 -0
- tocparser/serializer.py +243 -0
- tocparser/times.py +92 -0
- tocparser-0.1.0.dist-info/METADATA +205 -0
- tocparser-0.1.0.dist-info/RECORD +12 -0
- tocparser-0.1.0.dist-info/WHEEL +4 -0
- tocparser-0.1.0.dist-info/licenses/LICENSE.txt +201 -0
tocparser/parser.py
ADDED
|
@@ -0,0 +1,695 @@
|
|
|
1
|
+
"""Turn TOC file text into :class:`~tocparser.models.Toc` objects."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from collections.abc import Iterator, Mapping
|
|
7
|
+
from contextlib import contextmanager
|
|
8
|
+
from functools import cache
|
|
9
|
+
from importlib import resources
|
|
10
|
+
from os import PathLike
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Literal, NamedTuple
|
|
13
|
+
|
|
14
|
+
from lark import Lark, Token, Transformer, v_args
|
|
15
|
+
from lark.exceptions import (
|
|
16
|
+
UnexpectedCharacters,
|
|
17
|
+
UnexpectedEOF,
|
|
18
|
+
UnexpectedInput,
|
|
19
|
+
UnexpectedToken,
|
|
20
|
+
VisitError,
|
|
21
|
+
)
|
|
22
|
+
from lark.tree import Meta
|
|
23
|
+
|
|
24
|
+
from tocparser.errors import Loc, TocError, TocParseError, TocValidationError
|
|
25
|
+
from tocparser.models import (
|
|
26
|
+
BINARY_DATA_TOO_LONG,
|
|
27
|
+
CD_TEXT_ALIASES,
|
|
28
|
+
LANGUAGE_CODE_EN,
|
|
29
|
+
MAX_BINARY_LENGTH,
|
|
30
|
+
PREGAP_IS_ZERO,
|
|
31
|
+
SILENCE_IS_ZERO,
|
|
32
|
+
ZERO_DATA_IS_ZERO,
|
|
33
|
+
CdText,
|
|
34
|
+
CdTextBlock,
|
|
35
|
+
CdTextEncoding,
|
|
36
|
+
CdTextItemName,
|
|
37
|
+
CdTextValue,
|
|
38
|
+
DataFile,
|
|
39
|
+
DataMode,
|
|
40
|
+
DiscType,
|
|
41
|
+
End,
|
|
42
|
+
Fifo,
|
|
43
|
+
File,
|
|
44
|
+
Silence,
|
|
45
|
+
Start,
|
|
46
|
+
SubChannelMode,
|
|
47
|
+
Toc,
|
|
48
|
+
Track,
|
|
49
|
+
TrackMode,
|
|
50
|
+
TrackStatement,
|
|
51
|
+
Zero,
|
|
52
|
+
validate_binary_value,
|
|
53
|
+
validate_block_number,
|
|
54
|
+
validate_catalog,
|
|
55
|
+
validate_first_track_number,
|
|
56
|
+
validate_isrc,
|
|
57
|
+
validate_language_code,
|
|
58
|
+
validate_language_number,
|
|
59
|
+
validate_non_zero_length,
|
|
60
|
+
)
|
|
61
|
+
from tocparser.times import Msf, Time
|
|
62
|
+
|
|
63
|
+
__all__ = ["parse", "parse_file"]
|
|
64
|
+
|
|
65
|
+
# cdrdao's scanner turns \" and \\ into the character and keeps \NNN as it is;
|
|
66
|
+
# a second pass then reads every backslash followed by three digits, including
|
|
67
|
+
# one that came from \\, as an octal byte. That pass, and the rule that a string
|
|
68
|
+
# is either ASCII with escapes or UTF-8 without, are Util::processMixedString in
|
|
69
|
+
# cdrdao 1.2.6's trackdb/util.cc.
|
|
70
|
+
_SCANNER_ESCAPE_RE = re.compile(r'\\(["\\])|(\\[0-9]{3})')
|
|
71
|
+
_OCTAL_RE = re.compile(r"\\([0-9]{3})")
|
|
72
|
+
_OCTAL_DIGITS_RE = re.compile(r"[0-7]*")
|
|
73
|
+
# The longest start of a string the grammar's STRING terminal would accept.
|
|
74
|
+
_STRING_PREFIX_RE = re.compile(r'"(?:\\[0-9]{3}|\\["\\]|[^"\\])*')
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@cache
|
|
78
|
+
def _get_parser() -> Lark:
|
|
79
|
+
grammar = resources.files(__package__).joinpath("grammar.lark").read_text()
|
|
80
|
+
return Lark(grammar, start="toc", parser="lalr", propagate_positions=True)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class _String(NamedTuple):
|
|
84
|
+
"""A decoded quoted string.
|
|
85
|
+
|
|
86
|
+
``raw`` is cdrdao's distinction between a string written in plain ASCII,
|
|
87
|
+
whose ``\\NNN`` escapes are bytes in the CD-TEXT block's encoding, and one
|
|
88
|
+
holding other characters, which is UTF-8 text and may not use escapes.
|
|
89
|
+
A raw string's escapes are decoded here as ISO-8859-1.
|
|
90
|
+
"""
|
|
91
|
+
|
|
92
|
+
text: str
|
|
93
|
+
raw: bool
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _decode_string(token: Token) -> _String:
|
|
97
|
+
text = _SCANNER_ESCAPE_RE.sub(lambda match: match[1] or match[2], str(token)[1:-1])
|
|
98
|
+
raw = text.isascii()
|
|
99
|
+
|
|
100
|
+
def octal(match: re.Match[str]) -> str:
|
|
101
|
+
if not raw:
|
|
102
|
+
raise TocValidationError("Illegal mixed UTF-8 and binary.")
|
|
103
|
+
# strtol() reads the octal digits it can and a char keeps the low byte.
|
|
104
|
+
digits = _OCTAL_DIGITS_RE.match(match[1])
|
|
105
|
+
assert digits is not None
|
|
106
|
+
return chr(int(digits[0] or "0", 8) & 0xFF)
|
|
107
|
+
|
|
108
|
+
return _String(_OCTAL_RE.sub(octal, text), raw)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _cd_text_string(value: _String, encoding: CdTextEncoding) -> str:
|
|
112
|
+
"""Read a raw string's bytes in its block's encoding, as cdrdao does.
|
|
113
|
+
|
|
114
|
+
cdrdao only warns about bytes the encoding cannot decode and keeps them,
|
|
115
|
+
but they have no text to hold here, nor to write back.
|
|
116
|
+
"""
|
|
117
|
+
if value.raw and encoding is CdTextEncoding.MS_JIS:
|
|
118
|
+
try:
|
|
119
|
+
return value.text.encode("latin-1").decode("cp932")
|
|
120
|
+
except UnicodeDecodeError:
|
|
121
|
+
raise TocValidationError(
|
|
122
|
+
f"CD-TEXT: Illegal byte sequence for {encoding.value}."
|
|
123
|
+
) from None
|
|
124
|
+
return value.text
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
class _Isrc(NamedTuple):
|
|
128
|
+
value: str
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
class _Copy(NamedTuple):
|
|
132
|
+
value: bool
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
class _PreEmphasis(NamedTuple):
|
|
136
|
+
value: bool
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class _Channels(NamedTuple):
|
|
140
|
+
value: Literal[2, 4]
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
class _Pregap(NamedTuple):
|
|
144
|
+
value: Time
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
class _Index(NamedTuple):
|
|
148
|
+
value: Time
|
|
149
|
+
line: int
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class _Statement(NamedTuple):
|
|
153
|
+
value: TrackStatement
|
|
154
|
+
line: int
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
class _Offset(NamedTuple):
|
|
158
|
+
value: int
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
class _CdTextItem(NamedTuple):
|
|
162
|
+
name: CdTextItemName
|
|
163
|
+
value: _String | list[int]
|
|
164
|
+
line: int
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
class _CdTextBlock(NamedTuple):
|
|
168
|
+
number: int
|
|
169
|
+
encoding: CdTextEncoding | None
|
|
170
|
+
items: list[_CdTextItem]
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
class _TrackCdText(NamedTuple):
|
|
174
|
+
cd_text: CdText
|
|
175
|
+
#: Lines of the items, keyed by their location inside the track.
|
|
176
|
+
lines: dict[Loc, int]
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
_Child = object
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _item_loc(number: int, name: CdTextItemName) -> Loc:
|
|
183
|
+
return ("cd_text", "blocks", number, "items", name.value)
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _line(token: Token) -> int:
|
|
187
|
+
# propagate_positions gives every token a position.
|
|
188
|
+
assert token.line is not None
|
|
189
|
+
return token.line
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
@v_args(meta=True)
|
|
193
|
+
class _TocTransformer(Transformer[Token, object]):
|
|
194
|
+
"""Builds models from the parse tree, reporting errors with a line number."""
|
|
195
|
+
|
|
196
|
+
def __init__(self, filename: str | None = None) -> None:
|
|
197
|
+
super().__init__()
|
|
198
|
+
self._filename = filename
|
|
199
|
+
# The encodings the disc's CD_TEXT sets, which its tracks follow.
|
|
200
|
+
self._encodings: dict[int, CdTextEncoding] = {}
|
|
201
|
+
# Lines of everything a Toc-level check can point at.
|
|
202
|
+
self._lines: dict[Loc, int] = {}
|
|
203
|
+
self._track_count = 0
|
|
204
|
+
|
|
205
|
+
@contextmanager
|
|
206
|
+
def _at(self, line: int | None) -> Iterator[None]:
|
|
207
|
+
"""Report a validation error raised inside at ``line``."""
|
|
208
|
+
try:
|
|
209
|
+
yield
|
|
210
|
+
except TocValidationError as exc:
|
|
211
|
+
raise TocValidationError(
|
|
212
|
+
exc.message,
|
|
213
|
+
line=exc.line if exc.line is not None else line,
|
|
214
|
+
filename=self._filename,
|
|
215
|
+
loc=exc.loc,
|
|
216
|
+
) from None
|
|
217
|
+
|
|
218
|
+
@contextmanager
|
|
219
|
+
def _located(
|
|
220
|
+
self, lines: Mapping[Loc, int], default: int | None, prefix: Loc = ()
|
|
221
|
+
) -> Iterator[None]:
|
|
222
|
+
"""Report a model's validation error at the line its ``loc`` came from."""
|
|
223
|
+
try:
|
|
224
|
+
yield
|
|
225
|
+
except TocValidationError as exc:
|
|
226
|
+
raise TocValidationError(
|
|
227
|
+
exc.message,
|
|
228
|
+
line=lines.get(exc.loc, default),
|
|
229
|
+
filename=self._filename,
|
|
230
|
+
loc=prefix + exc.loc,
|
|
231
|
+
) from None
|
|
232
|
+
|
|
233
|
+
def _string(self, token: Token) -> _String:
|
|
234
|
+
with self._at(token.line):
|
|
235
|
+
return _decode_string(token)
|
|
236
|
+
|
|
237
|
+
# -- disc level ------------------------------------------------------
|
|
238
|
+
|
|
239
|
+
def toc(self, meta: Meta, children: list[_Child]) -> Toc:
|
|
240
|
+
catalog: str | None = None
|
|
241
|
+
disc_types: list[DiscType] = []
|
|
242
|
+
first_track_number: int | None = None
|
|
243
|
+
cd_text: CdText | None = None
|
|
244
|
+
tracks: list[Track] = []
|
|
245
|
+
for child in children:
|
|
246
|
+
# DiscType is a str enum, so it must be tested before plain str.
|
|
247
|
+
if isinstance(child, DiscType):
|
|
248
|
+
disc_types.append(child)
|
|
249
|
+
elif isinstance(child, CdText):
|
|
250
|
+
cd_text = child
|
|
251
|
+
elif isinstance(child, Track):
|
|
252
|
+
tracks.append(child)
|
|
253
|
+
elif isinstance(child, int):
|
|
254
|
+
first_track_number = child
|
|
255
|
+
elif isinstance(child, str):
|
|
256
|
+
catalog = child
|
|
257
|
+
with self._located(self._lines, None):
|
|
258
|
+
return Toc(
|
|
259
|
+
catalog=catalog,
|
|
260
|
+
disc_type=disc_types[-1] if disc_types else DiscType.CD_DA,
|
|
261
|
+
superseded_disc_types=disc_types[:-1],
|
|
262
|
+
first_track_number=first_track_number,
|
|
263
|
+
cd_text=cd_text,
|
|
264
|
+
tracks=tracks,
|
|
265
|
+
)
|
|
266
|
+
|
|
267
|
+
def catalog(self, meta: Meta, children: list[Token]) -> str:
|
|
268
|
+
with self._at(meta.line):
|
|
269
|
+
return validate_catalog(self._string(children[0]).text)
|
|
270
|
+
|
|
271
|
+
def disc_type(self, meta: Meta, children: list[Token]) -> DiscType:
|
|
272
|
+
return DiscType(str(children[0]))
|
|
273
|
+
|
|
274
|
+
def first_track_no(self, meta: Meta, children: list[Token]) -> int:
|
|
275
|
+
line = _line(children[0])
|
|
276
|
+
with self._at(line):
|
|
277
|
+
return validate_first_track_number(int(children[0]))
|
|
278
|
+
|
|
279
|
+
# -- tracks ----------------------------------------------------------
|
|
280
|
+
|
|
281
|
+
def track(self, meta: Meta, children: list[_Child]) -> Track:
|
|
282
|
+
mode_token = children[0]
|
|
283
|
+
assert isinstance(mode_token, Token)
|
|
284
|
+
if str(mode_token) == DataMode.MODE0.value:
|
|
285
|
+
# cdrdao's grammar only allows MODE0 after ZERO.
|
|
286
|
+
raise TocParseError(
|
|
287
|
+
f'syntax error at "{mode_token}"',
|
|
288
|
+
line=mode_token.line,
|
|
289
|
+
column=mode_token.column,
|
|
290
|
+
filename=self._filename,
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
sub_channel_mode: SubChannelMode | None = None
|
|
294
|
+
isrc: str | None = None
|
|
295
|
+
copy_permitted: bool | None = None
|
|
296
|
+
pre_emphasis: bool | None = None
|
|
297
|
+
channels: Literal[2, 4] | None = None
|
|
298
|
+
cd_text: CdText | None = None
|
|
299
|
+
pregap: Time | None = None
|
|
300
|
+
statements: list[TrackStatement] = []
|
|
301
|
+
indexes: list[Time] = []
|
|
302
|
+
lines: dict[Loc, int] = {}
|
|
303
|
+
|
|
304
|
+
for child in children[1:]:
|
|
305
|
+
if isinstance(child, Token):
|
|
306
|
+
sub_channel_mode = SubChannelMode(str(child))
|
|
307
|
+
elif isinstance(child, _Isrc):
|
|
308
|
+
isrc = child.value
|
|
309
|
+
elif isinstance(child, _Copy):
|
|
310
|
+
copy_permitted = child.value
|
|
311
|
+
elif isinstance(child, _PreEmphasis):
|
|
312
|
+
pre_emphasis = child.value
|
|
313
|
+
elif isinstance(child, _Channels):
|
|
314
|
+
channels = child.value
|
|
315
|
+
elif isinstance(child, _TrackCdText):
|
|
316
|
+
cd_text = child.cd_text
|
|
317
|
+
lines.update(child.lines)
|
|
318
|
+
elif isinstance(child, _Pregap):
|
|
319
|
+
pregap = child.value
|
|
320
|
+
elif isinstance(child, _Statement):
|
|
321
|
+
lines[("statements", len(statements))] = child.line
|
|
322
|
+
statements.append(child.value)
|
|
323
|
+
elif isinstance(child, _Index):
|
|
324
|
+
lines[("indexes", len(indexes))] = child.line
|
|
325
|
+
indexes.append(child.value)
|
|
326
|
+
|
|
327
|
+
prefix: Loc = ("tracks", self._track_count)
|
|
328
|
+
self._track_count += 1
|
|
329
|
+
self._lines.update({prefix + loc: line for loc, line in lines.items()})
|
|
330
|
+
with self._located(lines, meta.line, prefix):
|
|
331
|
+
return Track(
|
|
332
|
+
mode=TrackMode(str(mode_token)),
|
|
333
|
+
sub_channel_mode=sub_channel_mode,
|
|
334
|
+
copy_permitted=copy_permitted,
|
|
335
|
+
pre_emphasis=pre_emphasis,
|
|
336
|
+
channels=channels,
|
|
337
|
+
isrc=isrc,
|
|
338
|
+
cd_text=cd_text,
|
|
339
|
+
pregap=pregap,
|
|
340
|
+
statements=statements,
|
|
341
|
+
indexes=indexes,
|
|
342
|
+
)
|
|
343
|
+
|
|
344
|
+
def isrc(self, meta: Meta, children: list[Token]) -> _Isrc:
|
|
345
|
+
with self._at(meta.line):
|
|
346
|
+
return _Isrc(validate_isrc(self._string(children[0]).text))
|
|
347
|
+
|
|
348
|
+
def copy(self, meta: Meta, children: list[Token]) -> _Copy:
|
|
349
|
+
return _Copy(not children)
|
|
350
|
+
|
|
351
|
+
def pre_emphasis(self, meta: Meta, children: list[Token]) -> _PreEmphasis:
|
|
352
|
+
return _PreEmphasis(not children)
|
|
353
|
+
|
|
354
|
+
def channels(self, meta: Meta, children: list[Token]) -> _Channels:
|
|
355
|
+
return _Channels(4 if str(children[0]).startswith("FOUR") else 2)
|
|
356
|
+
|
|
357
|
+
def pregap(self, meta: Meta, children: list[Time]) -> _Pregap:
|
|
358
|
+
with self._at(meta.line):
|
|
359
|
+
return _Pregap(validate_non_zero_length(children[0], PREGAP_IS_ZERO))
|
|
360
|
+
|
|
361
|
+
def index(self, meta: Meta, children: list[Time]) -> _Index:
|
|
362
|
+
return _Index(children[0], meta.line)
|
|
363
|
+
|
|
364
|
+
# -- track statements ------------------------------------------------
|
|
365
|
+
|
|
366
|
+
def file(self, meta: Meta, children: list[_Child]) -> _Statement:
|
|
367
|
+
keyword = children[0]
|
|
368
|
+
assert isinstance(keyword, Token)
|
|
369
|
+
filename_token = children[1]
|
|
370
|
+
assert isinstance(filename_token, Token)
|
|
371
|
+
rest = children[2:]
|
|
372
|
+
|
|
373
|
+
swap = False
|
|
374
|
+
if rest and isinstance(rest[0], Token) and rest[0].type == "SWAP":
|
|
375
|
+
swap = True
|
|
376
|
+
rest = rest[1:]
|
|
377
|
+
offset: int | None = None
|
|
378
|
+
if rest and isinstance(rest[0], _Offset):
|
|
379
|
+
offset = rest[0].value
|
|
380
|
+
rest = rest[1:]
|
|
381
|
+
|
|
382
|
+
statement = File(
|
|
383
|
+
filename=self._string(filename_token).text,
|
|
384
|
+
start=rest[0], # type: ignore[arg-type]
|
|
385
|
+
length=rest[1] if len(rest) > 1 else None, # type: ignore[arg-type]
|
|
386
|
+
swap=swap,
|
|
387
|
+
offset=offset,
|
|
388
|
+
audiofile=str(keyword) == "AUDIOFILE",
|
|
389
|
+
)
|
|
390
|
+
return _Statement(statement, meta.line)
|
|
391
|
+
|
|
392
|
+
def datafile(self, meta: Meta, children: list[_Child]) -> _Statement:
|
|
393
|
+
filename_token = children[0]
|
|
394
|
+
assert isinstance(filename_token, Token)
|
|
395
|
+
rest = children[1:]
|
|
396
|
+
offset: int | None = None
|
|
397
|
+
if rest and isinstance(rest[0], _Offset):
|
|
398
|
+
offset = rest[0].value
|
|
399
|
+
rest = rest[1:]
|
|
400
|
+
statement = DataFile(
|
|
401
|
+
filename=self._string(filename_token).text,
|
|
402
|
+
length=rest[0] if rest else None, # type: ignore[arg-type]
|
|
403
|
+
offset=offset,
|
|
404
|
+
)
|
|
405
|
+
return _Statement(statement, meta.line)
|
|
406
|
+
|
|
407
|
+
def fifo(self, meta: Meta, children: list[_Child]) -> _Statement:
|
|
408
|
+
filename_token = children[0]
|
|
409
|
+
assert isinstance(filename_token, Token)
|
|
410
|
+
statement = Fifo(
|
|
411
|
+
filename=self._string(filename_token).text,
|
|
412
|
+
length=children[1], # type: ignore[arg-type]
|
|
413
|
+
)
|
|
414
|
+
return _Statement(statement, meta.line)
|
|
415
|
+
|
|
416
|
+
def silence(self, meta: Meta, children: list[Time]) -> _Statement:
|
|
417
|
+
with self._at(meta.line):
|
|
418
|
+
length = validate_non_zero_length(children[0], SILENCE_IS_ZERO)
|
|
419
|
+
return _Statement(Silence(length=length), meta.line)
|
|
420
|
+
|
|
421
|
+
def zero(self, meta: Meta, children: list[_Child]) -> _Statement:
|
|
422
|
+
data_mode: DataMode | None = None
|
|
423
|
+
sub_channel_mode: SubChannelMode | None = None
|
|
424
|
+
rest = list(children)
|
|
425
|
+
if rest and isinstance(rest[0], Token) and rest[0].type == "MODE":
|
|
426
|
+
data_mode = DataMode(str(rest[0]))
|
|
427
|
+
rest = rest[1:]
|
|
428
|
+
if rest and isinstance(rest[0], Token) and rest[0].type == "SUB_CHANNEL_MODE":
|
|
429
|
+
sub_channel_mode = SubChannelMode(str(rest[0]))
|
|
430
|
+
rest = rest[1:]
|
|
431
|
+
with self._at(meta.line):
|
|
432
|
+
length = validate_non_zero_length(
|
|
433
|
+
rest[0], # type: ignore[arg-type]
|
|
434
|
+
ZERO_DATA_IS_ZERO,
|
|
435
|
+
)
|
|
436
|
+
statement = Zero(
|
|
437
|
+
length=length, data_mode=data_mode, sub_channel_mode=sub_channel_mode
|
|
438
|
+
)
|
|
439
|
+
return _Statement(statement, meta.line)
|
|
440
|
+
|
|
441
|
+
def start(self, meta: Meta, children: list[Time]) -> _Statement:
|
|
442
|
+
return _Statement(Start(position=children[0] if children else None), meta.line)
|
|
443
|
+
|
|
444
|
+
def end(self, meta: Meta, children: list[Time]) -> _Statement:
|
|
445
|
+
return _Statement(End(position=children[0] if children else None), meta.line)
|
|
446
|
+
|
|
447
|
+
def offset(self, meta: Meta, children: list[Token]) -> _Offset:
|
|
448
|
+
return _Offset(int(children[0]))
|
|
449
|
+
|
|
450
|
+
def msf(self, meta: Meta, children: list[Token]) -> Msf:
|
|
451
|
+
with self._at(meta.line):
|
|
452
|
+
return Msf.parse(str(children[0]))
|
|
453
|
+
|
|
454
|
+
def time(self, meta: Meta, children: list[Token]) -> Time:
|
|
455
|
+
token = children[0]
|
|
456
|
+
if token.type == "MSF":
|
|
457
|
+
with self._at(meta.line):
|
|
458
|
+
return Msf.parse(str(token))
|
|
459
|
+
return int(token)
|
|
460
|
+
|
|
461
|
+
# -- CD-TEXT ---------------------------------------------------------
|
|
462
|
+
|
|
463
|
+
def cd_text_global(self, meta: Meta, children: list[_Child]) -> CdText:
|
|
464
|
+
language_map: dict[int, int] = {}
|
|
465
|
+
blocks: list[_CdTextBlock] = []
|
|
466
|
+
for child in children:
|
|
467
|
+
if isinstance(child, dict):
|
|
468
|
+
language_map = child
|
|
469
|
+
else:
|
|
470
|
+
assert isinstance(child, _CdTextBlock)
|
|
471
|
+
blocks.append(child)
|
|
472
|
+
# Only the disc's blocks set encodings; its tracks follow them.
|
|
473
|
+
for block in blocks:
|
|
474
|
+
if block.encoding is not None:
|
|
475
|
+
self._encodings[block.number] = block.encoding
|
|
476
|
+
built, lines = self._cd_text_blocks(blocks)
|
|
477
|
+
self._lines.update(lines)
|
|
478
|
+
with self._located(lines, meta.line, ("cd_text",)):
|
|
479
|
+
return CdText(language_map=language_map, blocks=built)
|
|
480
|
+
|
|
481
|
+
def cd_text_track(self, meta: Meta, children: list[_Child]) -> _TrackCdText:
|
|
482
|
+
blocks = [child for child in children if isinstance(child, _CdTextBlock)]
|
|
483
|
+
built, lines = self._cd_text_blocks(blocks)
|
|
484
|
+
with self._located(lines, meta.line, ("cd_text",)):
|
|
485
|
+
return _TrackCdText(CdText(blocks=built), lines)
|
|
486
|
+
|
|
487
|
+
def _cd_text_blocks(
|
|
488
|
+
self, blocks: list[_CdTextBlock]
|
|
489
|
+
) -> tuple[dict[int, CdTextBlock], dict[Loc, int]]:
|
|
490
|
+
"""Merge blocks the way cdrdao's CD-TEXT container does.
|
|
491
|
+
|
|
492
|
+
cdrdao files every item under its language and pack type, so a block
|
|
493
|
+
number given twice adds to the first block, and an item given twice,
|
|
494
|
+
under either spelling of its pack, keeps the last value. cdrdao drops an
|
|
495
|
+
empty string before filing it, so one never replaces an earlier value;
|
|
496
|
+
a lone empty string is kept, so that it is written back.
|
|
497
|
+
"""
|
|
498
|
+
encodings: dict[int, CdTextEncoding | None] = {}
|
|
499
|
+
items: dict[int, dict[CdTextItemName, CdTextValue]] = {}
|
|
500
|
+
lines: dict[Loc, int] = {}
|
|
501
|
+
for block in blocks:
|
|
502
|
+
if block.encoding is not None or block.number not in encodings:
|
|
503
|
+
encodings[block.number] = block.encoding
|
|
504
|
+
block_items = items.setdefault(block.number, {})
|
|
505
|
+
encoding = self._encodings.get(block.number, CdTextEncoding.ISO_8859_1)
|
|
506
|
+
for item in block.items:
|
|
507
|
+
value: CdTextValue
|
|
508
|
+
if isinstance(item.value, _String):
|
|
509
|
+
with self._at(item.line):
|
|
510
|
+
value = _cd_text_string(item.value, encoding)
|
|
511
|
+
else:
|
|
512
|
+
value = item.value
|
|
513
|
+
alias = CD_TEXT_ALIASES.get(item.name)
|
|
514
|
+
if value == "" and (item.name in block_items or alias in block_items):
|
|
515
|
+
continue
|
|
516
|
+
if alias is not None and alias in block_items:
|
|
517
|
+
del block_items[alias]
|
|
518
|
+
del lines[_item_loc(block.number, alias)]
|
|
519
|
+
block_items[item.name] = value
|
|
520
|
+
lines[_item_loc(block.number, item.name)] = item.line
|
|
521
|
+
built = {
|
|
522
|
+
number: CdTextBlock(encoding=encodings[number], items=block_items)
|
|
523
|
+
for number, block_items in items.items()
|
|
524
|
+
}
|
|
525
|
+
return built, lines
|
|
526
|
+
|
|
527
|
+
def language_map(
|
|
528
|
+
self, meta: Meta, children: list[tuple[int, int]]
|
|
529
|
+
) -> dict[int, int]:
|
|
530
|
+
return dict(children)
|
|
531
|
+
|
|
532
|
+
def language_map_entry(self, meta: Meta, children: list[_Child]) -> tuple[int, int]:
|
|
533
|
+
number_token = children[0]
|
|
534
|
+
assert isinstance(number_token, Token)
|
|
535
|
+
code = children[1]
|
|
536
|
+
assert isinstance(code, int)
|
|
537
|
+
with self._at(number_token.line):
|
|
538
|
+
number = validate_language_number(int(number_token))
|
|
539
|
+
with self._at(meta.end_line):
|
|
540
|
+
return number, validate_language_code(code)
|
|
541
|
+
|
|
542
|
+
def language_code(self, meta: Meta, children: list[Token]) -> int:
|
|
543
|
+
token = children[0]
|
|
544
|
+
return LANGUAGE_CODE_EN if token.type == "LANGUAGE_EN" else int(token)
|
|
545
|
+
|
|
546
|
+
def cd_text_block(self, meta: Meta, children: list[_Child]) -> _CdTextBlock:
|
|
547
|
+
number_token = children[0]
|
|
548
|
+
assert isinstance(number_token, Token)
|
|
549
|
+
with self._at(number_token.line):
|
|
550
|
+
number = validate_block_number(int(number_token))
|
|
551
|
+
encoding: CdTextEncoding | None = None
|
|
552
|
+
items: list[_CdTextItem] = []
|
|
553
|
+
for child in children[1:]:
|
|
554
|
+
if isinstance(child, Token):
|
|
555
|
+
encoding = CdTextEncoding(str(child))
|
|
556
|
+
else:
|
|
557
|
+
assert isinstance(child, _CdTextItem)
|
|
558
|
+
items.append(child)
|
|
559
|
+
return _CdTextBlock(number, encoding, items)
|
|
560
|
+
|
|
561
|
+
def cd_text_item(self, meta: Meta, children: list[_Child]) -> _CdTextItem:
|
|
562
|
+
name_token = children[0]
|
|
563
|
+
assert isinstance(name_token, Token)
|
|
564
|
+
value = children[1]
|
|
565
|
+
assert isinstance(value, (_String, list))
|
|
566
|
+
return _CdTextItem(CdTextItemName(str(name_token)), value, _line(name_token))
|
|
567
|
+
|
|
568
|
+
def cd_text_value(self, meta: Meta, children: list[_Child]) -> _String | list[int]:
|
|
569
|
+
child = children[0]
|
|
570
|
+
if isinstance(child, Token):
|
|
571
|
+
return self._string(child)
|
|
572
|
+
assert isinstance(child, list)
|
|
573
|
+
return child
|
|
574
|
+
|
|
575
|
+
def binary_data(self, meta: Meta, children: list[Token]) -> list[int]:
|
|
576
|
+
values: list[int] = []
|
|
577
|
+
for position, token in enumerate(children):
|
|
578
|
+
with self._at(token.line):
|
|
579
|
+
values.append(validate_binary_value(int(token)))
|
|
580
|
+
if position == MAX_BINARY_LENGTH:
|
|
581
|
+
raise TocValidationError(BINARY_DATA_TOO_LONG)
|
|
582
|
+
return values
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def parse(text: str, *, filename: str | None = None) -> Toc:
|
|
586
|
+
"""Parse TOC file *text*.
|
|
587
|
+
|
|
588
|
+
Raises :class:`~tocparser.errors.TocParseError` for syntax errors and
|
|
589
|
+
:class:`~tocparser.errors.TocValidationError` for input that parses but
|
|
590
|
+
breaks a rule cdrdao enforces.
|
|
591
|
+
"""
|
|
592
|
+
try:
|
|
593
|
+
tree = _get_parser().parse(text)
|
|
594
|
+
except UnexpectedInput as exc:
|
|
595
|
+
raise _syntax_error(exc, text, filename) from None
|
|
596
|
+
|
|
597
|
+
try:
|
|
598
|
+
result = _TocTransformer(filename).transform(tree)
|
|
599
|
+
except VisitError as exc:
|
|
600
|
+
raise _unwrap(exc) from None
|
|
601
|
+
assert isinstance(result, Toc)
|
|
602
|
+
return result
|
|
603
|
+
|
|
604
|
+
|
|
605
|
+
def parse_file(path: str | PathLike[str], *, encoding: str = "utf-8") -> Toc:
|
|
606
|
+
"""Parse the TOC file at *path*.
|
|
607
|
+
|
|
608
|
+
cdrdao 1.2.5 and later write TOC files in UTF-8. Bytes that do not decode
|
|
609
|
+
in ``encoding`` raise :class:`~tocparser.errors.TocParseError`.
|
|
610
|
+
"""
|
|
611
|
+
file = Path(path)
|
|
612
|
+
data = file.read_bytes()
|
|
613
|
+
try:
|
|
614
|
+
text = data.decode(encoding)
|
|
615
|
+
except UnicodeDecodeError as exc:
|
|
616
|
+
# Everything before the bad byte decodes, and gives its line.
|
|
617
|
+
before = _universal_newlines(data[: exc.start].decode(encoding))
|
|
618
|
+
raise TocParseError(
|
|
619
|
+
f"Cannot decode byte 0x{data[exc.start]:02x} as {encoding}",
|
|
620
|
+
line=before.count("\n") + 1,
|
|
621
|
+
filename=str(file),
|
|
622
|
+
) from None
|
|
623
|
+
return parse(_universal_newlines(text), filename=str(file))
|
|
624
|
+
|
|
625
|
+
|
|
626
|
+
def _universal_newlines(text: str) -> str:
|
|
627
|
+
"""Line endings as a file opened in text mode would give them."""
|
|
628
|
+
return text.replace("\r\n", "\n").replace("\r", "\n")
|
|
629
|
+
|
|
630
|
+
|
|
631
|
+
def _unwrap(error: VisitError) -> BaseException:
|
|
632
|
+
original: BaseException = error
|
|
633
|
+
while isinstance(original, VisitError):
|
|
634
|
+
original = original.orig_exc
|
|
635
|
+
return original if isinstance(original, TocError) else error
|
|
636
|
+
|
|
637
|
+
|
|
638
|
+
def _syntax_error(
|
|
639
|
+
error: UnexpectedInput, text: str, filename: str | None
|
|
640
|
+
) -> TocParseError:
|
|
641
|
+
"""Word a syntax error the way cdrdao does."""
|
|
642
|
+
if isinstance(error, UnexpectedEOF) or (
|
|
643
|
+
isinstance(error, UnexpectedToken) and error.token.type == "$END"
|
|
644
|
+
):
|
|
645
|
+
return TocParseError(
|
|
646
|
+
'syntax error at "EOF"', line=text.count("\n") + 1, filename=filename
|
|
647
|
+
)
|
|
648
|
+
if isinstance(error, UnexpectedToken):
|
|
649
|
+
# cdrdao's scanner makes a string's opening quote a token of its own.
|
|
650
|
+
shown = '"' if error.token.type == "STRING" else str(error.token)
|
|
651
|
+
return TocParseError(
|
|
652
|
+
f'syntax error at "{shown}"',
|
|
653
|
+
line=error.line,
|
|
654
|
+
column=error.column,
|
|
655
|
+
filename=filename,
|
|
656
|
+
)
|
|
657
|
+
if isinstance(error, UnexpectedCharacters):
|
|
658
|
+
if error.char == '"':
|
|
659
|
+
return _broken_string(error, text, filename)
|
|
660
|
+
return TocParseError(
|
|
661
|
+
f"Illegal token: {error.char}",
|
|
662
|
+
line=error.line,
|
|
663
|
+
column=error.column,
|
|
664
|
+
filename=filename,
|
|
665
|
+
)
|
|
666
|
+
return TocParseError("syntax error", filename=filename) # pragma: no cover
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
def _broken_string(
|
|
670
|
+
error: UnexpectedCharacters, text: str, filename: str | None
|
|
671
|
+
) -> TocParseError:
|
|
672
|
+
"""A string that does not lex: an illegal backslash, or no closing quote."""
|
|
673
|
+
position = error.pos_in_stream
|
|
674
|
+
assert position is not None
|
|
675
|
+
prefix = _STRING_PREFIX_RE.match(text, position)
|
|
676
|
+
assert prefix is not None
|
|
677
|
+
end = prefix.end()
|
|
678
|
+
if end == len(text):
|
|
679
|
+
return TocParseError(
|
|
680
|
+
'syntax error at "EOF"', line=error.line, filename=filename
|
|
681
|
+
)
|
|
682
|
+
if text[end] == "\\":
|
|
683
|
+
# A backslash starting no escape cdrdao knows.
|
|
684
|
+
return TocParseError(
|
|
685
|
+
"Illegal token: \\",
|
|
686
|
+
line=text.count("\n", 0, end) + 1,
|
|
687
|
+
column=end - text.rfind("\n", 0, end),
|
|
688
|
+
filename=filename,
|
|
689
|
+
)
|
|
690
|
+
# A complete string where the grammar allows none. Lark's contextual lexer
|
|
691
|
+
# reports that as an unexpected STRING token instead, so this only guards
|
|
692
|
+
# against the lexer changing its mind.
|
|
693
|
+
return TocParseError( # pragma: no cover
|
|
694
|
+
'syntax error at """', line=error.line, column=error.column, filename=filename
|
|
695
|
+
)
|
tocparser/py.typed
ADDED
|
File without changes
|