tocparser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tocparser/parser.py ADDED
@@ -0,0 +1,695 @@
1
+ """Turn TOC file text into :class:`~tocparser.models.Toc` objects."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Iterator, Mapping
7
+ from contextlib import contextmanager
8
+ from functools import cache
9
+ from importlib import resources
10
+ from os import PathLike
11
+ from pathlib import Path
12
+ from typing import Literal, NamedTuple
13
+
14
+ from lark import Lark, Token, Transformer, v_args
15
+ from lark.exceptions import (
16
+ UnexpectedCharacters,
17
+ UnexpectedEOF,
18
+ UnexpectedInput,
19
+ UnexpectedToken,
20
+ VisitError,
21
+ )
22
+ from lark.tree import Meta
23
+
24
+ from tocparser.errors import Loc, TocError, TocParseError, TocValidationError
25
+ from tocparser.models import (
26
+ BINARY_DATA_TOO_LONG,
27
+ CD_TEXT_ALIASES,
28
+ LANGUAGE_CODE_EN,
29
+ MAX_BINARY_LENGTH,
30
+ PREGAP_IS_ZERO,
31
+ SILENCE_IS_ZERO,
32
+ ZERO_DATA_IS_ZERO,
33
+ CdText,
34
+ CdTextBlock,
35
+ CdTextEncoding,
36
+ CdTextItemName,
37
+ CdTextValue,
38
+ DataFile,
39
+ DataMode,
40
+ DiscType,
41
+ End,
42
+ Fifo,
43
+ File,
44
+ Silence,
45
+ Start,
46
+ SubChannelMode,
47
+ Toc,
48
+ Track,
49
+ TrackMode,
50
+ TrackStatement,
51
+ Zero,
52
+ validate_binary_value,
53
+ validate_block_number,
54
+ validate_catalog,
55
+ validate_first_track_number,
56
+ validate_isrc,
57
+ validate_language_code,
58
+ validate_language_number,
59
+ validate_non_zero_length,
60
+ )
61
+ from tocparser.times import Msf, Time
62
+
63
+ __all__ = ["parse", "parse_file"]
64
+
65
+ # cdrdao's scanner turns \" and \\ into the character and keeps \NNN as it is;
66
+ # a second pass then reads every backslash followed by three digits, including
67
+ # one that came from \\, as an octal byte. That pass, and the rule that a string
68
+ # is either ASCII with escapes or UTF-8 without, are Util::processMixedString in
69
+ # cdrdao 1.2.6's trackdb/util.cc.
70
+ _SCANNER_ESCAPE_RE = re.compile(r'\\(["\\])|(\\[0-9]{3})')
71
+ _OCTAL_RE = re.compile(r"\\([0-9]{3})")
72
+ _OCTAL_DIGITS_RE = re.compile(r"[0-7]*")
73
+ # The longest start of a string the grammar's STRING terminal would accept.
74
+ _STRING_PREFIX_RE = re.compile(r'"(?:\\[0-9]{3}|\\["\\]|[^"\\])*')
75
+
76
+
77
+ @cache
78
+ def _get_parser() -> Lark:
79
+ grammar = resources.files(__package__).joinpath("grammar.lark").read_text()
80
+ return Lark(grammar, start="toc", parser="lalr", propagate_positions=True)
81
+
82
+
83
+ class _String(NamedTuple):
84
+ """A decoded quoted string.
85
+
86
+ ``raw`` is cdrdao's distinction between a string written in plain ASCII,
87
+ whose ``\\NNN`` escapes are bytes in the CD-TEXT block's encoding, and one
88
+ holding other characters, which is UTF-8 text and may not use escapes.
89
+ A raw string's escapes are decoded here as ISO-8859-1.
90
+ """
91
+
92
+ text: str
93
+ raw: bool
94
+
95
+
96
+ def _decode_string(token: Token) -> _String:
97
+ text = _SCANNER_ESCAPE_RE.sub(lambda match: match[1] or match[2], str(token)[1:-1])
98
+ raw = text.isascii()
99
+
100
+ def octal(match: re.Match[str]) -> str:
101
+ if not raw:
102
+ raise TocValidationError("Illegal mixed UTF-8 and binary.")
103
+ # strtol() reads the octal digits it can and a char keeps the low byte.
104
+ digits = _OCTAL_DIGITS_RE.match(match[1])
105
+ assert digits is not None
106
+ return chr(int(digits[0] or "0", 8) & 0xFF)
107
+
108
+ return _String(_OCTAL_RE.sub(octal, text), raw)
109
+
110
+
111
+ def _cd_text_string(value: _String, encoding: CdTextEncoding) -> str:
112
+ """Read a raw string's bytes in its block's encoding, as cdrdao does.
113
+
114
+ cdrdao only warns about bytes the encoding cannot decode and keeps them,
115
+ but they have no text to hold here, nor to write back.
116
+ """
117
+ if value.raw and encoding is CdTextEncoding.MS_JIS:
118
+ try:
119
+ return value.text.encode("latin-1").decode("cp932")
120
+ except UnicodeDecodeError:
121
+ raise TocValidationError(
122
+ f"CD-TEXT: Illegal byte sequence for {encoding.value}."
123
+ ) from None
124
+ return value.text
125
+
126
+
127
+ class _Isrc(NamedTuple):
128
+ value: str
129
+
130
+
131
+ class _Copy(NamedTuple):
132
+ value: bool
133
+
134
+
135
+ class _PreEmphasis(NamedTuple):
136
+ value: bool
137
+
138
+
139
+ class _Channels(NamedTuple):
140
+ value: Literal[2, 4]
141
+
142
+
143
+ class _Pregap(NamedTuple):
144
+ value: Time
145
+
146
+
147
+ class _Index(NamedTuple):
148
+ value: Time
149
+ line: int
150
+
151
+
152
+ class _Statement(NamedTuple):
153
+ value: TrackStatement
154
+ line: int
155
+
156
+
157
+ class _Offset(NamedTuple):
158
+ value: int
159
+
160
+
161
+ class _CdTextItem(NamedTuple):
162
+ name: CdTextItemName
163
+ value: _String | list[int]
164
+ line: int
165
+
166
+
167
+ class _CdTextBlock(NamedTuple):
168
+ number: int
169
+ encoding: CdTextEncoding | None
170
+ items: list[_CdTextItem]
171
+
172
+
173
+ class _TrackCdText(NamedTuple):
174
+ cd_text: CdText
175
+ #: Lines of the items, keyed by their location inside the track.
176
+ lines: dict[Loc, int]
177
+
178
+
179
+ _Child = object
180
+
181
+
182
+ def _item_loc(number: int, name: CdTextItemName) -> Loc:
183
+ return ("cd_text", "blocks", number, "items", name.value)
184
+
185
+
186
+ def _line(token: Token) -> int:
187
+ # propagate_positions gives every token a position.
188
+ assert token.line is not None
189
+ return token.line
190
+
191
+
192
+ @v_args(meta=True)
193
+ class _TocTransformer(Transformer[Token, object]):
194
+ """Builds models from the parse tree, reporting errors with a line number."""
195
+
196
+ def __init__(self, filename: str | None = None) -> None:
197
+ super().__init__()
198
+ self._filename = filename
199
+ # The encodings the disc's CD_TEXT sets, which its tracks follow.
200
+ self._encodings: dict[int, CdTextEncoding] = {}
201
+ # Lines of everything a Toc-level check can point at.
202
+ self._lines: dict[Loc, int] = {}
203
+ self._track_count = 0
204
+
205
+ @contextmanager
206
+ def _at(self, line: int | None) -> Iterator[None]:
207
+ """Report a validation error raised inside at ``line``."""
208
+ try:
209
+ yield
210
+ except TocValidationError as exc:
211
+ raise TocValidationError(
212
+ exc.message,
213
+ line=exc.line if exc.line is not None else line,
214
+ filename=self._filename,
215
+ loc=exc.loc,
216
+ ) from None
217
+
218
+ @contextmanager
219
+ def _located(
220
+ self, lines: Mapping[Loc, int], default: int | None, prefix: Loc = ()
221
+ ) -> Iterator[None]:
222
+ """Report a model's validation error at the line its ``loc`` came from."""
223
+ try:
224
+ yield
225
+ except TocValidationError as exc:
226
+ raise TocValidationError(
227
+ exc.message,
228
+ line=lines.get(exc.loc, default),
229
+ filename=self._filename,
230
+ loc=prefix + exc.loc,
231
+ ) from None
232
+
233
+ def _string(self, token: Token) -> _String:
234
+ with self._at(token.line):
235
+ return _decode_string(token)
236
+
237
+ # -- disc level ------------------------------------------------------
238
+
239
+ def toc(self, meta: Meta, children: list[_Child]) -> Toc:
240
+ catalog: str | None = None
241
+ disc_types: list[DiscType] = []
242
+ first_track_number: int | None = None
243
+ cd_text: CdText | None = None
244
+ tracks: list[Track] = []
245
+ for child in children:
246
+ # DiscType is a str enum, so it must be tested before plain str.
247
+ if isinstance(child, DiscType):
248
+ disc_types.append(child)
249
+ elif isinstance(child, CdText):
250
+ cd_text = child
251
+ elif isinstance(child, Track):
252
+ tracks.append(child)
253
+ elif isinstance(child, int):
254
+ first_track_number = child
255
+ elif isinstance(child, str):
256
+ catalog = child
257
+ with self._located(self._lines, None):
258
+ return Toc(
259
+ catalog=catalog,
260
+ disc_type=disc_types[-1] if disc_types else DiscType.CD_DA,
261
+ superseded_disc_types=disc_types[:-1],
262
+ first_track_number=first_track_number,
263
+ cd_text=cd_text,
264
+ tracks=tracks,
265
+ )
266
+
267
+ def catalog(self, meta: Meta, children: list[Token]) -> str:
268
+ with self._at(meta.line):
269
+ return validate_catalog(self._string(children[0]).text)
270
+
271
+ def disc_type(self, meta: Meta, children: list[Token]) -> DiscType:
272
+ return DiscType(str(children[0]))
273
+
274
+ def first_track_no(self, meta: Meta, children: list[Token]) -> int:
275
+ line = _line(children[0])
276
+ with self._at(line):
277
+ return validate_first_track_number(int(children[0]))
278
+
279
+ # -- tracks ----------------------------------------------------------
280
+
281
+ def track(self, meta: Meta, children: list[_Child]) -> Track:
282
+ mode_token = children[0]
283
+ assert isinstance(mode_token, Token)
284
+ if str(mode_token) == DataMode.MODE0.value:
285
+ # cdrdao's grammar only allows MODE0 after ZERO.
286
+ raise TocParseError(
287
+ f'syntax error at "{mode_token}"',
288
+ line=mode_token.line,
289
+ column=mode_token.column,
290
+ filename=self._filename,
291
+ )
292
+
293
+ sub_channel_mode: SubChannelMode | None = None
294
+ isrc: str | None = None
295
+ copy_permitted: bool | None = None
296
+ pre_emphasis: bool | None = None
297
+ channels: Literal[2, 4] | None = None
298
+ cd_text: CdText | None = None
299
+ pregap: Time | None = None
300
+ statements: list[TrackStatement] = []
301
+ indexes: list[Time] = []
302
+ lines: dict[Loc, int] = {}
303
+
304
+ for child in children[1:]:
305
+ if isinstance(child, Token):
306
+ sub_channel_mode = SubChannelMode(str(child))
307
+ elif isinstance(child, _Isrc):
308
+ isrc = child.value
309
+ elif isinstance(child, _Copy):
310
+ copy_permitted = child.value
311
+ elif isinstance(child, _PreEmphasis):
312
+ pre_emphasis = child.value
313
+ elif isinstance(child, _Channels):
314
+ channels = child.value
315
+ elif isinstance(child, _TrackCdText):
316
+ cd_text = child.cd_text
317
+ lines.update(child.lines)
318
+ elif isinstance(child, _Pregap):
319
+ pregap = child.value
320
+ elif isinstance(child, _Statement):
321
+ lines[("statements", len(statements))] = child.line
322
+ statements.append(child.value)
323
+ elif isinstance(child, _Index):
324
+ lines[("indexes", len(indexes))] = child.line
325
+ indexes.append(child.value)
326
+
327
+ prefix: Loc = ("tracks", self._track_count)
328
+ self._track_count += 1
329
+ self._lines.update({prefix + loc: line for loc, line in lines.items()})
330
+ with self._located(lines, meta.line, prefix):
331
+ return Track(
332
+ mode=TrackMode(str(mode_token)),
333
+ sub_channel_mode=sub_channel_mode,
334
+ copy_permitted=copy_permitted,
335
+ pre_emphasis=pre_emphasis,
336
+ channels=channels,
337
+ isrc=isrc,
338
+ cd_text=cd_text,
339
+ pregap=pregap,
340
+ statements=statements,
341
+ indexes=indexes,
342
+ )
343
+
344
+ def isrc(self, meta: Meta, children: list[Token]) -> _Isrc:
345
+ with self._at(meta.line):
346
+ return _Isrc(validate_isrc(self._string(children[0]).text))
347
+
348
+ def copy(self, meta: Meta, children: list[Token]) -> _Copy:
349
+ return _Copy(not children)
350
+
351
+ def pre_emphasis(self, meta: Meta, children: list[Token]) -> _PreEmphasis:
352
+ return _PreEmphasis(not children)
353
+
354
+ def channels(self, meta: Meta, children: list[Token]) -> _Channels:
355
+ return _Channels(4 if str(children[0]).startswith("FOUR") else 2)
356
+
357
+ def pregap(self, meta: Meta, children: list[Time]) -> _Pregap:
358
+ with self._at(meta.line):
359
+ return _Pregap(validate_non_zero_length(children[0], PREGAP_IS_ZERO))
360
+
361
+ def index(self, meta: Meta, children: list[Time]) -> _Index:
362
+ return _Index(children[0], meta.line)
363
+
364
+ # -- track statements ------------------------------------------------
365
+
366
+ def file(self, meta: Meta, children: list[_Child]) -> _Statement:
367
+ keyword = children[0]
368
+ assert isinstance(keyword, Token)
369
+ filename_token = children[1]
370
+ assert isinstance(filename_token, Token)
371
+ rest = children[2:]
372
+
373
+ swap = False
374
+ if rest and isinstance(rest[0], Token) and rest[0].type == "SWAP":
375
+ swap = True
376
+ rest = rest[1:]
377
+ offset: int | None = None
378
+ if rest and isinstance(rest[0], _Offset):
379
+ offset = rest[0].value
380
+ rest = rest[1:]
381
+
382
+ statement = File(
383
+ filename=self._string(filename_token).text,
384
+ start=rest[0], # type: ignore[arg-type]
385
+ length=rest[1] if len(rest) > 1 else None, # type: ignore[arg-type]
386
+ swap=swap,
387
+ offset=offset,
388
+ audiofile=str(keyword) == "AUDIOFILE",
389
+ )
390
+ return _Statement(statement, meta.line)
391
+
392
+ def datafile(self, meta: Meta, children: list[_Child]) -> _Statement:
393
+ filename_token = children[0]
394
+ assert isinstance(filename_token, Token)
395
+ rest = children[1:]
396
+ offset: int | None = None
397
+ if rest and isinstance(rest[0], _Offset):
398
+ offset = rest[0].value
399
+ rest = rest[1:]
400
+ statement = DataFile(
401
+ filename=self._string(filename_token).text,
402
+ length=rest[0] if rest else None, # type: ignore[arg-type]
403
+ offset=offset,
404
+ )
405
+ return _Statement(statement, meta.line)
406
+
407
+ def fifo(self, meta: Meta, children: list[_Child]) -> _Statement:
408
+ filename_token = children[0]
409
+ assert isinstance(filename_token, Token)
410
+ statement = Fifo(
411
+ filename=self._string(filename_token).text,
412
+ length=children[1], # type: ignore[arg-type]
413
+ )
414
+ return _Statement(statement, meta.line)
415
+
416
+ def silence(self, meta: Meta, children: list[Time]) -> _Statement:
417
+ with self._at(meta.line):
418
+ length = validate_non_zero_length(children[0], SILENCE_IS_ZERO)
419
+ return _Statement(Silence(length=length), meta.line)
420
+
421
+ def zero(self, meta: Meta, children: list[_Child]) -> _Statement:
422
+ data_mode: DataMode | None = None
423
+ sub_channel_mode: SubChannelMode | None = None
424
+ rest = list(children)
425
+ if rest and isinstance(rest[0], Token) and rest[0].type == "MODE":
426
+ data_mode = DataMode(str(rest[0]))
427
+ rest = rest[1:]
428
+ if rest and isinstance(rest[0], Token) and rest[0].type == "SUB_CHANNEL_MODE":
429
+ sub_channel_mode = SubChannelMode(str(rest[0]))
430
+ rest = rest[1:]
431
+ with self._at(meta.line):
432
+ length = validate_non_zero_length(
433
+ rest[0], # type: ignore[arg-type]
434
+ ZERO_DATA_IS_ZERO,
435
+ )
436
+ statement = Zero(
437
+ length=length, data_mode=data_mode, sub_channel_mode=sub_channel_mode
438
+ )
439
+ return _Statement(statement, meta.line)
440
+
441
+ def start(self, meta: Meta, children: list[Time]) -> _Statement:
442
+ return _Statement(Start(position=children[0] if children else None), meta.line)
443
+
444
+ def end(self, meta: Meta, children: list[Time]) -> _Statement:
445
+ return _Statement(End(position=children[0] if children else None), meta.line)
446
+
447
+ def offset(self, meta: Meta, children: list[Token]) -> _Offset:
448
+ return _Offset(int(children[0]))
449
+
450
+ def msf(self, meta: Meta, children: list[Token]) -> Msf:
451
+ with self._at(meta.line):
452
+ return Msf.parse(str(children[0]))
453
+
454
+ def time(self, meta: Meta, children: list[Token]) -> Time:
455
+ token = children[0]
456
+ if token.type == "MSF":
457
+ with self._at(meta.line):
458
+ return Msf.parse(str(token))
459
+ return int(token)
460
+
461
+ # -- CD-TEXT ---------------------------------------------------------
462
+
463
+ def cd_text_global(self, meta: Meta, children: list[_Child]) -> CdText:
464
+ language_map: dict[int, int] = {}
465
+ blocks: list[_CdTextBlock] = []
466
+ for child in children:
467
+ if isinstance(child, dict):
468
+ language_map = child
469
+ else:
470
+ assert isinstance(child, _CdTextBlock)
471
+ blocks.append(child)
472
+ # Only the disc's blocks set encodings; its tracks follow them.
473
+ for block in blocks:
474
+ if block.encoding is not None:
475
+ self._encodings[block.number] = block.encoding
476
+ built, lines = self._cd_text_blocks(blocks)
477
+ self._lines.update(lines)
478
+ with self._located(lines, meta.line, ("cd_text",)):
479
+ return CdText(language_map=language_map, blocks=built)
480
+
481
+ def cd_text_track(self, meta: Meta, children: list[_Child]) -> _TrackCdText:
482
+ blocks = [child for child in children if isinstance(child, _CdTextBlock)]
483
+ built, lines = self._cd_text_blocks(blocks)
484
+ with self._located(lines, meta.line, ("cd_text",)):
485
+ return _TrackCdText(CdText(blocks=built), lines)
486
+
487
+ def _cd_text_blocks(
488
+ self, blocks: list[_CdTextBlock]
489
+ ) -> tuple[dict[int, CdTextBlock], dict[Loc, int]]:
490
+ """Merge blocks the way cdrdao's CD-TEXT container does.
491
+
492
+ cdrdao files every item under its language and pack type, so a block
493
+ number given twice adds to the first block, and an item given twice,
494
+ under either spelling of its pack, keeps the last value. cdrdao drops an
495
+ empty string before filing it, so one never replaces an earlier value;
496
+ a lone empty string is kept, so that it is written back.
497
+ """
498
+ encodings: dict[int, CdTextEncoding | None] = {}
499
+ items: dict[int, dict[CdTextItemName, CdTextValue]] = {}
500
+ lines: dict[Loc, int] = {}
501
+ for block in blocks:
502
+ if block.encoding is not None or block.number not in encodings:
503
+ encodings[block.number] = block.encoding
504
+ block_items = items.setdefault(block.number, {})
505
+ encoding = self._encodings.get(block.number, CdTextEncoding.ISO_8859_1)
506
+ for item in block.items:
507
+ value: CdTextValue
508
+ if isinstance(item.value, _String):
509
+ with self._at(item.line):
510
+ value = _cd_text_string(item.value, encoding)
511
+ else:
512
+ value = item.value
513
+ alias = CD_TEXT_ALIASES.get(item.name)
514
+ if value == "" and (item.name in block_items or alias in block_items):
515
+ continue
516
+ if alias is not None and alias in block_items:
517
+ del block_items[alias]
518
+ del lines[_item_loc(block.number, alias)]
519
+ block_items[item.name] = value
520
+ lines[_item_loc(block.number, item.name)] = item.line
521
+ built = {
522
+ number: CdTextBlock(encoding=encodings[number], items=block_items)
523
+ for number, block_items in items.items()
524
+ }
525
+ return built, lines
526
+
527
+ def language_map(
528
+ self, meta: Meta, children: list[tuple[int, int]]
529
+ ) -> dict[int, int]:
530
+ return dict(children)
531
+
532
+ def language_map_entry(self, meta: Meta, children: list[_Child]) -> tuple[int, int]:
533
+ number_token = children[0]
534
+ assert isinstance(number_token, Token)
535
+ code = children[1]
536
+ assert isinstance(code, int)
537
+ with self._at(number_token.line):
538
+ number = validate_language_number(int(number_token))
539
+ with self._at(meta.end_line):
540
+ return number, validate_language_code(code)
541
+
542
+ def language_code(self, meta: Meta, children: list[Token]) -> int:
543
+ token = children[0]
544
+ return LANGUAGE_CODE_EN if token.type == "LANGUAGE_EN" else int(token)
545
+
546
+ def cd_text_block(self, meta: Meta, children: list[_Child]) -> _CdTextBlock:
547
+ number_token = children[0]
548
+ assert isinstance(number_token, Token)
549
+ with self._at(number_token.line):
550
+ number = validate_block_number(int(number_token))
551
+ encoding: CdTextEncoding | None = None
552
+ items: list[_CdTextItem] = []
553
+ for child in children[1:]:
554
+ if isinstance(child, Token):
555
+ encoding = CdTextEncoding(str(child))
556
+ else:
557
+ assert isinstance(child, _CdTextItem)
558
+ items.append(child)
559
+ return _CdTextBlock(number, encoding, items)
560
+
561
+ def cd_text_item(self, meta: Meta, children: list[_Child]) -> _CdTextItem:
562
+ name_token = children[0]
563
+ assert isinstance(name_token, Token)
564
+ value = children[1]
565
+ assert isinstance(value, (_String, list))
566
+ return _CdTextItem(CdTextItemName(str(name_token)), value, _line(name_token))
567
+
568
+ def cd_text_value(self, meta: Meta, children: list[_Child]) -> _String | list[int]:
569
+ child = children[0]
570
+ if isinstance(child, Token):
571
+ return self._string(child)
572
+ assert isinstance(child, list)
573
+ return child
574
+
575
+ def binary_data(self, meta: Meta, children: list[Token]) -> list[int]:
576
+ values: list[int] = []
577
+ for position, token in enumerate(children):
578
+ with self._at(token.line):
579
+ values.append(validate_binary_value(int(token)))
580
+ if position == MAX_BINARY_LENGTH:
581
+ raise TocValidationError(BINARY_DATA_TOO_LONG)
582
+ return values
583
+
584
+
585
+ def parse(text: str, *, filename: str | None = None) -> Toc:
586
+ """Parse TOC file *text*.
587
+
588
+ Raises :class:`~tocparser.errors.TocParseError` for syntax errors and
589
+ :class:`~tocparser.errors.TocValidationError` for input that parses but
590
+ breaks a rule cdrdao enforces.
591
+ """
592
+ try:
593
+ tree = _get_parser().parse(text)
594
+ except UnexpectedInput as exc:
595
+ raise _syntax_error(exc, text, filename) from None
596
+
597
+ try:
598
+ result = _TocTransformer(filename).transform(tree)
599
+ except VisitError as exc:
600
+ raise _unwrap(exc) from None
601
+ assert isinstance(result, Toc)
602
+ return result
603
+
604
+
605
+ def parse_file(path: str | PathLike[str], *, encoding: str = "utf-8") -> Toc:
606
+ """Parse the TOC file at *path*.
607
+
608
+ cdrdao 1.2.5 and later write TOC files in UTF-8. Bytes that do not decode
609
+ in ``encoding`` raise :class:`~tocparser.errors.TocParseError`.
610
+ """
611
+ file = Path(path)
612
+ data = file.read_bytes()
613
+ try:
614
+ text = data.decode(encoding)
615
+ except UnicodeDecodeError as exc:
616
+ # Everything before the bad byte decodes, and gives its line.
617
+ before = _universal_newlines(data[: exc.start].decode(encoding))
618
+ raise TocParseError(
619
+ f"Cannot decode byte 0x{data[exc.start]:02x} as {encoding}",
620
+ line=before.count("\n") + 1,
621
+ filename=str(file),
622
+ ) from None
623
+ return parse(_universal_newlines(text), filename=str(file))
624
+
625
+
626
+ def _universal_newlines(text: str) -> str:
627
+ """Line endings as a file opened in text mode would give them."""
628
+ return text.replace("\r\n", "\n").replace("\r", "\n")
629
+
630
+
631
+ def _unwrap(error: VisitError) -> BaseException:
632
+ original: BaseException = error
633
+ while isinstance(original, VisitError):
634
+ original = original.orig_exc
635
+ return original if isinstance(original, TocError) else error
636
+
637
+
638
+ def _syntax_error(
639
+ error: UnexpectedInput, text: str, filename: str | None
640
+ ) -> TocParseError:
641
+ """Word a syntax error the way cdrdao does."""
642
+ if isinstance(error, UnexpectedEOF) or (
643
+ isinstance(error, UnexpectedToken) and error.token.type == "$END"
644
+ ):
645
+ return TocParseError(
646
+ 'syntax error at "EOF"', line=text.count("\n") + 1, filename=filename
647
+ )
648
+ if isinstance(error, UnexpectedToken):
649
+ # cdrdao's scanner makes a string's opening quote a token of its own.
650
+ shown = '"' if error.token.type == "STRING" else str(error.token)
651
+ return TocParseError(
652
+ f'syntax error at "{shown}"',
653
+ line=error.line,
654
+ column=error.column,
655
+ filename=filename,
656
+ )
657
+ if isinstance(error, UnexpectedCharacters):
658
+ if error.char == '"':
659
+ return _broken_string(error, text, filename)
660
+ return TocParseError(
661
+ f"Illegal token: {error.char}",
662
+ line=error.line,
663
+ column=error.column,
664
+ filename=filename,
665
+ )
666
+ return TocParseError("syntax error", filename=filename) # pragma: no cover
667
+
668
+
669
+ def _broken_string(
670
+ error: UnexpectedCharacters, text: str, filename: str | None
671
+ ) -> TocParseError:
672
+ """A string that does not lex: an illegal backslash, or no closing quote."""
673
+ position = error.pos_in_stream
674
+ assert position is not None
675
+ prefix = _STRING_PREFIX_RE.match(text, position)
676
+ assert prefix is not None
677
+ end = prefix.end()
678
+ if end == len(text):
679
+ return TocParseError(
680
+ 'syntax error at "EOF"', line=error.line, filename=filename
681
+ )
682
+ if text[end] == "\\":
683
+ # A backslash starting no escape cdrdao knows.
684
+ return TocParseError(
685
+ "Illegal token: \\",
686
+ line=text.count("\n", 0, end) + 1,
687
+ column=end - text.rfind("\n", 0, end),
688
+ filename=filename,
689
+ )
690
+ # A complete string where the grammar allows none. Lark's contextual lexer
691
+ # reports that as an unexpected STRING token instead, so this only guards
692
+ # against the lexer changing its mind.
693
+ return TocParseError( # pragma: no cover
694
+ 'syntax error at """', line=error.line, column=error.column, filename=filename
695
+ )
tocparser/py.typed ADDED
File without changes