tocparser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tocparser/__init__.py +88 -0
- tocparser/errors.py +80 -0
- tocparser/grammar.lark +89 -0
- tocparser/models.py +739 -0
- tocparser/parser.py +695 -0
- tocparser/py.typed +0 -0
- tocparser/serializer.py +243 -0
- tocparser/times.py +92 -0
- tocparser-0.1.0.dist-info/METADATA +205 -0
- tocparser-0.1.0.dist-info/RECORD +12 -0
- tocparser-0.1.0.dist-info/WHEEL +4 -0
- tocparser-0.1.0.dist-info/licenses/LICENSE.txt +201 -0
tocparser/serializer.py
ADDED
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
"""Render :class:`~tocparser.models.Toc` objects back to TOC file text.
|
|
2
|
+
|
|
3
|
+
The layout follows the files cdrdao itself writes: two-space nesting inside
|
|
4
|
+
``CD_TEXT``, binary CD-TEXT arrays wrapped at twelve values per line, and both
|
|
5
|
+
of the comments cdrdao generates -- a ``// Track N`` header before each track
|
|
6
|
+
and a ``// length in bytes:`` annotation on data lengths. Comments written by
|
|
7
|
+
hand are not preserved, so serializing is meaning-preserving rather than
|
|
8
|
+
byte-exact. Text is written as UTF-8, as cdrdao 1.2.5 and later write it.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from os import PathLike
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from tocparser.errors import TocValidationError
|
|
18
|
+
from tocparser.models import (
|
|
19
|
+
CdText,
|
|
20
|
+
CdTextValue,
|
|
21
|
+
DataFile,
|
|
22
|
+
Fifo,
|
|
23
|
+
File,
|
|
24
|
+
Silence,
|
|
25
|
+
Start,
|
|
26
|
+
Toc,
|
|
27
|
+
Track,
|
|
28
|
+
TrackMode,
|
|
29
|
+
TrackStatement,
|
|
30
|
+
Zero,
|
|
31
|
+
)
|
|
32
|
+
from tocparser.times import Msf, Time
|
|
33
|
+
|
|
34
|
+
__all__ = ["dump", "dumps"]
|
|
35
|
+
|
|
36
|
+
_INDENT = " "
|
|
37
|
+
_BINARY_VALUES_PER_LINE = 12
|
|
38
|
+
_BINARY_CONTINUATION_INDENT = " " * 15
|
|
39
|
+
|
|
40
|
+
# A backslash before three digits, any other backslash, and a quote.
|
|
41
|
+
_TO_ESCAPE_RE = re.compile(r'(\\(?=[0-9]{3}))|(\\)|(")')
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def escape(text: str) -> str:
|
|
45
|
+
"""Escape *text* so cdrdao reads it back unchanged.
|
|
46
|
+
|
|
47
|
+
A quote and a backslash are escaped with a backslash, except a backslash
|
|
48
|
+
followed by three digits: cdrdao reads that as an octal escape even when
|
|
49
|
+
the backslash was itself escaped, so it is written as ``\\134``, the octal
|
|
50
|
+
escape for a backslash. cdrdao rejects octal escapes in text holding
|
|
51
|
+
non-ASCII characters, so such text cannot be written at all.
|
|
52
|
+
"""
|
|
53
|
+
|
|
54
|
+
def replace(match: re.Match[str]) -> str:
|
|
55
|
+
if match[1] is not None:
|
|
56
|
+
if not text.isascii():
|
|
57
|
+
raise TocValidationError(
|
|
58
|
+
f"Cannot write {text!r}: a backslash followed by three digits "
|
|
59
|
+
"needs an octal escape, which cdrdao rejects in non-ASCII text."
|
|
60
|
+
)
|
|
61
|
+
return "\\134"
|
|
62
|
+
return "\\" + match[0]
|
|
63
|
+
|
|
64
|
+
return _TO_ESCAPE_RE.sub(replace, text)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _quote(text: str) -> str:
|
|
68
|
+
return f'"{escape(text)}"'
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _binary(name: str, values: list[int], indent: str) -> list[str]:
|
|
72
|
+
if not values:
|
|
73
|
+
return [f"{indent}{name} {{}}"]
|
|
74
|
+
lines: list[str] = []
|
|
75
|
+
for offset in range(0, len(values), _BINARY_VALUES_PER_LINE):
|
|
76
|
+
chunk = values[offset : offset + _BINARY_VALUES_PER_LINE]
|
|
77
|
+
if offset == 0:
|
|
78
|
+
# The opening brace supplies the padding of the first value, so
|
|
79
|
+
# cdrdao writes that one bare and pads the rest to two columns.
|
|
80
|
+
rendered = str(chunk[0]) + "".join(f", {value:2d}" for value in chunk[1:])
|
|
81
|
+
prefix = f"{indent}{name} {{ "
|
|
82
|
+
else:
|
|
83
|
+
# On a continued line every value is padded, the first included,
|
|
84
|
+
# which only shows once a line begins with more than one digit.
|
|
85
|
+
rendered = ", ".join(f"{value:2d}" for value in chunk)
|
|
86
|
+
prefix = _BINARY_CONTINUATION_INDENT
|
|
87
|
+
suffix = "," if offset + _BINARY_VALUES_PER_LINE < len(values) else "}"
|
|
88
|
+
lines.append(f"{prefix}{rendered}{suffix}")
|
|
89
|
+
return lines
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _cd_text_value(name: str, value: CdTextValue, indent: str) -> list[str]:
|
|
93
|
+
if isinstance(value, str):
|
|
94
|
+
return [f"{indent}{name} {_quote(value)}"]
|
|
95
|
+
return _binary(name, value, indent)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _cd_text(cd_text: CdText) -> list[str]:
|
|
99
|
+
lines = ["CD_TEXT {"]
|
|
100
|
+
if cd_text.language_map:
|
|
101
|
+
lines.append(f"{_INDENT}LANGUAGE_MAP {{")
|
|
102
|
+
for number, code in cd_text.language_map.items():
|
|
103
|
+
lines.append(f"{_INDENT * 2}{number}: {code}")
|
|
104
|
+
lines.append(f"{_INDENT}}}")
|
|
105
|
+
for number, block in cd_text.blocks.items():
|
|
106
|
+
lines.append(f"{_INDENT}LANGUAGE {number} {{")
|
|
107
|
+
if block.encoding is not None:
|
|
108
|
+
lines.append(f"{_INDENT * 2}{block.encoding.value}")
|
|
109
|
+
for name, value in block.items.items():
|
|
110
|
+
lines.extend(_cd_text_value(name.value, value, _INDENT * 2))
|
|
111
|
+
lines.append(f"{_INDENT}}}")
|
|
112
|
+
lines.append("}")
|
|
113
|
+
return lines
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _byte_length(length: Time, track: Track) -> int:
|
|
117
|
+
"""The count cdrdao annotates a data length with.
|
|
118
|
+
|
|
119
|
+
On an audio track cdrdao counts samples rather than bytes while still
|
|
120
|
+
labeling the comment "length in bytes"; that quirk is reproduced here so
|
|
121
|
+
the output matches.
|
|
122
|
+
"""
|
|
123
|
+
if not isinstance(length, Msf):
|
|
124
|
+
return length
|
|
125
|
+
if track.mode is TrackMode.AUDIO:
|
|
126
|
+
return length.to_samples()
|
|
127
|
+
return length.to_bytes(track.block_size)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _length_comment(length: Time, track: Track) -> str:
|
|
131
|
+
return f" // length in bytes: {_byte_length(length, track)}"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _statement(statement: TrackStatement, track: Track) -> str:
|
|
135
|
+
if isinstance(statement, File):
|
|
136
|
+
parts = [
|
|
137
|
+
"AUDIOFILE" if statement.audiofile else "FILE",
|
|
138
|
+
_quote(statement.filename),
|
|
139
|
+
]
|
|
140
|
+
if statement.swap:
|
|
141
|
+
parts.append("SWAP")
|
|
142
|
+
if statement.offset is not None:
|
|
143
|
+
parts.append(f"#{statement.offset}")
|
|
144
|
+
parts.append(str(statement.start))
|
|
145
|
+
if statement.length is not None:
|
|
146
|
+
parts.append(str(statement.length))
|
|
147
|
+
return " ".join(parts)
|
|
148
|
+
|
|
149
|
+
if isinstance(statement, DataFile):
|
|
150
|
+
parts = ["DATAFILE", _quote(statement.filename)]
|
|
151
|
+
if statement.offset is not None:
|
|
152
|
+
parts.append(f"#{statement.offset}")
|
|
153
|
+
if statement.length is None:
|
|
154
|
+
return " ".join(parts)
|
|
155
|
+
parts.append(str(statement.length))
|
|
156
|
+
line = " ".join(parts)
|
|
157
|
+
# cdrdao annotates data lengths, but not ones it wrote as FILE.
|
|
158
|
+
if track.mode is not TrackMode.AUDIO:
|
|
159
|
+
line += _length_comment(statement.length, track)
|
|
160
|
+
return line
|
|
161
|
+
|
|
162
|
+
if isinstance(statement, Fifo):
|
|
163
|
+
line = f"FIFO {_quote(statement.filename)} {statement.length}"
|
|
164
|
+
# cdrdao only annotates a FIFO whose length lands on a block boundary,
|
|
165
|
+
# which is exactly when it writes the length as MSF.
|
|
166
|
+
if isinstance(statement.length, Msf):
|
|
167
|
+
line += _length_comment(statement.length, track)
|
|
168
|
+
return line
|
|
169
|
+
|
|
170
|
+
if isinstance(statement, Silence):
|
|
171
|
+
return f"SILENCE {statement.length}"
|
|
172
|
+
|
|
173
|
+
if isinstance(statement, Zero):
|
|
174
|
+
parts = ["ZERO"]
|
|
175
|
+
if statement.data_mode is not None:
|
|
176
|
+
parts.append(statement.data_mode.value)
|
|
177
|
+
if statement.sub_channel_mode is not None:
|
|
178
|
+
parts.append(statement.sub_channel_mode.value)
|
|
179
|
+
parts.append(str(statement.length))
|
|
180
|
+
return " ".join(parts)
|
|
181
|
+
|
|
182
|
+
if isinstance(statement, Start):
|
|
183
|
+
return "START" if statement.position is None else f"START {statement.position}"
|
|
184
|
+
|
|
185
|
+
# Every other statement returned above, so this is an End.
|
|
186
|
+
return "END" if statement.position is None else f"END {statement.position}"
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _track(track: Track) -> list[str]:
|
|
190
|
+
header = f"TRACK {track.mode.value}"
|
|
191
|
+
if track.sub_channel_mode is not None:
|
|
192
|
+
header += f" {track.sub_channel_mode.value}"
|
|
193
|
+
lines = [header]
|
|
194
|
+
|
|
195
|
+
if track.copy_permitted is not None:
|
|
196
|
+
lines.append("COPY" if track.copy_permitted else "NO COPY")
|
|
197
|
+
if track.pre_emphasis is not None:
|
|
198
|
+
lines.append("PRE_EMPHASIS" if track.pre_emphasis else "NO PRE_EMPHASIS")
|
|
199
|
+
if track.channels is not None:
|
|
200
|
+
lines.append(
|
|
201
|
+
"FOUR_CHANNEL_AUDIO" if track.channels == 4 else "TWO_CHANNEL_AUDIO"
|
|
202
|
+
)
|
|
203
|
+
if track.isrc is not None:
|
|
204
|
+
lines.append(f"ISRC {_quote(track.isrc)}")
|
|
205
|
+
if track.cd_text is not None:
|
|
206
|
+
lines.extend(_cd_text(track.cd_text))
|
|
207
|
+
if track.pregap is not None:
|
|
208
|
+
lines.append(f"PREGAP {track.pregap}")
|
|
209
|
+
|
|
210
|
+
lines.extend(_statement(statement, track) for statement in track.statements)
|
|
211
|
+
lines.extend(f"INDEX {index}" for index in track.indexes)
|
|
212
|
+
return lines
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def dumps(toc: Toc) -> str:
|
|
216
|
+
"""Render *toc* as TOC file text.
|
|
217
|
+
|
|
218
|
+
Models can be changed after they are built, so *toc* is validated again
|
|
219
|
+
first: an invalid one raises instead of producing a file cdrdao rejects.
|
|
220
|
+
"""
|
|
221
|
+
toc = Toc.model_validate(toc.model_dump())
|
|
222
|
+
lines = [flag.value for flag in toc.superseded_disc_types]
|
|
223
|
+
lines += [toc.disc_type.value, ""]
|
|
224
|
+
if toc.catalog is not None:
|
|
225
|
+
lines.append(f"CATALOG {_quote(toc.catalog)}")
|
|
226
|
+
if toc.first_track_number is not None:
|
|
227
|
+
lines.append(f"FIRST_TRACK_NO {toc.first_track_number}")
|
|
228
|
+
if toc.cd_text is not None:
|
|
229
|
+
lines.extend(_cd_text(toc.cd_text))
|
|
230
|
+
|
|
231
|
+
number = toc.first_track_number if toc.first_track_number is not None else 1
|
|
232
|
+
for offset, track in enumerate(toc.tracks):
|
|
233
|
+
lines.append("")
|
|
234
|
+
lines.append(f"// Track {number + offset}")
|
|
235
|
+
lines.extend(_track(track))
|
|
236
|
+
lines.append("")
|
|
237
|
+
|
|
238
|
+
return "\n".join(lines) + "\n"
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def dump(toc: Toc, path: str | PathLike[str], *, encoding: str = "utf-8") -> None:
|
|
242
|
+
"""Write *toc* to the file at *path*."""
|
|
243
|
+
Path(path).write_text(dumps(toc), encoding=encoding)
|
tocparser/times.py
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Time values used in TOC files.
|
|
2
|
+
|
|
3
|
+
A position or a length is written either as an ``MM:SS:FF`` triple counting
|
|
4
|
+
frames (also called blocks) at 75 frames per second, or as a bare integer.
|
|
5
|
+
A bare integer means *samples* for audio tracks and *bytes* for data tracks,
|
|
6
|
+
so the two forms are not interchangeable: a sample count that is not frame
|
|
7
|
+
aligned has no MSF representation. Both forms are therefore preserved as
|
|
8
|
+
written, and :data:`Time` is the union of the two.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import re
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
from typing import TypeAlias
|
|
16
|
+
|
|
17
|
+
from tocparser.errors import TocValidationError
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"FRAMES_PER_SECOND",
|
|
21
|
+
"SAMPLES_PER_FRAME",
|
|
22
|
+
"SECONDS_PER_MINUTE",
|
|
23
|
+
"Msf",
|
|
24
|
+
"Time",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
FRAMES_PER_SECOND = 75
|
|
28
|
+
SECONDS_PER_MINUTE = 60
|
|
29
|
+
#: 44100 Hz / 75 frames per second.
|
|
30
|
+
SAMPLES_PER_FRAME = 588
|
|
31
|
+
|
|
32
|
+
_MSF_RE = re.compile(r"\A([0-9]+):([0-9]+):([0-9]+)\Z")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass(frozen=True, order=True)
|
|
36
|
+
class Msf:
|
|
37
|
+
"""A position or length expressed as minutes, seconds and frames."""
|
|
38
|
+
|
|
39
|
+
minutes: int
|
|
40
|
+
seconds: int
|
|
41
|
+
frames: int
|
|
42
|
+
|
|
43
|
+
def __post_init__(self) -> None:
|
|
44
|
+
for field in (self.minutes, self.seconds, self.frames):
|
|
45
|
+
if isinstance(field, bool) or not isinstance(field, int):
|
|
46
|
+
raise TypeError(f"MSF fields must be integers, not {field!r}")
|
|
47
|
+
if self.minutes < 0:
|
|
48
|
+
raise TocValidationError(f"Illegal minute field: {self.minutes}")
|
|
49
|
+
if not 0 <= self.seconds < SECONDS_PER_MINUTE:
|
|
50
|
+
raise TocValidationError(f"Illegal second field: {self.seconds}")
|
|
51
|
+
if not 0 <= self.frames < FRAMES_PER_SECOND:
|
|
52
|
+
raise TocValidationError(f"Illegal fraction field: {self.frames}")
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def total_frames(self) -> int:
|
|
56
|
+
"""The value as a plain frame (block) count."""
|
|
57
|
+
return (
|
|
58
|
+
self.minutes * SECONDS_PER_MINUTE + self.seconds
|
|
59
|
+
) * FRAMES_PER_SECOND + self.frames
|
|
60
|
+
|
|
61
|
+
@classmethod
|
|
62
|
+
def from_frames(cls, frames: int) -> Msf:
|
|
63
|
+
"""Build an :class:`Msf` from a frame count."""
|
|
64
|
+
if frames < 0:
|
|
65
|
+
raise TocValidationError(f"Illegal frame count: {frames}")
|
|
66
|
+
minutes, rest = divmod(frames, FRAMES_PER_SECOND * SECONDS_PER_MINUTE)
|
|
67
|
+
seconds, remaining_frames = divmod(rest, FRAMES_PER_SECOND)
|
|
68
|
+
return cls(minutes, seconds, remaining_frames)
|
|
69
|
+
|
|
70
|
+
@classmethod
|
|
71
|
+
def parse(cls, text: str) -> Msf:
|
|
72
|
+
"""Parse an ``MM:SS:FF`` string."""
|
|
73
|
+
match = _MSF_RE.match(text)
|
|
74
|
+
if match is None:
|
|
75
|
+
raise TocValidationError(f"Illegal MSF value: {text}")
|
|
76
|
+
return cls(int(match[1]), int(match[2]), int(match[3]))
|
|
77
|
+
|
|
78
|
+
def to_samples(self) -> int:
|
|
79
|
+
"""The value as a count of 16-bit stereo samples."""
|
|
80
|
+
return self.total_frames * SAMPLES_PER_FRAME
|
|
81
|
+
|
|
82
|
+
def to_bytes(self, block_size: int) -> int:
|
|
83
|
+
"""The value as a byte count, given a track's block size."""
|
|
84
|
+
return self.total_frames * block_size
|
|
85
|
+
|
|
86
|
+
def __str__(self) -> str:
|
|
87
|
+
return f"{self.minutes:02d}:{self.seconds:02d}:{self.frames:02d}"
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
#: A position or length, either frame-based (:class:`Msf`) or a raw
|
|
91
|
+
#: sample/byte count.
|
|
92
|
+
Time: TypeAlias = Msf | int
|
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tocparser
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Parse, validate, and write cdrdao CD TOC files
|
|
5
|
+
Keywords: audio,cd,cd-text,cdrdao,parser,toc
|
|
6
|
+
Author: Jean-Marc Fontaine
|
|
7
|
+
Author-email: Jean-Marc Fontaine <jm@jmfontaine.net>
|
|
8
|
+
License-Expression: Apache-2.0
|
|
9
|
+
License-File: LICENSE.txt
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.15
|
|
20
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: CD Audio
|
|
21
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Dist: lark>=1.2
|
|
24
|
+
Requires-Dist: pydantic>=2.9
|
|
25
|
+
Requires-Dist: pydantic>=2.14.0b1,<2.15 ; python_full_version == '3.15.*'
|
|
26
|
+
Requires-Python: >=3.10
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# tocparser
|
|
30
|
+
|
|
31
|
+
tocparser parses cdrdao's TOC files into Pydantic models and writes them back out.
|
|
32
|
+
|
|
33
|
+
- It supports the full TOC format. Older files mostly parse the same, with two
|
|
34
|
+
exceptions described below.
|
|
35
|
+
- It validates files like cdrdao, matching its error messages and line numbers. See
|
|
36
|
+
Limits for exceptions.
|
|
37
|
+
- Writing preserves every value. Unedited files usually serialize byte for byte
|
|
38
|
+
identically, though hand-written comments are dropped and some formatting is
|
|
39
|
+
normalized. See What Is Written Back for details.
|
|
40
|
+
- It is fully typed and depends only on `lark` and `pydantic`.
|
|
41
|
+
|
|
42
|
+
## Install
|
|
43
|
+
|
|
44
|
+
```
|
|
45
|
+
uv add tocparser
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Requires Python 3.10 or newer.
|
|
49
|
+
|
|
50
|
+
## Usage
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
from tocparser import dump, parse_file
|
|
54
|
+
|
|
55
|
+
toc = parse_file("the_downward_spiral.toc")
|
|
56
|
+
|
|
57
|
+
toc.disc_type # <DiscType.CD_DA: 'CD_DA'>
|
|
58
|
+
toc.catalog # '0602498647295'
|
|
59
|
+
toc.cd_text[0].title # 'The Downward Spiral - Deluxe Edition [ Disc 1 ] '
|
|
60
|
+
toc.cd_text[0].performer # 'Nine Inch Nails'
|
|
61
|
+
|
|
62
|
+
track = toc.tracks[4]
|
|
63
|
+
track.mode # <TrackMode.AUDIO: 'AUDIO'>
|
|
64
|
+
track.isrc # 'USIR19400529'
|
|
65
|
+
track.cd_text[0].title # 'Closer'
|
|
66
|
+
track.content[0].start # Msf(minutes=15, seconds=47, frames=32)
|
|
67
|
+
track.content[0].length # Msf(minutes=6, seconds=13, frames=23)
|
|
68
|
+
|
|
69
|
+
# "The Becoming" starts with a 00:02:30 pre-gap: 2 seconds and 30 frames.
|
|
70
|
+
toc.tracks[6].start # Start(kind='start', position=Msf(minutes=0, seconds=2, frames=30))
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
Models are ordinary Pydantic models, so `model_dump()`, `model_dump_json()` and
|
|
74
|
+
`Toc.model_validate_json()` all work and preserve the parsed values.
|
|
75
|
+
|
|
76
|
+
### Times
|
|
77
|
+
|
|
78
|
+
A position or length can be written as `MM:SS:FF`, where frames are counted at 75 per
|
|
79
|
+
second, or as a bare integer representing samples on an audio track and bytes on a data
|
|
80
|
+
track. These formats are not interchangeable. If a sample count is not frame-aligned, it
|
|
81
|
+
has no `MM:SS:FF` equivalent, so tocparser preserves the format in which it was written.
|
|
82
|
+
Frame-based values become `Msf`, while bare integers remain `int`.
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
from tocparser import Msf
|
|
86
|
+
|
|
87
|
+
position = Msf(15, 47, 32) # where "Closer" starts
|
|
88
|
+
position.total_frames # 71057
|
|
89
|
+
position.to_samples() # 41781516
|
|
90
|
+
position.to_bytes(2352) # 167126064
|
|
91
|
+
str(position) # '15:47:32'
|
|
92
|
+
Msf.from_frames(71057) # Msf(minutes=15, seconds=47, frames=32)
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
`Msf` is ordered and hashable, so positions sort and can be used as keys.
|
|
96
|
+
|
|
97
|
+
### CD-TEXT
|
|
98
|
+
|
|
99
|
+
CD-TEXT values are `str`, or `list[int]` for binary packs. Each language block uses
|
|
100
|
+
ISO-8859-1 unless the disc's `CD_TEXT` specifies an `ENCODING_*` value. Track blocks
|
|
101
|
+
follow the disc block with the same number, as they do in cdrdao.
|
|
102
|
+
`toc.cd_text_encoding(n)` returns the encoding used for block `n`.
|
|
103
|
+
|
|
104
|
+
Strings follow cdrdao's rules. Non-ASCII text is UTF-8 and must fit the block's
|
|
105
|
+
encoding: `TITLE "日本"` requires `ENCODING_MS_JIS`, because ISO-8859-1 cannot represent
|
|
106
|
+
Japanese. Plain ASCII strings may contain `\NNN` octal escapes, which represent bytes in
|
|
107
|
+
the block's encoding. For example, `"caf\351"` decodes to `'café'`, and under
|
|
108
|
+
`ENCODING_MS_JIS`, `"\223\372\226\173"` decodes to `'日本'`.
|
|
109
|
+
|
|
110
|
+
cdrdao treats `UPC_EAN` and `ISRC` as the same pack, as well as `RESERVED4` and
|
|
111
|
+
`CLOSED`. A block retains the spelling written last, and `block.upc_ean` and
|
|
112
|
+
`block.isrc` both access it.
|
|
113
|
+
|
|
114
|
+
### Errors
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
from tocparser import TocParseError, TocValidationError, parse
|
|
118
|
+
|
|
119
|
+
parse(
|
|
120
|
+
'CD_DA\nCATALOG "0602498647"\nTRACK AUDIO\nFILE "hurt.wav" 0\n',
|
|
121
|
+
filename="the_downward_spiral.toc",
|
|
122
|
+
)
|
|
123
|
+
# TocValidationError: the_downward_spiral.toc:2: Illegal catalog number: 0602498647.
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
`TocParseError` covers syntax errors and `TocValidationError` covers input that parses
|
|
127
|
+
but breaks a rule cdrdao enforces. Both derive from `TocError` and carry `line`, and
|
|
128
|
+
`TocParseError` also carries `column`. `parse_file` reads files as UTF-8 and raises
|
|
129
|
+
`TocParseError` for bytes that do not decode; pass `encoding=` for other encodings.
|
|
130
|
+
|
|
131
|
+
The models enforce the same rules, so building one by hand that cdrdao would reject
|
|
132
|
+
raises `TocValidationError`, whose `loc` locates the problem inside that model, e.g.
|
|
133
|
+
`("statements", 2)`. A value cdrdao's syntax cannot express, such as a value of the
|
|
134
|
+
wrong type or a negative number, raises Pydantic's `ValidationError` instead. Models can
|
|
135
|
+
be changed after they are built, so `dumps` validates its argument again before writing
|
|
136
|
+
it.
|
|
137
|
+
|
|
138
|
+
## What Is Written Back
|
|
139
|
+
|
|
140
|
+
Serializing preserves meaning, though not always the exact bytes. In practice, the
|
|
141
|
+
output is usually byte-for-byte identical. Differences are limited to the following:
|
|
142
|
+
|
|
143
|
+
- Hand-written comments are dropped. cdrdao-generated comments—the `// Track N` headers
|
|
144
|
+
and the `// length in bytes:` annotations on data lengths—are regenerated.
|
|
145
|
+
- `MM:SS:FF` values are zero-padded: `0:10:0` becomes `00:10:00`. Bare integers remain
|
|
146
|
+
unchanged.
|
|
147
|
+
- When a CD-TEXT item appears more than once, only its last instance is written. An
|
|
148
|
+
empty string does not replace an earlier value because cdrdao drops it. Repeated
|
|
149
|
+
LANGUAGE blocks are merged into the first, as in cdrdao's output.
|
|
150
|
+
- Text is written as UTF-8, so `\NNN` escapes are replaced by the characters they
|
|
151
|
+
represent. A backslash followed by three digits is written as `\134`, because cdrdao
|
|
152
|
+
would interpret even an escaped backslash as an octal escape. cdrdao does not accept
|
|
153
|
+
escapes in non-ASCII text, so such text cannot contain that sequence; `dumps` raises
|
|
154
|
+
an error if it does.
|
|
155
|
+
- Unwritten flags are not added. A track without a `COPY` line keeps
|
|
156
|
+
`copy_permitted is None` and has no `COPY` line in the output. The effective values
|
|
157
|
+
are available through `is_copy_permitted`, `has_pre_emphasis`, and `channel_count`.
|
|
158
|
+
|
|
159
|
+
## Limits
|
|
160
|
+
|
|
161
|
+
tocparser follows cdrdao 1.2.6, but some checks require the audio and data files
|
|
162
|
+
referenced by a TOC file. Because tocparser reads only the TOC file, it cannot check the
|
|
163
|
+
four-second minimum track length, an `INDEX` beyond the track end, `START` or `END` past
|
|
164
|
+
the track end, an `END` within the pre-gap, or a requested length that exceeds the file.
|
|
165
|
+
It also skips the CD-TEXT completeness checks that cdrdao performs only before writing a
|
|
166
|
+
disc. tocparser checks everything else that `cdrdao show-toc` checks, with these
|
|
167
|
+
differences:
|
|
168
|
+
|
|
169
|
+
- If a file contains several errors, tocparser raises one, but not necessarily the one
|
|
170
|
+
cdrdao reports first.
|
|
171
|
+
- For two syntax errors—a `PREGAP` after track data and a `LANGUAGE_MAP` inside a
|
|
172
|
+
track's `CD_TEXT`—cdrdao's parser stops at a different token than tocparser does.
|
|
173
|
+
- cdrdao reports a `FIFO` that mixes audio and data on line 0; tocparser reports the
|
|
174
|
+
`FIFO`'s own line.
|
|
175
|
+
- tocparser rejects `\NNN` bytes that are not text in their block's encoding—for
|
|
176
|
+
example, a lone CP932 lead byte or a byte above 127 in an `ENCODING_ASCII` block.
|
|
177
|
+
cdrdao keeps these bytes, at most issuing a warning, but could not read back the text
|
|
178
|
+
it would write for them.
|
|
179
|
+
- cdrdao drops CD-TEXT items with empty strings. tocparser preserves an empty item when
|
|
180
|
+
no other value is provided and writes it back. Like cdrdao, it also allows a track to
|
|
181
|
+
contain an empty item that is valid only at the disc level.
|
|
182
|
+
- Escapes in file names are decoded like CD-TEXT (as ISO-8859-1) and written back as
|
|
183
|
+
UTF-8.
|
|
184
|
+
|
|
185
|
+
## TOC Files from Older cdrdao
|
|
186
|
+
|
|
187
|
+
Files written by cdrdao 1.2.2 through 1.2.4 are mostly read the same way, with two
|
|
188
|
+
exceptions that also apply to later releases. Releases before 1.2.2 were not checked.
|
|
189
|
+
|
|
190
|
+
- cdrdao 1.2.2 through 1.2.4 wrote backslashes in strings unchanged. Since 1.2.5, a
|
|
191
|
+
backslash must begin an escape, so `"AC\DC"` is rejected with `Illegal token: \`, and
|
|
192
|
+
`"C:\\x"` now reads as `C:\x`, with one backslash rather than two. This applies to all
|
|
193
|
+
strings, including file names.
|
|
194
|
+
- cdrdao 1.2.2 through 1.2.4 wrote bytes above 127 as octal escapes, at least under the
|
|
195
|
+
default C locale, and omitted `ENCODING_*`, so the text is read as ISO-8859-1. Latin
|
|
196
|
+
text remains intact, but text in another encoding, such as Japanese, does not.
|
|
197
|
+
|
|
198
|
+
## Contributing
|
|
199
|
+
|
|
200
|
+
Contributions are welcome. See CONTRIBUTING.md for development setup, test comparisons
|
|
201
|
+
between tocparser and cdrdao, and the pull request process.
|
|
202
|
+
|
|
203
|
+
## License
|
|
204
|
+
|
|
205
|
+
tocparser is licensed under the Apache License 2.0.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
tocparser/__init__.py,sha256=Lfkz-I8Uw1zm_dPEw70Q3sueITHaSv-cywUQrfOjyAk,1507
|
|
2
|
+
tocparser/errors.py,sha256=0BytexY4nto_T_xQtK7OznGDuBUXRAMXDic2_H4tayU,2018
|
|
3
|
+
tocparser/grammar.lark,sha256=r2oZBBw5m3MMdvEwp-Ze5zeq8HuX2q7UUTq83OxamDo,3378
|
|
4
|
+
tocparser/models.py,sha256=DBMXRS8ukNiuROmYKKRP94xm-5mwl-v_04la5cEUPDY,24044
|
|
5
|
+
tocparser/parser.py,sha256=nUplKjFmSf_hir1l5iQ518GK4Kwqw8gLQ38EquQwDtE,25154
|
|
6
|
+
tocparser/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
tocparser/serializer.py,sha256=qBGJHsYYrK4Ki_un-xmZN3HWjTZ2G6IvnEK2KwbJQfY,8922
|
|
8
|
+
tocparser/times.py,sha256=fW74lwKNKswwHIIhTjuYiNTl8g34Lm7uWIoPK965Sao,3174
|
|
9
|
+
tocparser-0.1.0.dist-info/licenses/LICENSE.txt,sha256=ZoQz6-YuMRvk-tumxl2wKZptFZmNbfSXfIKx3nUz4Rs,11347
|
|
10
|
+
tocparser-0.1.0.dist-info/WHEEL,sha256=e4_1dyBeezi8ZjfxrZ3bnVOxFDa3ksqVqH0jTHkUZ3k,81
|
|
11
|
+
tocparser-0.1.0.dist-info/METADATA,sha256=6fzCg5ZC5kt_kuPVcLNkWwqLBb3ZPequpQbVZbPEat4,9478
|
|
12
|
+
tocparser-0.1.0.dist-info/RECORD,,
|