tinytag 2.1.1__tar.gz → 2.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of tinytag might be problematic. Click here for more details.
- {tinytag-2.1.1 → tinytag-2.2.0}/PKG-INFO +36 -4
- {tinytag-2.1.1 → tinytag-2.2.0}/README.md +32 -3
- {tinytag-2.1.1 → tinytag-2.2.0}/pyproject.toml +15 -2
- {tinytag-2.1.1 → tinytag-2.2.0}/tinytag/__init__.py +2 -0
- {tinytag-2.1.1 → tinytag-2.2.0}/tinytag/tinytag.py +324 -182
- {tinytag-2.1.1 → tinytag-2.2.0}/LICENSE +0 -0
- {tinytag-2.1.1 → tinytag-2.2.0}/tinytag/__main__.py +0 -0
- {tinytag-2.1.1 → tinytag-2.2.0}/tinytag/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tinytag
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.2.0
|
|
4
4
|
Summary: Read audio file metadata
|
|
5
5
|
Keywords: metadata,audio,music,mp3,m4a,wav,ogg,opus,flac,wma,aiff
|
|
6
6
|
Author: Tom Wallroth, Mat (mathiascode)
|
|
@@ -15,6 +15,7 @@ Classifier: Programming Language :: Python :: 3.10
|
|
|
15
15
|
Classifier: Programming Language :: Python :: 3.11
|
|
16
16
|
Classifier: Programming Language :: Python :: 3.12
|
|
17
17
|
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
18
19
|
Classifier: License :: OSI Approved :: MIT License
|
|
19
20
|
Classifier: Development Status :: 5 - Production/Stable
|
|
20
21
|
Classifier: Environment :: Web Environment
|
|
@@ -28,8 +29,10 @@ Classifier: Typing :: Typed
|
|
|
28
29
|
License-File: LICENSE
|
|
29
30
|
Requires-Dist: coverage ; extra == "tests"
|
|
30
31
|
Requires-Dist: mypy ; extra == "tests"
|
|
32
|
+
Requires-Dist: mypy<1.19.0 ; extra == "tests" and ( platform_python_implementation == 'PyPy')
|
|
31
33
|
Requires-Dist: pycodestyle ; extra == "tests"
|
|
32
34
|
Requires-Dist: pylint ; extra == "tests"
|
|
35
|
+
Requires-Dist: pyright ; extra == "tests"
|
|
33
36
|
Project-URL: Homepage, https://github.com/tinytag/tinytag
|
|
34
37
|
Provides-Extra: tests
|
|
35
38
|
|
|
@@ -44,8 +47,6 @@ tinytag is a Python library for reading audio file metadata
|
|
|
44
47
|
|
|
45
48
|
[](https://github.com/tinytag/tinytag/actions?query=workflow:%22Tests%22)
|
|
47
|
-
[](https://coveralls.io/r/tinytag/tinytag)
|
|
49
50
|
[](https://pypi.org/project/tinytag/)
|
|
51
52
|
[
|
|
|
422
429
|
|
|
423
430
|
## Changelog
|
|
424
431
|
|
|
432
|
+
### 2.2.0 (2025-12-15)
|
|
433
|
+
|
|
434
|
+
- Add support for movement, work and grouping fields
|
|
435
|
+
- ID3: Make synced lyrics available in 'other.lyrics' (LRC format)
|
|
436
|
+
- ID3: Continue reading after encountering empty frame
|
|
437
|
+
- ID3: Fix frame reading when image parsing is disabled
|
|
438
|
+
- ID3: Exclude more frames containing binary data
|
|
439
|
+
- ID3: Avoid unnecessary string decoding
|
|
440
|
+
- M4A: Support extended atom sizes
|
|
441
|
+
- M4A: Ensure all field names are lowercase
|
|
442
|
+
- OGG: Stop reading after reaching EOS page
|
|
443
|
+
- Vorbis: Map UNSYNCEDLYRICS field to other.lyrics
|
|
444
|
+
|
|
445
|
+
### 2.1.2 (2025-08-14)
|
|
446
|
+
|
|
447
|
+
- M4A: Add a few missing additional metadata fields
|
|
448
|
+
- M4A: Support '©com' composer atom
|
|
449
|
+
- M4A: Fix reading of multi-value custom fields
|
|
450
|
+
- M4A: Use correct encoding when reading data names
|
|
451
|
+
- ID3: Don't read entire file to determine duration
|
|
452
|
+
- ID3: Skip stray null byte before image data
|
|
453
|
+
- Add missing `__version__` attribute
|
|
454
|
+
- Avoid some unnecessary work in hot code paths
|
|
455
|
+
- Improve a few incomplete type hints
|
|
456
|
+
|
|
425
457
|
### 2.1.1 (2025-04-23)
|
|
426
458
|
|
|
427
459
|
- ID3: Stop removing 'b' character from strings
|
|
@@ -9,8 +9,6 @@ tinytag is a Python library for reading audio file metadata
|
|
|
9
9
|
|
|
10
10
|
[](https://github.com/tinytag/tinytag/actions?query=workflow:%22Tests%22)
|
|
12
|
-
[](https://coveralls.io/r/tinytag/tinytag)
|
|
14
12
|
[](https://pypi.org/project/tinytag/)
|
|
16
14
|
[
|
|
|
387
391
|
|
|
388
392
|
## Changelog
|
|
389
393
|
|
|
394
|
+
### 2.2.0 (2025-12-15)
|
|
395
|
+
|
|
396
|
+
- Add support for movement, work and grouping fields
|
|
397
|
+
- ID3: Make synced lyrics available in 'other.lyrics' (LRC format)
|
|
398
|
+
- ID3: Continue reading after encountering empty frame
|
|
399
|
+
- ID3: Fix frame reading when image parsing is disabled
|
|
400
|
+
- ID3: Exclude more frames containing binary data
|
|
401
|
+
- ID3: Avoid unnecessary string decoding
|
|
402
|
+
- M4A: Support extended atom sizes
|
|
403
|
+
- M4A: Ensure all field names are lowercase
|
|
404
|
+
- OGG: Stop reading after reaching EOS page
|
|
405
|
+
- Vorbis: Map UNSYNCEDLYRICS field to other.lyrics
|
|
406
|
+
|
|
407
|
+
### 2.1.2 (2025-08-14)
|
|
408
|
+
|
|
409
|
+
- M4A: Add a few missing additional metadata fields
|
|
410
|
+
- M4A: Support '©com' composer atom
|
|
411
|
+
- M4A: Fix reading of multi-value custom fields
|
|
412
|
+
- M4A: Use correct encoding when reading data names
|
|
413
|
+
- ID3: Don't read entire file to determine duration
|
|
414
|
+
- ID3: Skip stray null byte before image data
|
|
415
|
+
- Add missing `__version__` attribute
|
|
416
|
+
- Avoid some unnecessary work in hot code paths
|
|
417
|
+
- Improve a few incomplete type hints
|
|
418
|
+
|
|
390
419
|
### 2.1.1 (2025-04-23)
|
|
391
420
|
|
|
392
421
|
- ID3: Stop removing 'b' character from strings
|
|
@@ -7,7 +7,6 @@ build-backend = "flit_core.buildapi"
|
|
|
7
7
|
|
|
8
8
|
[project]
|
|
9
9
|
name = "tinytag"
|
|
10
|
-
version = "2.1.1"
|
|
11
10
|
description = "Read audio file metadata"
|
|
12
11
|
authors = [
|
|
13
12
|
{name = "Tom Wallroth"},
|
|
@@ -36,6 +35,7 @@ classifiers = [
|
|
|
36
35
|
"Programming Language :: Python :: 3.11",
|
|
37
36
|
"Programming Language :: Python :: 3.12",
|
|
38
37
|
"Programming Language :: Python :: 3.13",
|
|
38
|
+
"Programming Language :: Python :: 3.14",
|
|
39
39
|
"License :: OSI Approved :: MIT License",
|
|
40
40
|
"Development Status :: 5 - Production/Stable",
|
|
41
41
|
"Environment :: Web Environment",
|
|
@@ -50,6 +50,7 @@ classifiers = [
|
|
|
50
50
|
license = {file = "LICENSE"}
|
|
51
51
|
readme = "README.md"
|
|
52
52
|
requires-python = ">=3.7"
|
|
53
|
+
dynamic = ["version"]
|
|
53
54
|
|
|
54
55
|
[project.urls]
|
|
55
56
|
Homepage = "https://github.com/tinytag/tinytag"
|
|
@@ -58,8 +59,10 @@ Homepage = "https://github.com/tinytag/tinytag"
|
|
|
58
59
|
tests = [
|
|
59
60
|
"coverage",
|
|
60
61
|
"mypy",
|
|
62
|
+
"mypy<1.19.0; platform_python_implementation == 'PyPy'",
|
|
61
63
|
"pycodestyle",
|
|
62
|
-
"pylint"
|
|
64
|
+
"pylint",
|
|
65
|
+
"pyright"
|
|
63
66
|
]
|
|
64
67
|
|
|
65
68
|
[tool.flit.sdist]
|
|
@@ -114,3 +117,13 @@ py-version = "3.7"
|
|
|
114
117
|
|
|
115
118
|
[tool.mypy]
|
|
116
119
|
strict = true
|
|
120
|
+
|
|
121
|
+
[tool.coverage.report]
|
|
122
|
+
exclude_lines = [
|
|
123
|
+
"if TYPE_CHECKING:"
|
|
124
|
+
]
|
|
125
|
+
precision = 2
|
|
126
|
+
show_missing = true
|
|
127
|
+
|
|
128
|
+
[tool.coverage.run]
|
|
129
|
+
relative_files = true
|
|
@@ -34,15 +34,19 @@ from io import BytesIO
|
|
|
34
34
|
from os import PathLike, SEEK_CUR, SEEK_END, environ, fsdecode
|
|
35
35
|
from struct import unpack
|
|
36
36
|
|
|
37
|
+
TYPE_CHECKING = False
|
|
38
|
+
|
|
37
39
|
# Lazy imports for type checking
|
|
38
|
-
if
|
|
40
|
+
if TYPE_CHECKING:
|
|
39
41
|
from collections.abc import Callable, Iterator # pylint: disable-all
|
|
40
|
-
from typing import Any, BinaryIO, Dict, List
|
|
42
|
+
from typing import Any, BinaryIO, Dict, List, Union
|
|
41
43
|
|
|
42
44
|
_StringListDict = Dict[str, List[str]]
|
|
43
45
|
_ImageListDict = Dict[str, List["Image"]]
|
|
46
|
+
_DataTreeDict = Dict[
|
|
47
|
+
bytes, Union['_DataTreeDict', Callable[..., Dict[str, Any]]]]
|
|
44
48
|
else:
|
|
45
|
-
_StringListDict = _ImageListDict = dict
|
|
49
|
+
_StringListDict = _ImageListDict = _DataTreeDict = dict
|
|
46
50
|
|
|
47
51
|
# some of the parsers can print debug info
|
|
48
52
|
_DEBUG = bool(environ.get('TINYTAG_DEBUG'))
|
|
@@ -105,7 +109,7 @@ class TinyTag:
|
|
|
105
109
|
self._parse_tags = True
|
|
106
110
|
self._load_image = False
|
|
107
111
|
self._tags_parsed = False
|
|
108
|
-
self.__dict__: dict[str, str | float | Images | OtherFields]
|
|
112
|
+
self.__dict__: dict[str, str | float | Images | OtherFields | None]
|
|
109
113
|
|
|
110
114
|
@classmethod
|
|
111
115
|
def get(cls,
|
|
@@ -257,7 +261,7 @@ class TinyTag:
|
|
|
257
261
|
self._parse_duration = duration
|
|
258
262
|
self._load_image = image
|
|
259
263
|
if self._filehandler is None:
|
|
260
|
-
|
|
264
|
+
raise ValueError("File handle is required")
|
|
261
265
|
if tags:
|
|
262
266
|
self._parse_tag(self._filehandler)
|
|
263
267
|
if duration:
|
|
@@ -271,14 +275,11 @@ class TinyTag:
|
|
|
271
275
|
fieldname = fieldname[len(self._OTHER_PREFIX):]
|
|
272
276
|
if check_conflict and fieldname in self.__dict__:
|
|
273
277
|
fieldname = '_' + fieldname
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
other_values.append(value)
|
|
278
|
+
if fieldname not in self.other:
|
|
279
|
+
self.other[fieldname] = []
|
|
280
|
+
self.other[fieldname].append(str(value))
|
|
278
281
|
if _DEBUG:
|
|
279
|
-
print(
|
|
280
|
-
f'Setting other field "{fieldname}" to "{other_values!r}"')
|
|
281
|
-
self.other[fieldname] = other_values
|
|
282
|
+
print(f'Adding value "{value} to field "{fieldname}"')
|
|
282
283
|
return
|
|
283
284
|
old_value = self.__dict__.get(fieldname)
|
|
284
285
|
new_value = value
|
|
@@ -367,7 +368,7 @@ class Images:
|
|
|
367
368
|
self.media: Image | None = None
|
|
368
369
|
|
|
369
370
|
self.other: _ImageListDict = OtherImages()
|
|
370
|
-
self.__dict__: dict[str, Image | OtherImages]
|
|
371
|
+
self.__dict__: dict[str, Image | OtherImages | None]
|
|
371
372
|
|
|
372
373
|
@property
|
|
373
374
|
def any(self) -> Image | None:
|
|
@@ -486,9 +487,10 @@ class _MP4(TinyTag):
|
|
|
486
487
|
}
|
|
487
488
|
_VERSIONED_ATOMS = {b'meta', b'stsd'} # those have an extra 4 byte header
|
|
488
489
|
_FLAGGED_ATOMS = {b'stsd'} # these also have an extra 4 byte header
|
|
490
|
+
_ILST_PATH = [b'ftyp', b'moov', b'udta', b'meta', b'ilst']
|
|
489
491
|
|
|
490
|
-
_audio_data_tree:
|
|
491
|
-
_meta_data_tree:
|
|
492
|
+
_audio_data_tree: _DataTreeDict | None = None
|
|
493
|
+
_meta_data_tree: _DataTreeDict | None = None
|
|
492
494
|
|
|
493
495
|
def _determine_duration(self, fh: BinaryIO) -> None:
|
|
494
496
|
# https://developer.apple.com/library/mac/documentation/QuickTime/QTFF/QTFFChap3/qtff3.html
|
|
@@ -516,24 +518,32 @@ class _MP4(TinyTag):
|
|
|
516
518
|
b'\xa9ART': {b'data': _MP4._data_parser('artist')},
|
|
517
519
|
b'\xa9alb': {b'data': _MP4._data_parser('album')},
|
|
518
520
|
b'\xa9cmt': {b'data': _MP4._data_parser('comment')},
|
|
521
|
+
b'\xa9com': {b'data': _MP4._data_parser('composer')},
|
|
519
522
|
b'\xa9con': {b'data': _MP4._data_parser('other.conductor')},
|
|
520
|
-
# need test-data for this
|
|
521
|
-
# b'cpil': {b'data': _MP4._data_parser('other.compilation')},
|
|
522
523
|
b'\xa9day': {b'data': _MP4._data_parser('year')},
|
|
523
524
|
b'\xa9des': {b'data': _MP4._data_parser('other.description')},
|
|
524
525
|
b'\xa9dir': {b'data': _MP4._data_parser('other.director')},
|
|
525
526
|
b'\xa9gen': {b'data': _MP4._data_parser('genre')},
|
|
527
|
+
b'\xa9grp': {b'data': _MP4._data_parser('other.grouping')},
|
|
526
528
|
b'\xa9lyr': {b'data': _MP4._data_parser('other.lyrics')},
|
|
527
|
-
b'\
|
|
529
|
+
b'\xa9mvc': {
|
|
530
|
+
b'data': _MP4._data_parser('other.movement_total')
|
|
531
|
+
},
|
|
532
|
+
b'\xa9mvi': {b'data': _MP4._data_parser('other.movement')},
|
|
533
|
+
b'\xa9mvn': {
|
|
534
|
+
b'data': _MP4._data_parser('other.movement_name')
|
|
535
|
+
},
|
|
528
536
|
b'\xa9nam': {b'data': _MP4._data_parser('title')},
|
|
529
537
|
b'\xa9pub': {b'data': _MP4._data_parser('other.publisher')},
|
|
530
538
|
b'\xa9too': {b'data': _MP4._data_parser('other.encoded_by')},
|
|
539
|
+
b'\xa9wrk': {b'data': _MP4._data_parser('other.work')},
|
|
531
540
|
b'\xa9wrt': {b'data': _MP4._data_parser('composer')},
|
|
532
541
|
b'aART': {b'data': _MP4._data_parser('albumartist')},
|
|
533
542
|
b'cprt': {b'data': _MP4._data_parser('other.copyright')},
|
|
534
543
|
b'desc': {b'data': _MP4._data_parser('other.description')},
|
|
535
544
|
b'disk': {b'data': _MP4._nums_parser('disc', 'disc_total')},
|
|
536
545
|
b'gnre': {b'data': _MP4._parse_id3v1_genre},
|
|
546
|
+
b'shwm': {b'data': _MP4._data_parser('other.show_movement')},
|
|
537
547
|
b'trkn': {b'data': _MP4._nums_parser('track', 'track_total')},
|
|
538
548
|
b'tmpo': {b'data': _MP4._data_parser('other.bpm')},
|
|
539
549
|
b'covr': {b'data': _MP4._parse_cover_image},
|
|
@@ -543,16 +553,21 @@ class _MP4(TinyTag):
|
|
|
543
553
|
|
|
544
554
|
def _traverse_atoms(self,
|
|
545
555
|
fh: BinaryIO,
|
|
546
|
-
path:
|
|
556
|
+
path: _DataTreeDict,
|
|
547
557
|
stop_pos: int | None = None,
|
|
548
558
|
curr_path: list[bytes] | None = None) -> None:
|
|
549
|
-
header_len = 8
|
|
559
|
+
header_len = ext_size_len = 8
|
|
550
560
|
atom_header = fh.read(header_len)
|
|
551
561
|
while len(atom_header) == header_len:
|
|
552
|
-
atom_size = unpack('>I', atom_header[:4])[0]
|
|
562
|
+
atom_size = unpack('>I', atom_header[:4])[0]
|
|
553
563
|
atom_type = atom_header[4:]
|
|
554
564
|
if curr_path is None: # keep track how we traversed in the tree
|
|
555
565
|
curr_path = [atom_type]
|
|
566
|
+
if atom_size == 1: # 64-bit size
|
|
567
|
+
ext_size_header = fh.read(ext_size_len)
|
|
568
|
+
if len(ext_size_header) == ext_size_len:
|
|
569
|
+
atom_size = unpack('>Q', ext_size_header)[0] - ext_size_len
|
|
570
|
+
atom_size -= header_len
|
|
556
571
|
if atom_size <= 0: # empty atom, jump to next one
|
|
557
572
|
atom_header = fh.read(header_len)
|
|
558
573
|
continue
|
|
@@ -562,8 +577,10 @@ class _MP4(TinyTag):
|
|
|
562
577
|
f'atom: {atom_type!r} len: {atom_size + header_len}')
|
|
563
578
|
if atom_type in self._VERSIONED_ATOMS: # jump atom version for now
|
|
564
579
|
fh.seek(4, SEEK_CUR)
|
|
580
|
+
atom_size -= 4
|
|
565
581
|
if atom_type in self._FLAGGED_ATOMS: # jump atom flags for now
|
|
566
582
|
fh.seek(4, SEEK_CUR)
|
|
583
|
+
atom_size -= 4
|
|
567
584
|
sub_path = path.get(atom_type, None)
|
|
568
585
|
# if the path leaf is a dict, traverse deeper into the tree:
|
|
569
586
|
if isinstance(sub_path, dict):
|
|
@@ -575,13 +592,27 @@ class _MP4(TinyTag):
|
|
|
575
592
|
for fieldname, value in sub_path(fh.read(atom_size)).items():
|
|
576
593
|
if _DEBUG:
|
|
577
594
|
print(' ' * 4 * len(curr_path), 'FIELD: ', fieldname)
|
|
578
|
-
if
|
|
595
|
+
if isinstance(value, Image):
|
|
579
596
|
if self._load_image:
|
|
580
597
|
# pylint: disable=protected-access
|
|
581
598
|
self.images._set_field(
|
|
582
599
|
fieldname[len('images.'):], value)
|
|
583
|
-
elif
|
|
600
|
+
elif isinstance(value, list):
|
|
601
|
+
for subval in value:
|
|
602
|
+
self._set_field(fieldname, subval)
|
|
603
|
+
else:
|
|
584
604
|
self._set_field(fieldname, value)
|
|
605
|
+
# unknown data atom, try to parse it
|
|
606
|
+
elif curr_path == self._ILST_PATH:
|
|
607
|
+
atom_end_pos = fh.tell() + atom_size
|
|
608
|
+
field_name = (
|
|
609
|
+
self._OTHER_PREFIX + atom_type.decode('latin-1').lower()
|
|
610
|
+
)
|
|
611
|
+
fh.seek(-header_len, SEEK_CUR)
|
|
612
|
+
self._traverse_atoms(
|
|
613
|
+
fh,
|
|
614
|
+
path={atom_type: {b'data': self._data_parser(field_name)}},
|
|
615
|
+
stop_pos=atom_end_pos, curr_path=curr_path + [atom_type])
|
|
585
616
|
# if no action was specified using dict or callable, jump over atom
|
|
586
617
|
else:
|
|
587
618
|
fh.seek(atom_size, SEEK_CUR)
|
|
@@ -591,12 +622,8 @@ class _MP4(TinyTag):
|
|
|
591
622
|
atom_header = fh.read(header_len) # read next atom
|
|
592
623
|
|
|
593
624
|
@classmethod
|
|
594
|
-
def _data_parser(
|
|
595
|
-
|
|
596
|
-
) -> Callable[[bytes], dict[str, int | str | bytes | None]]:
|
|
597
|
-
def _parse_data_atom(
|
|
598
|
-
data_atom: bytes
|
|
599
|
-
) -> dict[str, int | str | bytes | None]:
|
|
625
|
+
def _data_parser(cls, fieldname: str) -> Callable[[bytes], dict[str, str]]:
|
|
626
|
+
def _parse_data_atom(data_atom: bytes) -> dict[str, str]:
|
|
600
627
|
data_type = unpack('>I', data_atom[:4])[0]
|
|
601
628
|
data = data_atom[8:]
|
|
602
629
|
value = None
|
|
@@ -607,7 +634,9 @@ class _MP4(TinyTag):
|
|
|
607
634
|
data_len = len(data)
|
|
608
635
|
if data_len in fmts:
|
|
609
636
|
value = str(unpack(fmts[data_len], data)[0])
|
|
610
|
-
|
|
637
|
+
if value:
|
|
638
|
+
return {fieldname: value}
|
|
639
|
+
return {}
|
|
611
640
|
return _parse_data_atom
|
|
612
641
|
|
|
613
642
|
@classmethod
|
|
@@ -645,13 +674,11 @@ class _MP4(TinyTag):
|
|
|
645
674
|
break
|
|
646
675
|
|
|
647
676
|
@classmethod
|
|
648
|
-
def _parse_custom_field(
|
|
649
|
-
cls, data: bytes
|
|
650
|
-
) -> dict[str, int | str | bytes | None]:
|
|
677
|
+
def _parse_custom_field(cls, data: bytes) -> dict[str, list[str]]:
|
|
651
678
|
fh = BytesIO(data)
|
|
652
679
|
header_len = 8
|
|
653
680
|
field_name = None
|
|
654
|
-
|
|
681
|
+
values = []
|
|
655
682
|
atom_header = fh.read(header_len)
|
|
656
683
|
while len(atom_header) == header_len:
|
|
657
684
|
atom_size = unpack('>I', atom_header[:4])[0] - header_len
|
|
@@ -662,15 +689,18 @@ class _MP4(TinyTag):
|
|
|
662
689
|
# pylint: disable=protected-access
|
|
663
690
|
field_name = cls._CUSTOM_FIELD_NAME_MAPPING.get(
|
|
664
691
|
field_name, TinyTag._OTHER_PREFIX + field_name)
|
|
665
|
-
elif atom_type == b'data':
|
|
692
|
+
elif atom_type == b'data' and field_name:
|
|
666
693
|
data_atom = fh.read(atom_size)
|
|
694
|
+
parser = cls._data_parser(field_name)
|
|
695
|
+
atom_values = parser(data_atom)
|
|
696
|
+
if field_name in atom_values:
|
|
697
|
+
values.append(atom_values[field_name])
|
|
667
698
|
else:
|
|
668
699
|
fh.seek(atom_size, SEEK_CUR)
|
|
669
700
|
atom_header = fh.read(header_len) # read next atom
|
|
670
|
-
if
|
|
671
|
-
return {}
|
|
672
|
-
|
|
673
|
-
return parser(data_atom)
|
|
701
|
+
if field_name and values:
|
|
702
|
+
return {field_name: values}
|
|
703
|
+
return {}
|
|
674
704
|
|
|
675
705
|
@classmethod
|
|
676
706
|
def _parse_audio_sample_entry_mp4a(cls, data: bytes) -> dict[str, int]:
|
|
@@ -727,31 +757,35 @@ class _ID3(TinyTag):
|
|
|
727
757
|
_ID3_MAPPING = {
|
|
728
758
|
# Mapping from Frame ID to a field of the TinyTag
|
|
729
759
|
# https://exiftool.org/TagNames/ID3.html
|
|
730
|
-
'COMM': 'comment', 'COM': 'comment',
|
|
731
|
-
'TRCK': 'track', 'TRK': 'track',
|
|
732
|
-
'TYER': 'year', 'TYE': 'year', 'TDRC': 'year',
|
|
733
|
-
'TALB': 'album', 'TAL': 'album',
|
|
734
|
-
'TPE1': 'artist', 'TP1': 'artist',
|
|
735
|
-
'TIT2': 'title', 'TT2': 'title',
|
|
736
|
-
'TCON': 'genre', 'TCO': 'genre',
|
|
737
|
-
'TPOS': 'disc', 'TPA': 'disc',
|
|
738
|
-
'TPE2': 'albumartist', 'TP2': 'albumartist',
|
|
739
|
-
'TCOM': 'composer', 'TCM': 'composer',
|
|
740
|
-
'WOAR': 'other.url', 'WAR': 'other.url',
|
|
741
|
-
'TSRC': 'other.isrc', 'TRC': 'other.isrc',
|
|
742
|
-
'TCOP': 'other.copyright', 'TCR': 'other.copyright',
|
|
743
|
-
'TBPM': 'other.bpm', 'TBP': 'other.bpm',
|
|
744
|
-
'TKEY': 'other.initial_key', 'TKE': 'other.initial_key',
|
|
745
|
-
'TLAN': 'other.language', 'TLA': 'other.language',
|
|
746
|
-
'TPUB': 'other.publisher', 'TPB': 'other.publisher',
|
|
747
|
-
'USLT': 'other.lyrics', 'ULT': 'other.lyrics',
|
|
748
|
-
'TPE3': 'other.conductor', 'TP3': 'other.conductor',
|
|
749
|
-
'TEXT': 'other.lyricist', 'TXT': 'other.lyricist',
|
|
750
|
-
'TSST': 'other.set_subtitle',
|
|
751
|
-
'TENC': 'other.encoded_by', 'TEN': 'other.encoded_by',
|
|
752
|
-
'TSSE': 'other.encoder_settings', 'TSS': 'other.encoder_settings',
|
|
753
|
-
'TMED': 'other.media', 'TMT': 'other.media',
|
|
754
|
-
'WCOP': 'other.license',
|
|
760
|
+
b'COMM': 'comment', b'COM': 'comment',
|
|
761
|
+
b'TRCK': 'track', b'TRK': 'track',
|
|
762
|
+
b'TYER': 'year', b'TYE': 'year', b'TDRC': 'year',
|
|
763
|
+
b'TALB': 'album', b'TAL': 'album',
|
|
764
|
+
b'TPE1': 'artist', b'TP1': 'artist',
|
|
765
|
+
b'TIT2': 'title', b'TT2': 'title',
|
|
766
|
+
b'TCON': 'genre', b'TCO': 'genre',
|
|
767
|
+
b'TPOS': 'disc', b'TPA': 'disc',
|
|
768
|
+
b'TPE2': 'albumartist', b'TP2': 'albumartist',
|
|
769
|
+
b'TCOM': 'composer', b'TCM': 'composer',
|
|
770
|
+
b'WOAR': 'other.url', b'WAR': 'other.url',
|
|
771
|
+
b'TSRC': 'other.isrc', b'TRC': 'other.isrc',
|
|
772
|
+
b'TCOP': 'other.copyright', b'TCR': 'other.copyright',
|
|
773
|
+
b'TBPM': 'other.bpm', b'TBP': 'other.bpm',
|
|
774
|
+
b'TKEY': 'other.initial_key', b'TKE': 'other.initial_key',
|
|
775
|
+
b'TLAN': 'other.language', b'TLA': 'other.language',
|
|
776
|
+
b'TPUB': 'other.publisher', b'TPB': 'other.publisher',
|
|
777
|
+
b'USLT': 'other.lyrics', b'ULT': 'other.lyrics',
|
|
778
|
+
b'TPE3': 'other.conductor', b'TP3': 'other.conductor',
|
|
779
|
+
b'TEXT': 'other.lyricist', b'TXT': 'other.lyricist',
|
|
780
|
+
b'TSST': 'other.set_subtitle',
|
|
781
|
+
b'TENC': 'other.encoded_by', b'TEN': 'other.encoded_by',
|
|
782
|
+
b'TSSE': 'other.encoder_settings', b'TSS': 'other.encoder_settings',
|
|
783
|
+
b'TMED': 'other.media', b'TMT': 'other.media',
|
|
784
|
+
b'WCOP': 'other.license',
|
|
785
|
+
b'MVNM': 'other.movement_name',
|
|
786
|
+
b'MVIN': 'other.movement',
|
|
787
|
+
b'GRP1': 'modern_grouping', b'GP1': 'modern_grouping',
|
|
788
|
+
b'TIT1': 'legacy_grouping', b'TT1': 'legacy_grouping',
|
|
755
789
|
}
|
|
756
790
|
_ID3_MAPPING_CUSTOM = {
|
|
757
791
|
'artists': 'artist',
|
|
@@ -759,23 +793,40 @@ class _ID3(TinyTag):
|
|
|
759
793
|
'license': 'other.license',
|
|
760
794
|
'barcode': 'other.barcode',
|
|
761
795
|
'catalognumber': 'other.catalog_number',
|
|
796
|
+
'showmovement': 'other.show_movement'
|
|
762
797
|
}
|
|
763
|
-
|
|
764
|
-
|
|
798
|
+
_EMPTY_FRAME_IDS = {b'\x00\x00\x00\x00', b'\x00\x00\x00'}
|
|
799
|
+
_IMAGE_FRAME_IDS = {b'APIC', b'PIC'}
|
|
800
|
+
_CUSTOM_FRAME_IDS = {b'TXXX', b'TXX'}
|
|
801
|
+
_SYNCED_LYRICS_FRAME_IDS = {b'SYLT', b'SLT'}
|
|
765
802
|
_IGNORED_FRAME_IDS = {
|
|
766
|
-
'AENC', 'CRA',
|
|
767
|
-
'
|
|
768
|
-
'
|
|
769
|
-
'
|
|
770
|
-
'
|
|
771
|
-
'
|
|
772
|
-
'
|
|
773
|
-
'
|
|
774
|
-
'
|
|
775
|
-
'
|
|
776
|
-
'
|
|
777
|
-
'
|
|
778
|
-
'
|
|
803
|
+
b'AENC', b'CRA',
|
|
804
|
+
b'APIC', b'PIC',
|
|
805
|
+
b'ASPI',
|
|
806
|
+
b'ATXT',
|
|
807
|
+
b'CHAP',
|
|
808
|
+
b'COMR',
|
|
809
|
+
b'CRM',
|
|
810
|
+
b'CTOC',
|
|
811
|
+
b'ENCR',
|
|
812
|
+
b'EQU2', b'EQU',
|
|
813
|
+
b'ETCO', b'ETC',
|
|
814
|
+
b'GEOB', b'GEO',
|
|
815
|
+
b'GRID',
|
|
816
|
+
b'LINK', b'LNK',
|
|
817
|
+
b'MCDI', b'MCI',
|
|
818
|
+
b'MLLT', b'MLL',
|
|
819
|
+
b'PCNT', b'CNT',
|
|
820
|
+
b'POPM', b'POP',
|
|
821
|
+
b'POSS',
|
|
822
|
+
b'PRIV',
|
|
823
|
+
b'RBUF', b'BUF',
|
|
824
|
+
b'RGAD',
|
|
825
|
+
b'RVA2', b'RVA',
|
|
826
|
+
b'RVRB', b'REV',
|
|
827
|
+
b'SEEK',
|
|
828
|
+
b'SIGN',
|
|
829
|
+
b'SYTC', b'STC',
|
|
779
830
|
}
|
|
780
831
|
_ID3V1_TAG_SIZE = 128
|
|
781
832
|
_MAX_ESTIMATION_SEC = 30.0
|
|
@@ -825,9 +876,9 @@ class _ID3(TinyTag):
|
|
|
825
876
|
'Psybient',
|
|
826
877
|
)
|
|
827
878
|
_ID3V2_2_IMAGE_FORMATS = {
|
|
828
|
-
'bmp': 'image/bmp',
|
|
829
|
-
'jpg': 'image/jpeg',
|
|
830
|
-
'png': 'image/png',
|
|
879
|
+
b'bmp': 'image/bmp',
|
|
880
|
+
b'jpg': 'image/jpeg',
|
|
881
|
+
b'png': 'image/png',
|
|
831
882
|
}
|
|
832
883
|
_IMAGE_TYPES = (
|
|
833
884
|
'other.generic',
|
|
@@ -892,6 +943,8 @@ class _ID3(TinyTag):
|
|
|
892
943
|
super().__init__()
|
|
893
944
|
# save position after the ID3 tag for duration measurement speedup
|
|
894
945
|
self._bytepos_after_id3v2 = -1
|
|
946
|
+
self._modern_grouping_values: list[str] = []
|
|
947
|
+
self._legacy_grouping_values: list[str] = []
|
|
895
948
|
|
|
896
949
|
@staticmethod
|
|
897
950
|
def _parse_xing_header(fh: BinaryIO) -> tuple[int, int]:
|
|
@@ -917,20 +970,17 @@ class _ID3(TinyTag):
|
|
|
917
970
|
max_estimation_frames = (
|
|
918
971
|
(self._MAX_ESTIMATION_SEC * 44100) // self._SAMPLES_PER_FRAME)
|
|
919
972
|
frame_size_accu = 0
|
|
920
|
-
audio_offset =
|
|
973
|
+
audio_offset = self._bytepos_after_id3v2
|
|
921
974
|
frames = 0 # count frames for determining mp3 duration
|
|
922
975
|
bitrate_accu = 0 # add up bitrates to find average bitrate to detect
|
|
923
976
|
last_bitrates = set() # CBR mp3s (multiple frames with same bitrates)
|
|
924
977
|
# seek to first position after id3 tag (speedup for large header)
|
|
925
978
|
first_mpeg_id = None
|
|
926
979
|
fh.seek(self._bytepos_after_id3v2)
|
|
927
|
-
file_offset = fh.tell()
|
|
928
|
-
walker = BytesIO(fh.read())
|
|
929
980
|
while True:
|
|
930
981
|
# reading through garbage until 11 '1' sync-bits are found
|
|
931
|
-
header =
|
|
982
|
+
header = fh.read(4)
|
|
932
983
|
header_len = len(header)
|
|
933
|
-
walker.seek(-header_len, SEEK_CUR)
|
|
934
984
|
if header_len < 4:
|
|
935
985
|
if frames:
|
|
936
986
|
self.bitrate = bitrate_accu / frames
|
|
@@ -949,10 +999,12 @@ class _ID3(TinyTag):
|
|
|
949
999
|
or mpeg_id == 1):
|
|
950
1000
|
# invalid frame, find next sync header
|
|
951
1001
|
idx = header.find(b'\xFF', 1)
|
|
952
|
-
|
|
953
|
-
|
|
954
|
-
|
|
955
|
-
|
|
1002
|
+
next_offset = header_len
|
|
1003
|
+
if idx != -1:
|
|
1004
|
+
next_offset -= idx
|
|
1005
|
+
fh.seek(idx - header_len, SEEK_CUR)
|
|
1006
|
+
if frames == 0:
|
|
1007
|
+
audio_offset += next_offset
|
|
956
1008
|
continue
|
|
957
1009
|
if first_mpeg_id is None:
|
|
958
1010
|
first_mpeg_id = mpeg_id
|
|
@@ -964,12 +1016,12 @@ class _ID3(TinyTag):
|
|
|
964
1016
|
# all the info we need, otherwise parse multiple frames to find the
|
|
965
1017
|
# accurate average bitrate
|
|
966
1018
|
if frames == 0 and self._USE_XING_HEADER:
|
|
967
|
-
|
|
968
|
-
frame_content =
|
|
1019
|
+
prev_offset = header_len + audio_offset
|
|
1020
|
+
frame_content = fh.read(frame_length)
|
|
969
1021
|
xing_header_offset = frame_content.find(b'Xing')
|
|
970
1022
|
if xing_header_offset != -1:
|
|
971
|
-
|
|
972
|
-
xframes, byte_count = self._parse_xing_header(
|
|
1023
|
+
fh.seek(prev_offset + xing_header_offset)
|
|
1024
|
+
xframes, byte_count = self._parse_xing_header(fh)
|
|
973
1025
|
if xframes > 0 and byte_count > 0:
|
|
974
1026
|
# MPEG-2 Audio Layer III uses 576 samples per frame
|
|
975
1027
|
samples_pf = self._SAMPLES_PER_FRAME
|
|
@@ -978,12 +1030,10 @@ class _ID3(TinyTag):
|
|
|
978
1030
|
self.duration = dur = xframes * samples_pf / samplerate
|
|
979
1031
|
self.bitrate = byte_count * 8 / dur / 1000
|
|
980
1032
|
return
|
|
981
|
-
|
|
1033
|
+
fh.seek(prev_offset)
|
|
982
1034
|
|
|
983
1035
|
frames += 1 # it's most probably a mp3 frame
|
|
984
1036
|
bitrate_accu += frame_br
|
|
985
|
-
if frames == 1:
|
|
986
|
-
audio_offset = file_offset + walker.tell()
|
|
987
1037
|
if frames <= self._CBR_DETECTION_FRAME_COUNT:
|
|
988
1038
|
last_bitrates.add(frame_br)
|
|
989
1039
|
|
|
@@ -1002,7 +1052,7 @@ class _ID3(TinyTag):
|
|
|
1002
1052
|
return
|
|
1003
1053
|
|
|
1004
1054
|
if frame_length > 1: # jump over current frame body
|
|
1005
|
-
|
|
1055
|
+
fh.seek(frame_length - header_len, SEEK_CUR)
|
|
1006
1056
|
if self.samplerate:
|
|
1007
1057
|
self.duration = frames * self._SAMPLES_PER_FRAME / self.samplerate
|
|
1008
1058
|
|
|
@@ -1039,13 +1089,15 @@ class _ID3(TinyTag):
|
|
|
1039
1089
|
fh.seek(extd_size - 6, SEEK_CUR) # jump over extended_header
|
|
1040
1090
|
while parsed_size < size:
|
|
1041
1091
|
frame_size = self._parse_frame(fh, size, id3version=major)
|
|
1042
|
-
if frame_size ==
|
|
1092
|
+
if frame_size == -1:
|
|
1043
1093
|
break
|
|
1044
1094
|
parsed_size += frame_size
|
|
1045
1095
|
fh.seek(end_pos)
|
|
1096
|
+
self._set_grouping_work_fields()
|
|
1046
1097
|
|
|
1047
1098
|
def _parse_id3v1(self, fh: BinaryIO) -> None:
|
|
1048
|
-
|
|
1099
|
+
content = fh.read(3 + 30 + 30 + 30 + 4 + 30 + 1)
|
|
1100
|
+
if content[:3] != b'TAG': # check if this is an ID3 v1 tag
|
|
1049
1101
|
return
|
|
1050
1102
|
|
|
1051
1103
|
def asciidecode(x: bytes) -> str:
|
|
@@ -1053,24 +1105,23 @@ class _ID3(TinyTag):
|
|
|
1053
1105
|
x.decode(self._default_encoding or 'latin1', 'replace'))
|
|
1054
1106
|
# Only set fields that were not set by ID3v2 tags, as ID3v1
|
|
1055
1107
|
# tags are more likely to be outdated or have encoding issues
|
|
1056
|
-
fields = fh.read(30 + 30 + 30 + 4 + 30 + 1)
|
|
1057
1108
|
if not self.title:
|
|
1058
|
-
value = asciidecode(
|
|
1109
|
+
value = asciidecode(content[3:33])
|
|
1059
1110
|
if value:
|
|
1060
1111
|
self._set_field('title', value)
|
|
1061
1112
|
if not self.artist:
|
|
1062
|
-
value = asciidecode(
|
|
1113
|
+
value = asciidecode(content[33:63])
|
|
1063
1114
|
if value:
|
|
1064
1115
|
self._set_field('artist', value)
|
|
1065
1116
|
if not self.album:
|
|
1066
|
-
value = asciidecode(
|
|
1117
|
+
value = asciidecode(content[63:93])
|
|
1067
1118
|
if value:
|
|
1068
1119
|
self._set_field('album', value)
|
|
1069
1120
|
if not self.year:
|
|
1070
|
-
value = asciidecode(
|
|
1121
|
+
value = asciidecode(content[93:97])
|
|
1071
1122
|
if value:
|
|
1072
1123
|
self._set_field('year', value)
|
|
1073
|
-
comment =
|
|
1124
|
+
comment = content[97:127]
|
|
1074
1125
|
if b'\x00\x00' < comment[-2:] < b'\x01\x00':
|
|
1075
1126
|
if self.track is None:
|
|
1076
1127
|
self._set_field('track', ord(comment[-1:]))
|
|
@@ -1080,11 +1131,11 @@ class _ID3(TinyTag):
|
|
|
1080
1131
|
if value:
|
|
1081
1132
|
self._set_field('comment', value)
|
|
1082
1133
|
if not self.genre:
|
|
1083
|
-
genre_id = ord(
|
|
1134
|
+
genre_id = ord(content[127:128])
|
|
1084
1135
|
if genre_id < len(self._ID3V1_GENRES):
|
|
1085
1136
|
self._set_field('genre', self._ID3V1_GENRES[genre_id])
|
|
1086
1137
|
|
|
1087
|
-
def
|
|
1138
|
+
def _parse_custom_field(self, content: str) -> bool:
|
|
1088
1139
|
custom_field_name, separator, value = content.partition('\x00')
|
|
1089
1140
|
custom_field_name_lower = custom_field_name.lower()
|
|
1090
1141
|
value = value.lstrip('\ufeff')
|
|
@@ -1096,6 +1147,18 @@ class _ID3(TinyTag):
|
|
|
1096
1147
|
return True
|
|
1097
1148
|
return False
|
|
1098
1149
|
|
|
1150
|
+
def _set_grouping_work_fields(self) -> None:
|
|
1151
|
+
# iTunes 12.5.4.42 added a new GRP1 frame for 'grouping', and
|
|
1152
|
+
# repurposed the TIT1 frame for 'work'. Handle this mess here.
|
|
1153
|
+
if self._modern_grouping_values:
|
|
1154
|
+
for value in self._modern_grouping_values:
|
|
1155
|
+
self._set_field('other.grouping', value)
|
|
1156
|
+
for value in self._legacy_grouping_values:
|
|
1157
|
+
self._set_field('other.work', value)
|
|
1158
|
+
return
|
|
1159
|
+
for value in self._legacy_grouping_values:
|
|
1160
|
+
self._set_field('other.grouping', value)
|
|
1161
|
+
|
|
1099
1162
|
@classmethod
|
|
1100
1163
|
def _create_tag_image(cls,
|
|
1101
1164
|
data: bytes,
|
|
@@ -1115,6 +1178,75 @@ class _ID3(TinyTag):
|
|
|
1115
1178
|
image.description = description
|
|
1116
1179
|
return field_name, image
|
|
1117
1180
|
|
|
1181
|
+
def _parse_image(self,
|
|
1182
|
+
frame_id: bytes,
|
|
1183
|
+
content: bytes) -> tuple[str, Image]:
|
|
1184
|
+
# See section 4.14: http://id3.org/id3v2.4.0-frames
|
|
1185
|
+
encoding = content[:1]
|
|
1186
|
+
if frame_id == b'PIC': # ID3 v2.2:
|
|
1187
|
+
imgformat = content[1:4].lower()
|
|
1188
|
+
mime_type = self._ID3V2_2_IMAGE_FORMATS.get(imgformat)
|
|
1189
|
+
# skip encoding (1), imgformat (3), pictype(1)
|
|
1190
|
+
desc_start_pos = 5
|
|
1191
|
+
else: # ID3 v2.3+
|
|
1192
|
+
mime_start_pos = 1
|
|
1193
|
+
mime_end_pos = self._find_string_end_pos(
|
|
1194
|
+
content, start_pos=mime_start_pos)
|
|
1195
|
+
mime_type = self._decode_string(
|
|
1196
|
+
content[mime_start_pos:mime_end_pos]).lower()
|
|
1197
|
+
# skip mtype, pictype(1)
|
|
1198
|
+
desc_start_pos = mime_end_pos + 1
|
|
1199
|
+
pic_type = content[desc_start_pos - 1]
|
|
1200
|
+
desc_end_pos = self._find_string_end_pos(
|
|
1201
|
+
content, encoding, desc_start_pos)
|
|
1202
|
+
# skip stray null byte in broken file
|
|
1203
|
+
if (desc_end_pos + 1 < len(content)
|
|
1204
|
+
and content[desc_end_pos] == 0
|
|
1205
|
+
and content[desc_end_pos + 1] != 0):
|
|
1206
|
+
desc_end_pos += 1
|
|
1207
|
+
desc = self._decode_string(
|
|
1208
|
+
encoding + content[desc_start_pos:desc_end_pos])
|
|
1209
|
+
return self._create_tag_image(
|
|
1210
|
+
content[desc_end_pos:], pic_type, mime_type, desc)
|
|
1211
|
+
|
|
1212
|
+
@staticmethod
|
|
1213
|
+
def _lrc_timestamp(seconds: float) -> str:
|
|
1214
|
+
cs = int(seconds * 100)
|
|
1215
|
+
minutes, cs = divmod(cs, 6000)
|
|
1216
|
+
seconds, cs = divmod(cs, 100)
|
|
1217
|
+
return f"{minutes:02d}:{seconds:02d}.{cs:02d}"
|
|
1218
|
+
|
|
1219
|
+
def _parse_synced_lyrics(self, content: bytes) -> str:
|
|
1220
|
+
# Convert ID3 synced lyrics to LRC format
|
|
1221
|
+
content_length = len(content)
|
|
1222
|
+
encoding = content[:1]
|
|
1223
|
+
# skip language (3)
|
|
1224
|
+
timestamp_format = content[4:5]
|
|
1225
|
+
# skip content type (1)
|
|
1226
|
+
start_pos = 6
|
|
1227
|
+
end_pos = self._find_string_end_pos(content, encoding, start_pos)
|
|
1228
|
+
lyrics = ""
|
|
1229
|
+
offset = end_pos
|
|
1230
|
+
found_line = False
|
|
1231
|
+
while offset < content_length:
|
|
1232
|
+
end_pos = self._find_string_end_pos(content, encoding, offset)
|
|
1233
|
+
value = self._decode_string(
|
|
1234
|
+
encoding + content[offset:end_pos]).lstrip('\n')
|
|
1235
|
+
offset = end_pos
|
|
1236
|
+
time = unpack('>I', content[offset:offset + 4])[0]
|
|
1237
|
+
offset += 4
|
|
1238
|
+
if found_line:
|
|
1239
|
+
lyrics += '\n'
|
|
1240
|
+
found_line = True
|
|
1241
|
+
if timestamp_format == b'\x02':
|
|
1242
|
+
# time in milliseconds
|
|
1243
|
+
timestamp = self._lrc_timestamp(time / 1000)
|
|
1244
|
+
else:
|
|
1245
|
+
lyrics += value
|
|
1246
|
+
continue
|
|
1247
|
+
lyrics += f'[{timestamp}]{value}'
|
|
1248
|
+
return lyrics
|
|
1249
|
+
|
|
1118
1250
|
def _parse_frame(self,
|
|
1119
1251
|
fh: BinaryIO,
|
|
1120
1252
|
total_size: int,
|
|
@@ -1125,8 +1257,10 @@ class _ID3(TinyTag):
|
|
|
1125
1257
|
is_synchsafe_int = id3version == 4
|
|
1126
1258
|
header = fh.read(header_len)
|
|
1127
1259
|
if len(header) != header_len:
|
|
1128
|
-
return
|
|
1129
|
-
frame_id =
|
|
1260
|
+
return -1
|
|
1261
|
+
frame_id = header[:frame_size_bytes]
|
|
1262
|
+
if frame_id in self._EMPTY_FRAME_IDS:
|
|
1263
|
+
return -1
|
|
1130
1264
|
frame_size: int
|
|
1131
1265
|
if frame_size_bytes == 3:
|
|
1132
1266
|
frame_size = unpack('>I', b'\x00' + header[3:6])[0]
|
|
@@ -1135,25 +1269,22 @@ class _ID3(TinyTag):
|
|
|
1135
1269
|
else:
|
|
1136
1270
|
frame_size = unpack('>I', header[4:8])[0]
|
|
1137
1271
|
if _DEBUG:
|
|
1138
|
-
print(f'Found id3 Frame {frame_id} at '
|
|
1272
|
+
print(f'Found id3 Frame {frame_id!r} at '
|
|
1139
1273
|
f'{fh.tell()}-{fh.tell() + frame_size} of {self.filesize}')
|
|
1140
1274
|
if frame_size > total_size:
|
|
1141
1275
|
# invalid frame size, stop here
|
|
1142
|
-
return
|
|
1143
|
-
content = fh.read(frame_size)
|
|
1144
|
-
fieldname = self._ID3_MAPPING.get(frame_id)
|
|
1276
|
+
return -1
|
|
1145
1277
|
should_set_field = True
|
|
1146
|
-
if
|
|
1147
|
-
|
|
1148
|
-
return frame_size
|
|
1278
|
+
if self._parse_tags and frame_id in self._ID3_MAPPING:
|
|
1279
|
+
fieldname = self._ID3_MAPPING[frame_id]
|
|
1149
1280
|
language = fieldname in {'comment', 'other.lyrics'}
|
|
1150
|
-
value = self._decode_string(
|
|
1281
|
+
value = self._decode_string(fh.read(frame_size), language)
|
|
1151
1282
|
if not value:
|
|
1152
1283
|
return frame_size
|
|
1153
1284
|
if fieldname == "comment":
|
|
1154
1285
|
# check if comment is a key-value pair (used by iTunes)
|
|
1155
|
-
should_set_field = not self.
|
|
1156
|
-
elif fieldname in {'track', 'disc'}:
|
|
1286
|
+
should_set_field = not self._parse_custom_field(value)
|
|
1287
|
+
elif fieldname in {'track', 'disc', 'other.movement'}:
|
|
1157
1288
|
if '/' in value:
|
|
1158
1289
|
value, total = value.split('/')[:2]
|
|
1159
1290
|
if total.isdecimal():
|
|
@@ -1174,54 +1305,52 @@ class _ID3(TinyTag):
|
|
|
1174
1305
|
genre_id = int(parens_text)
|
|
1175
1306
|
if 0 <= genre_id < len(self._ID3V1_GENRES):
|
|
1176
1307
|
value = self._ID3V1_GENRES[genre_id]
|
|
1308
|
+
elif fieldname == 'modern_grouping':
|
|
1309
|
+
self._modern_grouping_values.append(value)
|
|
1310
|
+
should_set_field = False
|
|
1311
|
+
elif fieldname == 'legacy_grouping':
|
|
1312
|
+
self._legacy_grouping_values.append(value)
|
|
1313
|
+
should_set_field = False
|
|
1177
1314
|
if should_set_field:
|
|
1178
1315
|
self._set_field(fieldname, value)
|
|
1179
|
-
elif frame_id in self.
|
|
1316
|
+
elif self._parse_tags and frame_id in self._SYNCED_LYRICS_FRAME_IDS:
|
|
1317
|
+
lyrics = self._parse_synced_lyrics(fh.read(frame_size))
|
|
1318
|
+
self._set_field('other.lyrics', lyrics)
|
|
1319
|
+
elif self._parse_tags and frame_id in self._CUSTOM_FRAME_IDS:
|
|
1180
1320
|
# custom fields
|
|
1181
|
-
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
|
|
1185
|
-
elif frame_id in self._IMAGE_FRAME_IDS:
|
|
1186
|
-
if self._load_image:
|
|
1187
|
-
# See section 4.14: http://id3.org/id3v2.4.0-frames
|
|
1188
|
-
encoding = content[:1]
|
|
1189
|
-
if frame_id == 'PIC': # ID3 v2.2:
|
|
1190
|
-
imgformat = self._decode_string(content[1:4]).lower()
|
|
1191
|
-
mime_type = self._ID3V2_2_IMAGE_FORMATS.get(imgformat)
|
|
1192
|
-
# skip encoding (1), imgformat (3), pictype(1)
|
|
1193
|
-
desc_start_pos = 5
|
|
1194
|
-
else: # ID3 v2.3+
|
|
1195
|
-
mime_end_pos = content.index(b'\x00', 1)
|
|
1196
|
-
mime_type = self._decode_string(
|
|
1197
|
-
content[1:mime_end_pos]).lower()
|
|
1198
|
-
# skip mtype, pictype(1)
|
|
1199
|
-
desc_start_pos = mime_end_pos + 2
|
|
1200
|
-
pic_type = content[desc_start_pos - 1]
|
|
1201
|
-
# latin1 and utf-8 are 1 byte
|
|
1202
|
-
if encoding in {b'\x00', b'\x03'}:
|
|
1203
|
-
desc_end_pos = content.find(b'\x00', desc_start_pos) + 1
|
|
1204
|
-
else:
|
|
1205
|
-
desc_end_pos = 0
|
|
1206
|
-
for i in range(desc_start_pos, len(content), 2):
|
|
1207
|
-
if content[i:i + 2] == b'\x00\x00':
|
|
1208
|
-
desc_end_pos = i + 2
|
|
1209
|
-
break
|
|
1210
|
-
desc = self._decode_string(
|
|
1211
|
-
encoding + content[desc_start_pos:desc_end_pos])
|
|
1212
|
-
field_name, image = self._create_tag_image(
|
|
1213
|
-
content[desc_end_pos:], pic_type, mime_type, desc)
|
|
1214
|
-
# pylint: disable=protected-access
|
|
1215
|
-
self.images._set_field(field_name, image)
|
|
1216
|
-
elif frame_id not in self._IGNORED_FRAME_IDS:
|
|
1321
|
+
value = self._decode_string(fh.read(frame_size))
|
|
1322
|
+
if value:
|
|
1323
|
+
self._parse_custom_field(value)
|
|
1324
|
+
elif self._parse_tags and frame_id not in self._IGNORED_FRAME_IDS:
|
|
1217
1325
|
# unknown, try to add to other dict
|
|
1218
|
-
|
|
1219
|
-
|
|
1220
|
-
|
|
1221
|
-
self.
|
|
1222
|
-
|
|
1326
|
+
value = self._decode_string(fh.read(frame_size))
|
|
1327
|
+
if value:
|
|
1328
|
+
self._set_field(
|
|
1329
|
+
self._OTHER_PREFIX + frame_id.decode('latin-1').lower(),
|
|
1330
|
+
value)
|
|
1331
|
+
elif self._load_image and frame_id in self._IMAGE_FRAME_IDS:
|
|
1332
|
+
field_name, image = self._parse_image(
|
|
1333
|
+
frame_id, fh.read(frame_size))
|
|
1334
|
+
# pylint: disable=protected-access
|
|
1335
|
+
self.images._set_field(field_name, image)
|
|
1336
|
+
else: # skip frame
|
|
1337
|
+
fh.seek(frame_size, SEEK_CUR)
|
|
1223
1338
|
return frame_size
|
|
1224
1339
|
|
|
1340
|
+
@staticmethod
|
|
1341
|
+
def _find_string_end_pos(content: bytes,
|
|
1342
|
+
encoding: bytes = b'\x00',
|
|
1343
|
+
start_pos: int = 0) -> int:
|
|
1344
|
+
# latin1 and utf-8 are 1 byte
|
|
1345
|
+
if encoding in {b'\x00', b'\x03'}:
|
|
1346
|
+
return content.find(b'\x00', start_pos) + 1
|
|
1347
|
+
end_pos = 0
|
|
1348
|
+
for i in range(start_pos, len(content), 2):
|
|
1349
|
+
if content[i:i + 2] == b'\x00\x00':
|
|
1350
|
+
end_pos = i + 2
|
|
1351
|
+
break
|
|
1352
|
+
return end_pos
|
|
1353
|
+
|
|
1225
1354
|
def _decode_string(self, value: bytes, language: bool = False) -> str:
|
|
1226
1355
|
default_encoding = 'ISO-8859-1'
|
|
1227
1356
|
if self._default_encoding:
|
|
@@ -1294,6 +1423,7 @@ class _Ogg(TinyTag):
|
|
|
1294
1423
|
'copyright': 'other.copyright',
|
|
1295
1424
|
'isrc': 'other.isrc',
|
|
1296
1425
|
'lyrics': 'other.lyrics',
|
|
1426
|
+
'unsyncedlyrics': 'other.lyrics',
|
|
1297
1427
|
'publisher': 'other.publisher',
|
|
1298
1428
|
'language': 'other.language',
|
|
1299
1429
|
'director': 'other.director',
|
|
@@ -1310,6 +1440,13 @@ class _Ogg(TinyTag):
|
|
|
1310
1440
|
'license': 'other.license',
|
|
1311
1441
|
'barcode': 'other.barcode',
|
|
1312
1442
|
'catalognumber': 'other.catalog_number',
|
|
1443
|
+
'movementname': 'other.movement_name',
|
|
1444
|
+
'movement': 'other.movement',
|
|
1445
|
+
'movementtotal': 'other.movement_total',
|
|
1446
|
+
'showmovement': 'other.show_movement',
|
|
1447
|
+
'grouping': 'other.grouping',
|
|
1448
|
+
'contentgroup': 'other.grouping',
|
|
1449
|
+
'work': 'other.work'
|
|
1313
1450
|
}
|
|
1314
1451
|
|
|
1315
1452
|
def __init__(self) -> None:
|
|
@@ -1450,7 +1587,7 @@ class _Ogg(TinyTag):
|
|
|
1450
1587
|
elif value:
|
|
1451
1588
|
self._set_field(fieldname, value)
|
|
1452
1589
|
|
|
1453
|
-
def _parse_pages(self, fh: BinaryIO) -> Iterator[
|
|
1590
|
+
def _parse_pages(self, fh: BinaryIO) -> Iterator[bytearray]:
|
|
1454
1591
|
# for the spec, see: https://wiki.xiph.org/Ogg
|
|
1455
1592
|
packet_data = bytearray()
|
|
1456
1593
|
current_serial = None
|
|
@@ -1500,6 +1637,8 @@ class _Ogg(TinyTag):
|
|
|
1500
1637
|
else:
|
|
1501
1638
|
self._audio_size += last_audio_size
|
|
1502
1639
|
last_audio_size = audio_size
|
|
1640
|
+
if eos:
|
|
1641
|
+
break
|
|
1503
1642
|
page_header = fh.read(header_len)
|
|
1504
1643
|
|
|
1505
1644
|
|
|
@@ -1552,7 +1691,7 @@ class _Wave(TinyTag):
|
|
|
1552
1691
|
subchunk_size = unpack('I', chunk_header[4:])[0]
|
|
1553
1692
|
# IFF chunks are padded to an even number of bytes
|
|
1554
1693
|
subchunk_size += subchunk_size % 2
|
|
1555
|
-
if subchunk_id == b'fmt '
|
|
1694
|
+
if self._parse_duration and subchunk_id == b'fmt ':
|
|
1556
1695
|
chunk = fh.read(subchunk_size)
|
|
1557
1696
|
_format_tag, channels, samplerate = unpack('<HHI', chunk[:8])
|
|
1558
1697
|
bitdepth = unpack('<H', chunk[14:16])[0]
|
|
@@ -1563,14 +1702,14 @@ class _Wave(TinyTag):
|
|
|
1563
1702
|
self.bitrate = samplerate * channels * bitdepth / 1000
|
|
1564
1703
|
self.channels, self.samplerate, self.bitdepth = (
|
|
1565
1704
|
channels, samplerate, bitdepth)
|
|
1566
|
-
elif subchunk_id == b'data'
|
|
1705
|
+
elif self._parse_duration and subchunk_id == b'data':
|
|
1567
1706
|
if (self.channels is not None and self.samplerate is not None
|
|
1568
1707
|
and self.bitdepth is not None):
|
|
1569
1708
|
self.duration = (
|
|
1570
1709
|
subchunk_size / self.channels / self.samplerate
|
|
1571
1710
|
/ (self.bitdepth / 8))
|
|
1572
1711
|
fh.seek(subchunk_size, SEEK_CUR)
|
|
1573
|
-
elif subchunk_id == b'LIST'
|
|
1712
|
+
elif self._parse_tags and subchunk_id == b'LIST':
|
|
1574
1713
|
chunk = fh.read(subchunk_size)
|
|
1575
1714
|
if chunk.startswith(b'INFO'):
|
|
1576
1715
|
walker = BytesIO(chunk)
|
|
@@ -1582,8 +1721,8 @@ class _Wave(TinyTag):
|
|
|
1582
1721
|
data_length += data_length % 2
|
|
1583
1722
|
# strip zero-byte
|
|
1584
1723
|
data = walker.read(data_length).split(b'\x00', 1)[0]
|
|
1585
|
-
|
|
1586
|
-
|
|
1724
|
+
if field in self._RIFF_MAPPING:
|
|
1725
|
+
fieldname = self._RIFF_MAPPING[field]
|
|
1587
1726
|
value = data.decode('utf-8', 'replace')
|
|
1588
1727
|
if fieldname == 'track':
|
|
1589
1728
|
if value.isdecimal():
|
|
@@ -1591,7 +1730,7 @@ class _Wave(TinyTag):
|
|
|
1591
1730
|
else:
|
|
1592
1731
|
self._set_field(fieldname, value)
|
|
1593
1732
|
field = walker.read(4)
|
|
1594
|
-
elif subchunk_id in {b'id3 ', b'ID3 '}
|
|
1733
|
+
elif self._parse_tags and subchunk_id in {b'id3 ', b'ID3 '}:
|
|
1595
1734
|
# pylint: disable=protected-access
|
|
1596
1735
|
id3 = _ID3()
|
|
1597
1736
|
id3._filehandler = fh
|
|
@@ -1635,7 +1774,7 @@ class _Flac(TinyTag):
|
|
|
1635
1774
|
is_last_block = block_header[0] & 0x80
|
|
1636
1775
|
size = unpack('>I', b'\x00' + block_header[1:])[0]
|
|
1637
1776
|
# http://xiph.org/flac/format.html#metadata_block_streaminfo
|
|
1638
|
-
if block_type == self._STREAMINFO
|
|
1777
|
+
if self._parse_duration and block_type == self._STREAMINFO:
|
|
1639
1778
|
head = fh.read(size)
|
|
1640
1779
|
if len(head) < 34: # invalid streaminfo
|
|
1641
1780
|
break
|
|
@@ -1665,13 +1804,13 @@ class _Flac(TinyTag):
|
|
|
1665
1804
|
self.samplerate = sr
|
|
1666
1805
|
if duration > 0:
|
|
1667
1806
|
self.bitrate = self.filesize * 8 / duration / 1000
|
|
1668
|
-
elif block_type == self._VORBIS_COMMENT
|
|
1807
|
+
elif self._parse_tags and block_type == self._VORBIS_COMMENT:
|
|
1669
1808
|
# pylint: disable=protected-access
|
|
1670
1809
|
walker = BytesIO(fh.read(size))
|
|
1671
1810
|
oggtag = _Ogg()
|
|
1672
1811
|
oggtag._parse_vorbis_comment(walker)
|
|
1673
1812
|
self._update(oggtag)
|
|
1674
|
-
elif block_type == self._PICTURE
|
|
1813
|
+
elif self._load_image and block_type == self._PICTURE:
|
|
1675
1814
|
fieldname, value = self._parse_image(fh)
|
|
1676
1815
|
# pylint: disable=protected-access
|
|
1677
1816
|
self.images._set_field(fieldname, value)
|
|
@@ -1730,6 +1869,8 @@ class _Wma(TinyTag):
|
|
|
1730
1869
|
'WM/Media': 'other.media',
|
|
1731
1870
|
'WM/Barcode': 'other.barcode',
|
|
1732
1871
|
'WM/CatalogNo': 'other.catalog_number',
|
|
1872
|
+
'WM/ContentGroupDescription': 'other.grouping',
|
|
1873
|
+
'WM/Work': 'other.work'
|
|
1733
1874
|
}
|
|
1734
1875
|
_UNPACK_FORMATS = {
|
|
1735
1876
|
1: '<B',
|
|
@@ -1765,7 +1906,7 @@ class _Wma(TinyTag):
|
|
|
1765
1906
|
if object_size == 0 or object_size > self.filesize:
|
|
1766
1907
|
break # invalid object, stop parsing.
|
|
1767
1908
|
object_id = object_header[:16]
|
|
1768
|
-
if object_id == self._ASF_CONTENT_DESC
|
|
1909
|
+
if self._parse_tags and object_id == self._ASF_CONTENT_DESC:
|
|
1769
1910
|
walker = BytesIO(fh.read(object_size - header_len))
|
|
1770
1911
|
(title_length, author_length,
|
|
1771
1912
|
copyright_length, description_length,
|
|
@@ -1782,7 +1923,7 @@ class _Wma(TinyTag):
|
|
|
1782
1923
|
walker.read(length).decode('utf-16', 'replace'))
|
|
1783
1924
|
if not i_field_name.startswith('_') and value:
|
|
1784
1925
|
self._set_field(i_field_name, value)
|
|
1785
|
-
elif object_id == self._ASF_EXT_CONTENT_DESC
|
|
1926
|
+
elif self._parse_tags and object_id == self._ASF_EXT_CONTENT_DESC:
|
|
1786
1927
|
# http://web.archive.org/web/20131203084402/http://msdn.microsoft.com/en-us/library/bb643323.aspx#_Toc509555195
|
|
1787
1928
|
walker = BytesIO(fh.read(object_size - header_len))
|
|
1788
1929
|
descriptor_count = unpack('<H', walker.read(2))[0]
|
|
@@ -1804,8 +1945,9 @@ class _Wma(TinyTag):
|
|
|
1804
1945
|
walker.seek(value_len, SEEK_CUR) # skip other values
|
|
1805
1946
|
continue
|
|
1806
1947
|
# try to get normalized field name
|
|
1807
|
-
|
|
1808
|
-
|
|
1948
|
+
if name in self._ASF_MAPPING:
|
|
1949
|
+
field_name = self._ASF_MAPPING[name]
|
|
1950
|
+
else: # custom field
|
|
1809
1951
|
if name.startswith('WM/'):
|
|
1810
1952
|
name = name[3:]
|
|
1811
1953
|
field_name = self._OTHER_PREFIX + name.lower()
|
|
@@ -1814,13 +1956,13 @@ class _Wma(TinyTag):
|
|
|
1814
1956
|
self._set_field(field_name, int(value))
|
|
1815
1957
|
elif value:
|
|
1816
1958
|
self._set_field(field_name, value)
|
|
1817
|
-
elif object_id == self._ASF_FILE_PROP
|
|
1959
|
+
elif self._parse_duration and object_id == self._ASF_FILE_PROP:
|
|
1818
1960
|
data = fh.read(object_size - header_len)
|
|
1819
1961
|
play_duration = unpack('<Q', data[40:48])[0] / 10000000
|
|
1820
1962
|
preroll = unpack('<Q', data[56:64])[0] / 1000
|
|
1821
1963
|
# subtract the preroll to get the actual duration
|
|
1822
1964
|
self.duration = max(play_duration - preroll, 0.0)
|
|
1823
|
-
elif object_id == self._ASF_STREAM_PROPS
|
|
1965
|
+
elif self._parse_duration and object_id == self._ASF_STREAM_PROPS:
|
|
1824
1966
|
data = fh.read(object_size - header_len)
|
|
1825
1967
|
stream_type = data[:16]
|
|
1826
1968
|
if stream_type == self._STREAM_TYPE_ASF_AUDIO_MEDIA:
|
|
@@ -1876,11 +2018,11 @@ class _Aiff(TinyTag):
|
|
|
1876
2018
|
subchunk_size = unpack('>I', chunk_header[4:])[0]
|
|
1877
2019
|
# IFF chunks are padded to an even number of bytes
|
|
1878
2020
|
subchunk_size += subchunk_size % 2
|
|
1879
|
-
if subchunk_id in self._AIFF_MAPPING
|
|
2021
|
+
if self._parse_tags and subchunk_id in self._AIFF_MAPPING:
|
|
1880
2022
|
value = self._unpad(
|
|
1881
2023
|
fh.read(subchunk_size).decode('utf-8', 'replace'))
|
|
1882
2024
|
self._set_field(self._AIFF_MAPPING[subchunk_id], value)
|
|
1883
|
-
elif subchunk_id == b'COMM'
|
|
2025
|
+
elif self._parse_duration and subchunk_id == b'COMM':
|
|
1884
2026
|
chunk = fh.read(subchunk_size)
|
|
1885
2027
|
channels, num_frames, bitdepth = unpack('>hLh', chunk[:8])
|
|
1886
2028
|
self.channels, self.bitdepth = channels, bitdepth
|
|
@@ -1894,7 +2036,7 @@ class _Aiff(TinyTag):
|
|
|
1894
2036
|
sr, duration, bitrate)
|
|
1895
2037
|
except OverflowError:
|
|
1896
2038
|
pass
|
|
1897
|
-
elif subchunk_id in {b'id3 ', b'ID3 '}
|
|
2039
|
+
elif self._parse_tags and subchunk_id in {b'id3 ', b'ID3 '}:
|
|
1898
2040
|
# pylint: disable=protected-access
|
|
1899
2041
|
id3 = _ID3()
|
|
1900
2042
|
id3._filehandler = fh
|
|
File without changes
|
|
File without changes
|
|
File without changes
|