tinytag 2.1.1__tar.gz → 2.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of tinytag might be problematic. Click here for more details.

@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tinytag
3
- Version: 2.1.1
3
+ Version: 2.2.0
4
4
  Summary: Read audio file metadata
5
5
  Keywords: metadata,audio,music,mp3,m4a,wav,ogg,opus,flac,wma,aiff
6
6
  Author: Tom Wallroth, Mat (mathiascode)
@@ -15,6 +15,7 @@ Classifier: Programming Language :: Python :: 3.10
15
15
  Classifier: Programming Language :: Python :: 3.11
16
16
  Classifier: Programming Language :: Python :: 3.12
17
17
  Classifier: Programming Language :: Python :: 3.13
18
+ Classifier: Programming Language :: Python :: 3.14
18
19
  Classifier: License :: OSI Approved :: MIT License
19
20
  Classifier: Development Status :: 5 - Production/Stable
20
21
  Classifier: Environment :: Web Environment
@@ -28,8 +29,10 @@ Classifier: Typing :: Typed
28
29
  License-File: LICENSE
29
30
  Requires-Dist: coverage ; extra == "tests"
30
31
  Requires-Dist: mypy ; extra == "tests"
32
+ Requires-Dist: mypy<1.19.0 ; extra == "tests" and ( platform_python_implementation == 'PyPy')
31
33
  Requires-Dist: pycodestyle ; extra == "tests"
32
34
  Requires-Dist: pylint ; extra == "tests"
35
+ Requires-Dist: pyright ; extra == "tests"
33
36
  Project-URL: Homepage, https://github.com/tinytag/tinytag
34
37
  Provides-Extra: tests
35
38
 
@@ -44,8 +47,6 @@ tinytag is a Python library for reading audio file metadata
44
47
 
45
48
  [![Build Status](https://img.shields.io/github/actions/workflow/status/tinytag/tinytag/tests.yml
46
49
  )](https://github.com/tinytag/tinytag/actions?query=workflow:%22Tests%22)
47
- [![Coverage Status](https://img.shields.io/coverallsCoverage/github/tinytag/tinytag
48
- )](https://coveralls.io/r/tinytag/tinytag)
49
50
  [![PyPI Version](https://img.shields.io/pypi/v/tinytag
50
51
  )](https://pypi.org/project/tinytag/)
51
52
  [![PyPI Downloads](https://img.shields.io/pypi/dm/tinytag
@@ -183,7 +184,7 @@ These are helpful when you need quick access to common metadata.
183
184
 
184
185
  ### Additional Metadata
185
186
 
186
- For additional values of the same field type, non-common metadata fields, or
187
+ For additional values of the same field type, uncommon metadata fields, or
187
188
  metadata specific to certain file formats, use `other`:
188
189
 
189
190
  tag.other # a dictionary of additional fields
@@ -203,6 +204,7 @@ present when files provide such metadata:
203
204
  director
204
205
  encoded_by
205
206
  encoder_settings
207
+ grouping
206
208
  initial_key
207
209
  isrc
208
210
  language
@@ -210,9 +212,14 @@ present when files provide such metadata:
210
212
  lyricist
211
213
  lyrics
212
214
  media
215
+ movement
216
+ movement_name
217
+ movement_total
213
218
  publisher
214
219
  set_subtitle
220
+ show_movement
215
221
  url
222
+ work
216
223
 
217
224
  Additional `other` field names not documented above may be present, but are
218
225
  format-specific and may change or disappear in future tinytag releases. If
@@ -422,6 +429,31 @@ TinyTag.get(file_obj=your_file_obj)
422
429
 
423
430
  ## Changelog
424
431
 
432
+ ### 2.2.0 (2025-12-15)
433
+
434
+ - Add support for movement, work and grouping fields
435
+ - ID3: Make synced lyrics available in 'other.lyrics' (LRC format)
436
+ - ID3: Continue reading after encountering empty frame
437
+ - ID3: Fix frame reading when image parsing is disabled
438
+ - ID3: Exclude more frames containing binary data
439
+ - ID3: Avoid unnecessary string decoding
440
+ - M4A: Support extended atom sizes
441
+ - M4A: Ensure all field names are lowercase
442
+ - OGG: Stop reading after reaching EOS page
443
+ - Vorbis: Map UNSYNCEDLYRICS field to other.lyrics
444
+
445
+ ### 2.1.2 (2025-08-14)
446
+
447
+ - M4A: Add a few missing additional metadata fields
448
+ - M4A: Support '©com' composer atom
449
+ - M4A: Fix reading of multi-value custom fields
450
+ - M4A: Use correct encoding when reading data names
451
+ - ID3: Don't read entire file to determine duration
452
+ - ID3: Skip stray null byte before image data
453
+ - Add missing `__version__` attribute
454
+ - Avoid some unnecessary work in hot code paths
455
+ - Improve a few incomplete type hints
456
+
425
457
  ### 2.1.1 (2025-04-23)
426
458
 
427
459
  - ID3: Stop removing 'b' character from strings
@@ -9,8 +9,6 @@ tinytag is a Python library for reading audio file metadata
9
9
 
10
10
  [![Build Status](https://img.shields.io/github/actions/workflow/status/tinytag/tinytag/tests.yml
11
11
  )](https://github.com/tinytag/tinytag/actions?query=workflow:%22Tests%22)
12
- [![Coverage Status](https://img.shields.io/coverallsCoverage/github/tinytag/tinytag
13
- )](https://coveralls.io/r/tinytag/tinytag)
14
12
  [![PyPI Version](https://img.shields.io/pypi/v/tinytag
15
13
  )](https://pypi.org/project/tinytag/)
16
14
  [![PyPI Downloads](https://img.shields.io/pypi/dm/tinytag
@@ -148,7 +146,7 @@ These are helpful when you need quick access to common metadata.
148
146
 
149
147
  ### Additional Metadata
150
148
 
151
- For additional values of the same field type, non-common metadata fields, or
149
+ For additional values of the same field type, uncommon metadata fields, or
152
150
  metadata specific to certain file formats, use `other`:
153
151
 
154
152
  tag.other # a dictionary of additional fields
@@ -168,6 +166,7 @@ present when files provide such metadata:
168
166
  director
169
167
  encoded_by
170
168
  encoder_settings
169
+ grouping
171
170
  initial_key
172
171
  isrc
173
172
  language
@@ -175,9 +174,14 @@ present when files provide such metadata:
175
174
  lyricist
176
175
  lyrics
177
176
  media
177
+ movement
178
+ movement_name
179
+ movement_total
178
180
  publisher
179
181
  set_subtitle
182
+ show_movement
180
183
  url
184
+ work
181
185
 
182
186
  Additional `other` field names not documented above may be present, but are
183
187
  format-specific and may change or disappear in future tinytag releases. If
@@ -387,6 +391,31 @@ TinyTag.get(file_obj=your_file_obj)
387
391
 
388
392
  ## Changelog
389
393
 
394
+ ### 2.2.0 (2025-12-15)
395
+
396
+ - Add support for movement, work and grouping fields
397
+ - ID3: Make synced lyrics available in 'other.lyrics' (LRC format)
398
+ - ID3: Continue reading after encountering empty frame
399
+ - ID3: Fix frame reading when image parsing is disabled
400
+ - ID3: Exclude more frames containing binary data
401
+ - ID3: Avoid unnecessary string decoding
402
+ - M4A: Support extended atom sizes
403
+ - M4A: Ensure all field names are lowercase
404
+ - OGG: Stop reading after reaching EOS page
405
+ - Vorbis: Map UNSYNCEDLYRICS field to other.lyrics
406
+
407
+ ### 2.1.2 (2025-08-14)
408
+
409
+ - M4A: Add a few missing additional metadata fields
410
+ - M4A: Support '©com' composer atom
411
+ - M4A: Fix reading of multi-value custom fields
412
+ - M4A: Use correct encoding when reading data names
413
+ - ID3: Don't read entire file to determine duration
414
+ - ID3: Skip stray null byte before image data
415
+ - Add missing `__version__` attribute
416
+ - Avoid some unnecessary work in hot code paths
417
+ - Improve a few incomplete type hints
418
+
390
419
  ### 2.1.1 (2025-04-23)
391
420
 
392
421
  - ID3: Stop removing 'b' character from strings
@@ -7,7 +7,6 @@ build-backend = "flit_core.buildapi"
7
7
 
8
8
  [project]
9
9
  name = "tinytag"
10
- version = "2.1.1"
11
10
  description = "Read audio file metadata"
12
11
  authors = [
13
12
  {name = "Tom Wallroth"},
@@ -36,6 +35,7 @@ classifiers = [
36
35
  "Programming Language :: Python :: 3.11",
37
36
  "Programming Language :: Python :: 3.12",
38
37
  "Programming Language :: Python :: 3.13",
38
+ "Programming Language :: Python :: 3.14",
39
39
  "License :: OSI Approved :: MIT License",
40
40
  "Development Status :: 5 - Production/Stable",
41
41
  "Environment :: Web Environment",
@@ -50,6 +50,7 @@ classifiers = [
50
50
  license = {file = "LICENSE"}
51
51
  readme = "README.md"
52
52
  requires-python = ">=3.7"
53
+ dynamic = ["version"]
53
54
 
54
55
  [project.urls]
55
56
  Homepage = "https://github.com/tinytag/tinytag"
@@ -58,8 +59,10 @@ Homepage = "https://github.com/tinytag/tinytag"
58
59
  tests = [
59
60
  "coverage",
60
61
  "mypy",
62
+ "mypy<1.19.0; platform_python_implementation == 'PyPy'",
61
63
  "pycodestyle",
62
- "pylint"
64
+ "pylint",
65
+ "pyright"
63
66
  ]
64
67
 
65
68
  [tool.flit.sdist]
@@ -114,3 +117,13 @@ py-version = "3.7"
114
117
 
115
118
  [tool.mypy]
116
119
  strict = true
120
+
121
+ [tool.coverage.report]
122
+ exclude_lines = [
123
+ "if TYPE_CHECKING:"
124
+ ]
125
+ precision = 2
126
+ show_missing = true
127
+
128
+ [tool.coverage.run]
129
+ relative_files = true
@@ -3,6 +3,8 @@
3
3
 
4
4
  """Audio file metadata reader."""
5
5
 
6
+ __version__ = '2.2.0'
7
+
6
8
  from .tinytag import (
7
9
  TinyTag, Image, Images, OtherFields, OtherImages,
8
10
  TinyTagException, ParseError, UnsupportedFormatError
@@ -34,15 +34,19 @@ from io import BytesIO
34
34
  from os import PathLike, SEEK_CUR, SEEK_END, environ, fsdecode
35
35
  from struct import unpack
36
36
 
37
+ TYPE_CHECKING = False
38
+
37
39
  # Lazy imports for type checking
38
- if False: # pylint: disable=using-constant-test
40
+ if TYPE_CHECKING:
39
41
  from collections.abc import Callable, Iterator # pylint: disable-all
40
- from typing import Any, BinaryIO, Dict, List
42
+ from typing import Any, BinaryIO, Dict, List, Union
41
43
 
42
44
  _StringListDict = Dict[str, List[str]]
43
45
  _ImageListDict = Dict[str, List["Image"]]
46
+ _DataTreeDict = Dict[
47
+ bytes, Union['_DataTreeDict', Callable[..., Dict[str, Any]]]]
44
48
  else:
45
- _StringListDict = _ImageListDict = dict
49
+ _StringListDict = _ImageListDict = _DataTreeDict = dict
46
50
 
47
51
  # some of the parsers can print debug info
48
52
  _DEBUG = bool(environ.get('TINYTAG_DEBUG'))
@@ -105,7 +109,7 @@ class TinyTag:
105
109
  self._parse_tags = True
106
110
  self._load_image = False
107
111
  self._tags_parsed = False
108
- self.__dict__: dict[str, str | float | Images | OtherFields]
112
+ self.__dict__: dict[str, str | float | Images | OtherFields | None]
109
113
 
110
114
  @classmethod
111
115
  def get(cls,
@@ -257,7 +261,7 @@ class TinyTag:
257
261
  self._parse_duration = duration
258
262
  self._load_image = image
259
263
  if self._filehandler is None:
260
- return
264
+ raise ValueError("File handle is required")
261
265
  if tags:
262
266
  self._parse_tag(self._filehandler)
263
267
  if duration:
@@ -271,14 +275,11 @@ class TinyTag:
271
275
  fieldname = fieldname[len(self._OTHER_PREFIX):]
272
276
  if check_conflict and fieldname in self.__dict__:
273
277
  fieldname = '_' + fieldname
274
- other_values = self.other.get(fieldname, [])
275
- if not isinstance(value, str) or value in other_values:
276
- return
277
- other_values.append(value)
278
+ if fieldname not in self.other:
279
+ self.other[fieldname] = []
280
+ self.other[fieldname].append(str(value))
278
281
  if _DEBUG:
279
- print(
280
- f'Setting other field "{fieldname}" to "{other_values!r}"')
281
- self.other[fieldname] = other_values
282
+ print(f'Adding value "{value} to field "{fieldname}"')
282
283
  return
283
284
  old_value = self.__dict__.get(fieldname)
284
285
  new_value = value
@@ -367,7 +368,7 @@ class Images:
367
368
  self.media: Image | None = None
368
369
 
369
370
  self.other: _ImageListDict = OtherImages()
370
- self.__dict__: dict[str, Image | OtherImages]
371
+ self.__dict__: dict[str, Image | OtherImages | None]
371
372
 
372
373
  @property
373
374
  def any(self) -> Image | None:
@@ -486,9 +487,10 @@ class _MP4(TinyTag):
486
487
  }
487
488
  _VERSIONED_ATOMS = {b'meta', b'stsd'} # those have an extra 4 byte header
488
489
  _FLAGGED_ATOMS = {b'stsd'} # these also have an extra 4 byte header
490
+ _ILST_PATH = [b'ftyp', b'moov', b'udta', b'meta', b'ilst']
489
491
 
490
- _audio_data_tree: dict[bytes, Any] | None = None
491
- _meta_data_tree: dict[bytes, Any] | None = None
492
+ _audio_data_tree: _DataTreeDict | None = None
493
+ _meta_data_tree: _DataTreeDict | None = None
492
494
 
493
495
  def _determine_duration(self, fh: BinaryIO) -> None:
494
496
  # https://developer.apple.com/library/mac/documentation/QuickTime/QTFF/QTFFChap3/qtff3.html
@@ -516,24 +518,32 @@ class _MP4(TinyTag):
516
518
  b'\xa9ART': {b'data': _MP4._data_parser('artist')},
517
519
  b'\xa9alb': {b'data': _MP4._data_parser('album')},
518
520
  b'\xa9cmt': {b'data': _MP4._data_parser('comment')},
521
+ b'\xa9com': {b'data': _MP4._data_parser('composer')},
519
522
  b'\xa9con': {b'data': _MP4._data_parser('other.conductor')},
520
- # need test-data for this
521
- # b'cpil': {b'data': _MP4._data_parser('other.compilation')},
522
523
  b'\xa9day': {b'data': _MP4._data_parser('year')},
523
524
  b'\xa9des': {b'data': _MP4._data_parser('other.description')},
524
525
  b'\xa9dir': {b'data': _MP4._data_parser('other.director')},
525
526
  b'\xa9gen': {b'data': _MP4._data_parser('genre')},
527
+ b'\xa9grp': {b'data': _MP4._data_parser('other.grouping')},
526
528
  b'\xa9lyr': {b'data': _MP4._data_parser('other.lyrics')},
527
- b'\xa9mvn': {b'data': _MP4._data_parser('movement')},
529
+ b'\xa9mvc': {
530
+ b'data': _MP4._data_parser('other.movement_total')
531
+ },
532
+ b'\xa9mvi': {b'data': _MP4._data_parser('other.movement')},
533
+ b'\xa9mvn': {
534
+ b'data': _MP4._data_parser('other.movement_name')
535
+ },
528
536
  b'\xa9nam': {b'data': _MP4._data_parser('title')},
529
537
  b'\xa9pub': {b'data': _MP4._data_parser('other.publisher')},
530
538
  b'\xa9too': {b'data': _MP4._data_parser('other.encoded_by')},
539
+ b'\xa9wrk': {b'data': _MP4._data_parser('other.work')},
531
540
  b'\xa9wrt': {b'data': _MP4._data_parser('composer')},
532
541
  b'aART': {b'data': _MP4._data_parser('albumartist')},
533
542
  b'cprt': {b'data': _MP4._data_parser('other.copyright')},
534
543
  b'desc': {b'data': _MP4._data_parser('other.description')},
535
544
  b'disk': {b'data': _MP4._nums_parser('disc', 'disc_total')},
536
545
  b'gnre': {b'data': _MP4._parse_id3v1_genre},
546
+ b'shwm': {b'data': _MP4._data_parser('other.show_movement')},
537
547
  b'trkn': {b'data': _MP4._nums_parser('track', 'track_total')},
538
548
  b'tmpo': {b'data': _MP4._data_parser('other.bpm')},
539
549
  b'covr': {b'data': _MP4._parse_cover_image},
@@ -543,16 +553,21 @@ class _MP4(TinyTag):
543
553
 
544
554
  def _traverse_atoms(self,
545
555
  fh: BinaryIO,
546
- path: dict[bytes, Any],
556
+ path: _DataTreeDict,
547
557
  stop_pos: int | None = None,
548
558
  curr_path: list[bytes] | None = None) -> None:
549
- header_len = 8
559
+ header_len = ext_size_len = 8
550
560
  atom_header = fh.read(header_len)
551
561
  while len(atom_header) == header_len:
552
- atom_size = unpack('>I', atom_header[:4])[0] - header_len
562
+ atom_size = unpack('>I', atom_header[:4])[0]
553
563
  atom_type = atom_header[4:]
554
564
  if curr_path is None: # keep track how we traversed in the tree
555
565
  curr_path = [atom_type]
566
+ if atom_size == 1: # 64-bit size
567
+ ext_size_header = fh.read(ext_size_len)
568
+ if len(ext_size_header) == ext_size_len:
569
+ atom_size = unpack('>Q', ext_size_header)[0] - ext_size_len
570
+ atom_size -= header_len
556
571
  if atom_size <= 0: # empty atom, jump to next one
557
572
  atom_header = fh.read(header_len)
558
573
  continue
@@ -562,8 +577,10 @@ class _MP4(TinyTag):
562
577
  f'atom: {atom_type!r} len: {atom_size + header_len}')
563
578
  if atom_type in self._VERSIONED_ATOMS: # jump atom version for now
564
579
  fh.seek(4, SEEK_CUR)
580
+ atom_size -= 4
565
581
  if atom_type in self._FLAGGED_ATOMS: # jump atom flags for now
566
582
  fh.seek(4, SEEK_CUR)
583
+ atom_size -= 4
567
584
  sub_path = path.get(atom_type, None)
568
585
  # if the path leaf is a dict, traverse deeper into the tree:
569
586
  if isinstance(sub_path, dict):
@@ -575,13 +592,27 @@ class _MP4(TinyTag):
575
592
  for fieldname, value in sub_path(fh.read(atom_size)).items():
576
593
  if _DEBUG:
577
594
  print(' ' * 4 * len(curr_path), 'FIELD: ', fieldname)
578
- if fieldname.startswith('images.'):
595
+ if isinstance(value, Image):
579
596
  if self._load_image:
580
597
  # pylint: disable=protected-access
581
598
  self.images._set_field(
582
599
  fieldname[len('images.'):], value)
583
- elif fieldname:
600
+ elif isinstance(value, list):
601
+ for subval in value:
602
+ self._set_field(fieldname, subval)
603
+ else:
584
604
  self._set_field(fieldname, value)
605
+ # unknown data atom, try to parse it
606
+ elif curr_path == self._ILST_PATH:
607
+ atom_end_pos = fh.tell() + atom_size
608
+ field_name = (
609
+ self._OTHER_PREFIX + atom_type.decode('latin-1').lower()
610
+ )
611
+ fh.seek(-header_len, SEEK_CUR)
612
+ self._traverse_atoms(
613
+ fh,
614
+ path={atom_type: {b'data': self._data_parser(field_name)}},
615
+ stop_pos=atom_end_pos, curr_path=curr_path + [atom_type])
585
616
  # if no action was specified using dict or callable, jump over atom
586
617
  else:
587
618
  fh.seek(atom_size, SEEK_CUR)
@@ -591,12 +622,8 @@ class _MP4(TinyTag):
591
622
  atom_header = fh.read(header_len) # read next atom
592
623
 
593
624
  @classmethod
594
- def _data_parser(
595
- cls, fieldname: str
596
- ) -> Callable[[bytes], dict[str, int | str | bytes | None]]:
597
- def _parse_data_atom(
598
- data_atom: bytes
599
- ) -> dict[str, int | str | bytes | None]:
625
+ def _data_parser(cls, fieldname: str) -> Callable[[bytes], dict[str, str]]:
626
+ def _parse_data_atom(data_atom: bytes) -> dict[str, str]:
600
627
  data_type = unpack('>I', data_atom[:4])[0]
601
628
  data = data_atom[8:]
602
629
  value = None
@@ -607,7 +634,9 @@ class _MP4(TinyTag):
607
634
  data_len = len(data)
608
635
  if data_len in fmts:
609
636
  value = str(unpack(fmts[data_len], data)[0])
610
- return {fieldname: value}
637
+ if value:
638
+ return {fieldname: value}
639
+ return {}
611
640
  return _parse_data_atom
612
641
 
613
642
  @classmethod
@@ -645,13 +674,11 @@ class _MP4(TinyTag):
645
674
  break
646
675
 
647
676
  @classmethod
648
- def _parse_custom_field(
649
- cls, data: bytes
650
- ) -> dict[str, int | str | bytes | None]:
677
+ def _parse_custom_field(cls, data: bytes) -> dict[str, list[str]]:
651
678
  fh = BytesIO(data)
652
679
  header_len = 8
653
680
  field_name = None
654
- data_atom = b''
681
+ values = []
655
682
  atom_header = fh.read(header_len)
656
683
  while len(atom_header) == header_len:
657
684
  atom_size = unpack('>I', atom_header[:4])[0] - header_len
@@ -662,15 +689,18 @@ class _MP4(TinyTag):
662
689
  # pylint: disable=protected-access
663
690
  field_name = cls._CUSTOM_FIELD_NAME_MAPPING.get(
664
691
  field_name, TinyTag._OTHER_PREFIX + field_name)
665
- elif atom_type == b'data':
692
+ elif atom_type == b'data' and field_name:
666
693
  data_atom = fh.read(atom_size)
694
+ parser = cls._data_parser(field_name)
695
+ atom_values = parser(data_atom)
696
+ if field_name in atom_values:
697
+ values.append(atom_values[field_name])
667
698
  else:
668
699
  fh.seek(atom_size, SEEK_CUR)
669
700
  atom_header = fh.read(header_len) # read next atom
670
- if len(data_atom) < 8 or field_name is None:
671
- return {}
672
- parser = cls._data_parser(field_name)
673
- return parser(data_atom)
701
+ if field_name and values:
702
+ return {field_name: values}
703
+ return {}
674
704
 
675
705
  @classmethod
676
706
  def _parse_audio_sample_entry_mp4a(cls, data: bytes) -> dict[str, int]:
@@ -727,31 +757,35 @@ class _ID3(TinyTag):
727
757
  _ID3_MAPPING = {
728
758
  # Mapping from Frame ID to a field of the TinyTag
729
759
  # https://exiftool.org/TagNames/ID3.html
730
- 'COMM': 'comment', 'COM': 'comment',
731
- 'TRCK': 'track', 'TRK': 'track',
732
- 'TYER': 'year', 'TYE': 'year', 'TDRC': 'year',
733
- 'TALB': 'album', 'TAL': 'album',
734
- 'TPE1': 'artist', 'TP1': 'artist',
735
- 'TIT2': 'title', 'TT2': 'title',
736
- 'TCON': 'genre', 'TCO': 'genre',
737
- 'TPOS': 'disc', 'TPA': 'disc',
738
- 'TPE2': 'albumartist', 'TP2': 'albumartist',
739
- 'TCOM': 'composer', 'TCM': 'composer',
740
- 'WOAR': 'other.url', 'WAR': 'other.url',
741
- 'TSRC': 'other.isrc', 'TRC': 'other.isrc',
742
- 'TCOP': 'other.copyright', 'TCR': 'other.copyright',
743
- 'TBPM': 'other.bpm', 'TBP': 'other.bpm',
744
- 'TKEY': 'other.initial_key', 'TKE': 'other.initial_key',
745
- 'TLAN': 'other.language', 'TLA': 'other.language',
746
- 'TPUB': 'other.publisher', 'TPB': 'other.publisher',
747
- 'USLT': 'other.lyrics', 'ULT': 'other.lyrics',
748
- 'TPE3': 'other.conductor', 'TP3': 'other.conductor',
749
- 'TEXT': 'other.lyricist', 'TXT': 'other.lyricist',
750
- 'TSST': 'other.set_subtitle',
751
- 'TENC': 'other.encoded_by', 'TEN': 'other.encoded_by',
752
- 'TSSE': 'other.encoder_settings', 'TSS': 'other.encoder_settings',
753
- 'TMED': 'other.media', 'TMT': 'other.media',
754
- 'WCOP': 'other.license',
760
+ b'COMM': 'comment', b'COM': 'comment',
761
+ b'TRCK': 'track', b'TRK': 'track',
762
+ b'TYER': 'year', b'TYE': 'year', b'TDRC': 'year',
763
+ b'TALB': 'album', b'TAL': 'album',
764
+ b'TPE1': 'artist', b'TP1': 'artist',
765
+ b'TIT2': 'title', b'TT2': 'title',
766
+ b'TCON': 'genre', b'TCO': 'genre',
767
+ b'TPOS': 'disc', b'TPA': 'disc',
768
+ b'TPE2': 'albumartist', b'TP2': 'albumartist',
769
+ b'TCOM': 'composer', b'TCM': 'composer',
770
+ b'WOAR': 'other.url', b'WAR': 'other.url',
771
+ b'TSRC': 'other.isrc', b'TRC': 'other.isrc',
772
+ b'TCOP': 'other.copyright', b'TCR': 'other.copyright',
773
+ b'TBPM': 'other.bpm', b'TBP': 'other.bpm',
774
+ b'TKEY': 'other.initial_key', b'TKE': 'other.initial_key',
775
+ b'TLAN': 'other.language', b'TLA': 'other.language',
776
+ b'TPUB': 'other.publisher', b'TPB': 'other.publisher',
777
+ b'USLT': 'other.lyrics', b'ULT': 'other.lyrics',
778
+ b'TPE3': 'other.conductor', b'TP3': 'other.conductor',
779
+ b'TEXT': 'other.lyricist', b'TXT': 'other.lyricist',
780
+ b'TSST': 'other.set_subtitle',
781
+ b'TENC': 'other.encoded_by', b'TEN': 'other.encoded_by',
782
+ b'TSSE': 'other.encoder_settings', b'TSS': 'other.encoder_settings',
783
+ b'TMED': 'other.media', b'TMT': 'other.media',
784
+ b'WCOP': 'other.license',
785
+ b'MVNM': 'other.movement_name',
786
+ b'MVIN': 'other.movement',
787
+ b'GRP1': 'modern_grouping', b'GP1': 'modern_grouping',
788
+ b'TIT1': 'legacy_grouping', b'TT1': 'legacy_grouping',
755
789
  }
756
790
  _ID3_MAPPING_CUSTOM = {
757
791
  'artists': 'artist',
@@ -759,23 +793,40 @@ class _ID3(TinyTag):
759
793
  'license': 'other.license',
760
794
  'barcode': 'other.barcode',
761
795
  'catalognumber': 'other.catalog_number',
796
+ 'showmovement': 'other.show_movement'
762
797
  }
763
- _IMAGE_FRAME_IDS = {'APIC', 'PIC'}
764
- _CUSTOM_FRAME_IDS = {'TXXX', 'TXX'}
798
+ _EMPTY_FRAME_IDS = {b'\x00\x00\x00\x00', b'\x00\x00\x00'}
799
+ _IMAGE_FRAME_IDS = {b'APIC', b'PIC'}
800
+ _CUSTOM_FRAME_IDS = {b'TXXX', b'TXX'}
801
+ _SYNCED_LYRICS_FRAME_IDS = {b'SYLT', b'SLT'}
765
802
  _IGNORED_FRAME_IDS = {
766
- 'AENC', 'CRA',
767
- 'ATXT',
768
- 'CHAP',
769
- 'COMR',
770
- 'CRM',
771
- 'CTOC',
772
- 'ENCR',
773
- 'GEOB', 'GEO',
774
- 'GRID',
775
- 'MCDI', 'MCI',
776
- 'PRIV',
777
- 'RGAD',
778
- 'STC', 'SYTC'
803
+ b'AENC', b'CRA',
804
+ b'APIC', b'PIC',
805
+ b'ASPI',
806
+ b'ATXT',
807
+ b'CHAP',
808
+ b'COMR',
809
+ b'CRM',
810
+ b'CTOC',
811
+ b'ENCR',
812
+ b'EQU2', b'EQU',
813
+ b'ETCO', b'ETC',
814
+ b'GEOB', b'GEO',
815
+ b'GRID',
816
+ b'LINK', b'LNK',
817
+ b'MCDI', b'MCI',
818
+ b'MLLT', b'MLL',
819
+ b'PCNT', b'CNT',
820
+ b'POPM', b'POP',
821
+ b'POSS',
822
+ b'PRIV',
823
+ b'RBUF', b'BUF',
824
+ b'RGAD',
825
+ b'RVA2', b'RVA',
826
+ b'RVRB', b'REV',
827
+ b'SEEK',
828
+ b'SIGN',
829
+ b'SYTC', b'STC',
779
830
  }
780
831
  _ID3V1_TAG_SIZE = 128
781
832
  _MAX_ESTIMATION_SEC = 30.0
@@ -825,9 +876,9 @@ class _ID3(TinyTag):
825
876
  'Psybient',
826
877
  )
827
878
  _ID3V2_2_IMAGE_FORMATS = {
828
- 'bmp': 'image/bmp',
829
- 'jpg': 'image/jpeg',
830
- 'png': 'image/png',
879
+ b'bmp': 'image/bmp',
880
+ b'jpg': 'image/jpeg',
881
+ b'png': 'image/png',
831
882
  }
832
883
  _IMAGE_TYPES = (
833
884
  'other.generic',
@@ -892,6 +943,8 @@ class _ID3(TinyTag):
892
943
  super().__init__()
893
944
  # save position after the ID3 tag for duration measurement speedup
894
945
  self._bytepos_after_id3v2 = -1
946
+ self._modern_grouping_values: list[str] = []
947
+ self._legacy_grouping_values: list[str] = []
895
948
 
896
949
  @staticmethod
897
950
  def _parse_xing_header(fh: BinaryIO) -> tuple[int, int]:
@@ -917,20 +970,17 @@ class _ID3(TinyTag):
917
970
  max_estimation_frames = (
918
971
  (self._MAX_ESTIMATION_SEC * 44100) // self._SAMPLES_PER_FRAME)
919
972
  frame_size_accu = 0
920
- audio_offset = 0
973
+ audio_offset = self._bytepos_after_id3v2
921
974
  frames = 0 # count frames for determining mp3 duration
922
975
  bitrate_accu = 0 # add up bitrates to find average bitrate to detect
923
976
  last_bitrates = set() # CBR mp3s (multiple frames with same bitrates)
924
977
  # seek to first position after id3 tag (speedup for large header)
925
978
  first_mpeg_id = None
926
979
  fh.seek(self._bytepos_after_id3v2)
927
- file_offset = fh.tell()
928
- walker = BytesIO(fh.read())
929
980
  while True:
930
981
  # reading through garbage until 11 '1' sync-bits are found
931
- header = walker.read(4)
982
+ header = fh.read(4)
932
983
  header_len = len(header)
933
- walker.seek(-header_len, SEEK_CUR)
934
984
  if header_len < 4:
935
985
  if frames:
936
986
  self.bitrate = bitrate_accu / frames
@@ -949,10 +999,12 @@ class _ID3(TinyTag):
949
999
  or mpeg_id == 1):
950
1000
  # invalid frame, find next sync header
951
1001
  idx = header.find(b'\xFF', 1)
952
- if idx == -1:
953
- # not found: jump over the current peek buffer
954
- idx = header_len
955
- walker.seek(max(idx, 1), SEEK_CUR)
1002
+ next_offset = header_len
1003
+ if idx != -1:
1004
+ next_offset -= idx
1005
+ fh.seek(idx - header_len, SEEK_CUR)
1006
+ if frames == 0:
1007
+ audio_offset += next_offset
956
1008
  continue
957
1009
  if first_mpeg_id is None:
958
1010
  first_mpeg_id = mpeg_id
@@ -964,12 +1016,12 @@ class _ID3(TinyTag):
964
1016
  # all the info we need, otherwise parse multiple frames to find the
965
1017
  # accurate average bitrate
966
1018
  if frames == 0 and self._USE_XING_HEADER:
967
- walker_offset = walker.tell()
968
- frame_content = walker.read(frame_length)
1019
+ prev_offset = header_len + audio_offset
1020
+ frame_content = fh.read(frame_length)
969
1021
  xing_header_offset = frame_content.find(b'Xing')
970
1022
  if xing_header_offset != -1:
971
- walker.seek(walker_offset + xing_header_offset)
972
- xframes, byte_count = self._parse_xing_header(walker)
1023
+ fh.seek(prev_offset + xing_header_offset)
1024
+ xframes, byte_count = self._parse_xing_header(fh)
973
1025
  if xframes > 0 and byte_count > 0:
974
1026
  # MPEG-2 Audio Layer III uses 576 samples per frame
975
1027
  samples_pf = self._SAMPLES_PER_FRAME
@@ -978,12 +1030,10 @@ class _ID3(TinyTag):
978
1030
  self.duration = dur = xframes * samples_pf / samplerate
979
1031
  self.bitrate = byte_count * 8 / dur / 1000
980
1032
  return
981
- walker.seek(walker_offset)
1033
+ fh.seek(prev_offset)
982
1034
 
983
1035
  frames += 1 # it's most probably a mp3 frame
984
1036
  bitrate_accu += frame_br
985
- if frames == 1:
986
- audio_offset = file_offset + walker.tell()
987
1037
  if frames <= self._CBR_DETECTION_FRAME_COUNT:
988
1038
  last_bitrates.add(frame_br)
989
1039
 
@@ -1002,7 +1052,7 @@ class _ID3(TinyTag):
1002
1052
  return
1003
1053
 
1004
1054
  if frame_length > 1: # jump over current frame body
1005
- walker.seek(frame_length, SEEK_CUR)
1055
+ fh.seek(frame_length - header_len, SEEK_CUR)
1006
1056
  if self.samplerate:
1007
1057
  self.duration = frames * self._SAMPLES_PER_FRAME / self.samplerate
1008
1058
 
@@ -1039,13 +1089,15 @@ class _ID3(TinyTag):
1039
1089
  fh.seek(extd_size - 6, SEEK_CUR) # jump over extended_header
1040
1090
  while parsed_size < size:
1041
1091
  frame_size = self._parse_frame(fh, size, id3version=major)
1042
- if frame_size == 0:
1092
+ if frame_size == -1:
1043
1093
  break
1044
1094
  parsed_size += frame_size
1045
1095
  fh.seek(end_pos)
1096
+ self._set_grouping_work_fields()
1046
1097
 
1047
1098
  def _parse_id3v1(self, fh: BinaryIO) -> None:
1048
- if fh.read(3) != b'TAG': # check if this is an ID3 v1 tag
1099
+ content = fh.read(3 + 30 + 30 + 30 + 4 + 30 + 1)
1100
+ if content[:3] != b'TAG': # check if this is an ID3 v1 tag
1049
1101
  return
1050
1102
 
1051
1103
  def asciidecode(x: bytes) -> str:
@@ -1053,24 +1105,23 @@ class _ID3(TinyTag):
1053
1105
  x.decode(self._default_encoding or 'latin1', 'replace'))
1054
1106
  # Only set fields that were not set by ID3v2 tags, as ID3v1
1055
1107
  # tags are more likely to be outdated or have encoding issues
1056
- fields = fh.read(30 + 30 + 30 + 4 + 30 + 1)
1057
1108
  if not self.title:
1058
- value = asciidecode(fields[:30])
1109
+ value = asciidecode(content[3:33])
1059
1110
  if value:
1060
1111
  self._set_field('title', value)
1061
1112
  if not self.artist:
1062
- value = asciidecode(fields[30:60])
1113
+ value = asciidecode(content[33:63])
1063
1114
  if value:
1064
1115
  self._set_field('artist', value)
1065
1116
  if not self.album:
1066
- value = asciidecode(fields[60:90])
1117
+ value = asciidecode(content[63:93])
1067
1118
  if value:
1068
1119
  self._set_field('album', value)
1069
1120
  if not self.year:
1070
- value = asciidecode(fields[90:94])
1121
+ value = asciidecode(content[93:97])
1071
1122
  if value:
1072
1123
  self._set_field('year', value)
1073
- comment = fields[94:124]
1124
+ comment = content[97:127]
1074
1125
  if b'\x00\x00' < comment[-2:] < b'\x01\x00':
1075
1126
  if self.track is None:
1076
1127
  self._set_field('track', ord(comment[-1:]))
@@ -1080,11 +1131,11 @@ class _ID3(TinyTag):
1080
1131
  if value:
1081
1132
  self._set_field('comment', value)
1082
1133
  if not self.genre:
1083
- genre_id = ord(fields[124:125])
1134
+ genre_id = ord(content[127:128])
1084
1135
  if genre_id < len(self._ID3V1_GENRES):
1085
1136
  self._set_field('genre', self._ID3V1_GENRES[genre_id])
1086
1137
 
1087
- def __parse_custom_field(self, content: str) -> bool:
1138
+ def _parse_custom_field(self, content: str) -> bool:
1088
1139
  custom_field_name, separator, value = content.partition('\x00')
1089
1140
  custom_field_name_lower = custom_field_name.lower()
1090
1141
  value = value.lstrip('\ufeff')
@@ -1096,6 +1147,18 @@ class _ID3(TinyTag):
1096
1147
  return True
1097
1148
  return False
1098
1149
 
1150
+ def _set_grouping_work_fields(self) -> None:
1151
+ # iTunes 12.5.4.42 added a new GRP1 frame for 'grouping', and
1152
+ # repurposed the TIT1 frame for 'work'. Handle this mess here.
1153
+ if self._modern_grouping_values:
1154
+ for value in self._modern_grouping_values:
1155
+ self._set_field('other.grouping', value)
1156
+ for value in self._legacy_grouping_values:
1157
+ self._set_field('other.work', value)
1158
+ return
1159
+ for value in self._legacy_grouping_values:
1160
+ self._set_field('other.grouping', value)
1161
+
1099
1162
  @classmethod
1100
1163
  def _create_tag_image(cls,
1101
1164
  data: bytes,
@@ -1115,6 +1178,75 @@ class _ID3(TinyTag):
1115
1178
  image.description = description
1116
1179
  return field_name, image
1117
1180
 
1181
+ def _parse_image(self,
1182
+ frame_id: bytes,
1183
+ content: bytes) -> tuple[str, Image]:
1184
+ # See section 4.14: http://id3.org/id3v2.4.0-frames
1185
+ encoding = content[:1]
1186
+ if frame_id == b'PIC': # ID3 v2.2:
1187
+ imgformat = content[1:4].lower()
1188
+ mime_type = self._ID3V2_2_IMAGE_FORMATS.get(imgformat)
1189
+ # skip encoding (1), imgformat (3), pictype(1)
1190
+ desc_start_pos = 5
1191
+ else: # ID3 v2.3+
1192
+ mime_start_pos = 1
1193
+ mime_end_pos = self._find_string_end_pos(
1194
+ content, start_pos=mime_start_pos)
1195
+ mime_type = self._decode_string(
1196
+ content[mime_start_pos:mime_end_pos]).lower()
1197
+ # skip mtype, pictype(1)
1198
+ desc_start_pos = mime_end_pos + 1
1199
+ pic_type = content[desc_start_pos - 1]
1200
+ desc_end_pos = self._find_string_end_pos(
1201
+ content, encoding, desc_start_pos)
1202
+ # skip stray null byte in broken file
1203
+ if (desc_end_pos + 1 < len(content)
1204
+ and content[desc_end_pos] == 0
1205
+ and content[desc_end_pos + 1] != 0):
1206
+ desc_end_pos += 1
1207
+ desc = self._decode_string(
1208
+ encoding + content[desc_start_pos:desc_end_pos])
1209
+ return self._create_tag_image(
1210
+ content[desc_end_pos:], pic_type, mime_type, desc)
1211
+
1212
+ @staticmethod
1213
+ def _lrc_timestamp(seconds: float) -> str:
1214
+ cs = int(seconds * 100)
1215
+ minutes, cs = divmod(cs, 6000)
1216
+ seconds, cs = divmod(cs, 100)
1217
+ return f"{minutes:02d}:{seconds:02d}.{cs:02d}"
1218
+
1219
+ def _parse_synced_lyrics(self, content: bytes) -> str:
1220
+ # Convert ID3 synced lyrics to LRC format
1221
+ content_length = len(content)
1222
+ encoding = content[:1]
1223
+ # skip language (3)
1224
+ timestamp_format = content[4:5]
1225
+ # skip content type (1)
1226
+ start_pos = 6
1227
+ end_pos = self._find_string_end_pos(content, encoding, start_pos)
1228
+ lyrics = ""
1229
+ offset = end_pos
1230
+ found_line = False
1231
+ while offset < content_length:
1232
+ end_pos = self._find_string_end_pos(content, encoding, offset)
1233
+ value = self._decode_string(
1234
+ encoding + content[offset:end_pos]).lstrip('\n')
1235
+ offset = end_pos
1236
+ time = unpack('>I', content[offset:offset + 4])[0]
1237
+ offset += 4
1238
+ if found_line:
1239
+ lyrics += '\n'
1240
+ found_line = True
1241
+ if timestamp_format == b'\x02':
1242
+ # time in milliseconds
1243
+ timestamp = self._lrc_timestamp(time / 1000)
1244
+ else:
1245
+ lyrics += value
1246
+ continue
1247
+ lyrics += f'[{timestamp}]{value}'
1248
+ return lyrics
1249
+
1118
1250
  def _parse_frame(self,
1119
1251
  fh: BinaryIO,
1120
1252
  total_size: int,
@@ -1125,8 +1257,10 @@ class _ID3(TinyTag):
1125
1257
  is_synchsafe_int = id3version == 4
1126
1258
  header = fh.read(header_len)
1127
1259
  if len(header) != header_len:
1128
- return 0
1129
- frame_id = self._decode_string(header[:frame_size_bytes])
1260
+ return -1
1261
+ frame_id = header[:frame_size_bytes]
1262
+ if frame_id in self._EMPTY_FRAME_IDS:
1263
+ return -1
1130
1264
  frame_size: int
1131
1265
  if frame_size_bytes == 3:
1132
1266
  frame_size = unpack('>I', b'\x00' + header[3:6])[0]
@@ -1135,25 +1269,22 @@ class _ID3(TinyTag):
1135
1269
  else:
1136
1270
  frame_size = unpack('>I', header[4:8])[0]
1137
1271
  if _DEBUG:
1138
- print(f'Found id3 Frame {frame_id} at '
1272
+ print(f'Found id3 Frame {frame_id!r} at '
1139
1273
  f'{fh.tell()}-{fh.tell() + frame_size} of {self.filesize}')
1140
1274
  if frame_size > total_size:
1141
1275
  # invalid frame size, stop here
1142
- return 0
1143
- content = fh.read(frame_size)
1144
- fieldname = self._ID3_MAPPING.get(frame_id)
1276
+ return -1
1145
1277
  should_set_field = True
1146
- if fieldname:
1147
- if not self._parse_tags:
1148
- return frame_size
1278
+ if self._parse_tags and frame_id in self._ID3_MAPPING:
1279
+ fieldname = self._ID3_MAPPING[frame_id]
1149
1280
  language = fieldname in {'comment', 'other.lyrics'}
1150
- value = self._decode_string(content, language)
1281
+ value = self._decode_string(fh.read(frame_size), language)
1151
1282
  if not value:
1152
1283
  return frame_size
1153
1284
  if fieldname == "comment":
1154
1285
  # check if comment is a key-value pair (used by iTunes)
1155
- should_set_field = not self.__parse_custom_field(value)
1156
- elif fieldname in {'track', 'disc'}:
1286
+ should_set_field = not self._parse_custom_field(value)
1287
+ elif fieldname in {'track', 'disc', 'other.movement'}:
1157
1288
  if '/' in value:
1158
1289
  value, total = value.split('/')[:2]
1159
1290
  if total.isdecimal():
@@ -1174,54 +1305,52 @@ class _ID3(TinyTag):
1174
1305
  genre_id = int(parens_text)
1175
1306
  if 0 <= genre_id < len(self._ID3V1_GENRES):
1176
1307
  value = self._ID3V1_GENRES[genre_id]
1308
+ elif fieldname == 'modern_grouping':
1309
+ self._modern_grouping_values.append(value)
1310
+ should_set_field = False
1311
+ elif fieldname == 'legacy_grouping':
1312
+ self._legacy_grouping_values.append(value)
1313
+ should_set_field = False
1177
1314
  if should_set_field:
1178
1315
  self._set_field(fieldname, value)
1179
- elif frame_id in self._CUSTOM_FRAME_IDS:
1316
+ elif self._parse_tags and frame_id in self._SYNCED_LYRICS_FRAME_IDS:
1317
+ lyrics = self._parse_synced_lyrics(fh.read(frame_size))
1318
+ self._set_field('other.lyrics', lyrics)
1319
+ elif self._parse_tags and frame_id in self._CUSTOM_FRAME_IDS:
1180
1320
  # custom fields
1181
- if self._parse_tags:
1182
- value = self._decode_string(content)
1183
- if value:
1184
- self.__parse_custom_field(value)
1185
- elif frame_id in self._IMAGE_FRAME_IDS:
1186
- if self._load_image:
1187
- # See section 4.14: http://id3.org/id3v2.4.0-frames
1188
- encoding = content[:1]
1189
- if frame_id == 'PIC': # ID3 v2.2:
1190
- imgformat = self._decode_string(content[1:4]).lower()
1191
- mime_type = self._ID3V2_2_IMAGE_FORMATS.get(imgformat)
1192
- # skip encoding (1), imgformat (3), pictype(1)
1193
- desc_start_pos = 5
1194
- else: # ID3 v2.3+
1195
- mime_end_pos = content.index(b'\x00', 1)
1196
- mime_type = self._decode_string(
1197
- content[1:mime_end_pos]).lower()
1198
- # skip mtype, pictype(1)
1199
- desc_start_pos = mime_end_pos + 2
1200
- pic_type = content[desc_start_pos - 1]
1201
- # latin1 and utf-8 are 1 byte
1202
- if encoding in {b'\x00', b'\x03'}:
1203
- desc_end_pos = content.find(b'\x00', desc_start_pos) + 1
1204
- else:
1205
- desc_end_pos = 0
1206
- for i in range(desc_start_pos, len(content), 2):
1207
- if content[i:i + 2] == b'\x00\x00':
1208
- desc_end_pos = i + 2
1209
- break
1210
- desc = self._decode_string(
1211
- encoding + content[desc_start_pos:desc_end_pos])
1212
- field_name, image = self._create_tag_image(
1213
- content[desc_end_pos:], pic_type, mime_type, desc)
1214
- # pylint: disable=protected-access
1215
- self.images._set_field(field_name, image)
1216
- elif frame_id not in self._IGNORED_FRAME_IDS:
1321
+ value = self._decode_string(fh.read(frame_size))
1322
+ if value:
1323
+ self._parse_custom_field(value)
1324
+ elif self._parse_tags and frame_id not in self._IGNORED_FRAME_IDS:
1217
1325
  # unknown, try to add to other dict
1218
- if self._parse_tags:
1219
- value = self._decode_string(content)
1220
- if value:
1221
- self._set_field(
1222
- self._OTHER_PREFIX + frame_id.lower(), value)
1326
+ value = self._decode_string(fh.read(frame_size))
1327
+ if value:
1328
+ self._set_field(
1329
+ self._OTHER_PREFIX + frame_id.decode('latin-1').lower(),
1330
+ value)
1331
+ elif self._load_image and frame_id in self._IMAGE_FRAME_IDS:
1332
+ field_name, image = self._parse_image(
1333
+ frame_id, fh.read(frame_size))
1334
+ # pylint: disable=protected-access
1335
+ self.images._set_field(field_name, image)
1336
+ else: # skip frame
1337
+ fh.seek(frame_size, SEEK_CUR)
1223
1338
  return frame_size
1224
1339
 
1340
+ @staticmethod
1341
+ def _find_string_end_pos(content: bytes,
1342
+ encoding: bytes = b'\x00',
1343
+ start_pos: int = 0) -> int:
1344
+ # latin1 and utf-8 are 1 byte
1345
+ if encoding in {b'\x00', b'\x03'}:
1346
+ return content.find(b'\x00', start_pos) + 1
1347
+ end_pos = 0
1348
+ for i in range(start_pos, len(content), 2):
1349
+ if content[i:i + 2] == b'\x00\x00':
1350
+ end_pos = i + 2
1351
+ break
1352
+ return end_pos
1353
+
1225
1354
  def _decode_string(self, value: bytes, language: bool = False) -> str:
1226
1355
  default_encoding = 'ISO-8859-1'
1227
1356
  if self._default_encoding:
@@ -1294,6 +1423,7 @@ class _Ogg(TinyTag):
1294
1423
  'copyright': 'other.copyright',
1295
1424
  'isrc': 'other.isrc',
1296
1425
  'lyrics': 'other.lyrics',
1426
+ 'unsyncedlyrics': 'other.lyrics',
1297
1427
  'publisher': 'other.publisher',
1298
1428
  'language': 'other.language',
1299
1429
  'director': 'other.director',
@@ -1310,6 +1440,13 @@ class _Ogg(TinyTag):
1310
1440
  'license': 'other.license',
1311
1441
  'barcode': 'other.barcode',
1312
1442
  'catalognumber': 'other.catalog_number',
1443
+ 'movementname': 'other.movement_name',
1444
+ 'movement': 'other.movement',
1445
+ 'movementtotal': 'other.movement_total',
1446
+ 'showmovement': 'other.show_movement',
1447
+ 'grouping': 'other.grouping',
1448
+ 'contentgroup': 'other.grouping',
1449
+ 'work': 'other.work'
1313
1450
  }
1314
1451
 
1315
1452
  def __init__(self) -> None:
@@ -1450,7 +1587,7 @@ class _Ogg(TinyTag):
1450
1587
  elif value:
1451
1588
  self._set_field(fieldname, value)
1452
1589
 
1453
- def _parse_pages(self, fh: BinaryIO) -> Iterator[bytes]:
1590
+ def _parse_pages(self, fh: BinaryIO) -> Iterator[bytearray]:
1454
1591
  # for the spec, see: https://wiki.xiph.org/Ogg
1455
1592
  packet_data = bytearray()
1456
1593
  current_serial = None
@@ -1500,6 +1637,8 @@ class _Ogg(TinyTag):
1500
1637
  else:
1501
1638
  self._audio_size += last_audio_size
1502
1639
  last_audio_size = audio_size
1640
+ if eos:
1641
+ break
1503
1642
  page_header = fh.read(header_len)
1504
1643
 
1505
1644
 
@@ -1552,7 +1691,7 @@ class _Wave(TinyTag):
1552
1691
  subchunk_size = unpack('I', chunk_header[4:])[0]
1553
1692
  # IFF chunks are padded to an even number of bytes
1554
1693
  subchunk_size += subchunk_size % 2
1555
- if subchunk_id == b'fmt ' and self._parse_duration:
1694
+ if self._parse_duration and subchunk_id == b'fmt ':
1556
1695
  chunk = fh.read(subchunk_size)
1557
1696
  _format_tag, channels, samplerate = unpack('<HHI', chunk[:8])
1558
1697
  bitdepth = unpack('<H', chunk[14:16])[0]
@@ -1563,14 +1702,14 @@ class _Wave(TinyTag):
1563
1702
  self.bitrate = samplerate * channels * bitdepth / 1000
1564
1703
  self.channels, self.samplerate, self.bitdepth = (
1565
1704
  channels, samplerate, bitdepth)
1566
- elif subchunk_id == b'data' and self._parse_duration:
1705
+ elif self._parse_duration and subchunk_id == b'data':
1567
1706
  if (self.channels is not None and self.samplerate is not None
1568
1707
  and self.bitdepth is not None):
1569
1708
  self.duration = (
1570
1709
  subchunk_size / self.channels / self.samplerate
1571
1710
  / (self.bitdepth / 8))
1572
1711
  fh.seek(subchunk_size, SEEK_CUR)
1573
- elif subchunk_id == b'LIST' and self._parse_tags:
1712
+ elif self._parse_tags and subchunk_id == b'LIST':
1574
1713
  chunk = fh.read(subchunk_size)
1575
1714
  if chunk.startswith(b'INFO'):
1576
1715
  walker = BytesIO(chunk)
@@ -1582,8 +1721,8 @@ class _Wave(TinyTag):
1582
1721
  data_length += data_length % 2
1583
1722
  # strip zero-byte
1584
1723
  data = walker.read(data_length).split(b'\x00', 1)[0]
1585
- fieldname = self._RIFF_MAPPING.get(field)
1586
- if fieldname:
1724
+ if field in self._RIFF_MAPPING:
1725
+ fieldname = self._RIFF_MAPPING[field]
1587
1726
  value = data.decode('utf-8', 'replace')
1588
1727
  if fieldname == 'track':
1589
1728
  if value.isdecimal():
@@ -1591,7 +1730,7 @@ class _Wave(TinyTag):
1591
1730
  else:
1592
1731
  self._set_field(fieldname, value)
1593
1732
  field = walker.read(4)
1594
- elif subchunk_id in {b'id3 ', b'ID3 '} and self._parse_tags:
1733
+ elif self._parse_tags and subchunk_id in {b'id3 ', b'ID3 '}:
1595
1734
  # pylint: disable=protected-access
1596
1735
  id3 = _ID3()
1597
1736
  id3._filehandler = fh
@@ -1635,7 +1774,7 @@ class _Flac(TinyTag):
1635
1774
  is_last_block = block_header[0] & 0x80
1636
1775
  size = unpack('>I', b'\x00' + block_header[1:])[0]
1637
1776
  # http://xiph.org/flac/format.html#metadata_block_streaminfo
1638
- if block_type == self._STREAMINFO and self._parse_duration:
1777
+ if self._parse_duration and block_type == self._STREAMINFO:
1639
1778
  head = fh.read(size)
1640
1779
  if len(head) < 34: # invalid streaminfo
1641
1780
  break
@@ -1665,13 +1804,13 @@ class _Flac(TinyTag):
1665
1804
  self.samplerate = sr
1666
1805
  if duration > 0:
1667
1806
  self.bitrate = self.filesize * 8 / duration / 1000
1668
- elif block_type == self._VORBIS_COMMENT and self._parse_tags:
1807
+ elif self._parse_tags and block_type == self._VORBIS_COMMENT:
1669
1808
  # pylint: disable=protected-access
1670
1809
  walker = BytesIO(fh.read(size))
1671
1810
  oggtag = _Ogg()
1672
1811
  oggtag._parse_vorbis_comment(walker)
1673
1812
  self._update(oggtag)
1674
- elif block_type == self._PICTURE and self._load_image:
1813
+ elif self._load_image and block_type == self._PICTURE:
1675
1814
  fieldname, value = self._parse_image(fh)
1676
1815
  # pylint: disable=protected-access
1677
1816
  self.images._set_field(fieldname, value)
@@ -1730,6 +1869,8 @@ class _Wma(TinyTag):
1730
1869
  'WM/Media': 'other.media',
1731
1870
  'WM/Barcode': 'other.barcode',
1732
1871
  'WM/CatalogNo': 'other.catalog_number',
1872
+ 'WM/ContentGroupDescription': 'other.grouping',
1873
+ 'WM/Work': 'other.work'
1733
1874
  }
1734
1875
  _UNPACK_FORMATS = {
1735
1876
  1: '<B',
@@ -1765,7 +1906,7 @@ class _Wma(TinyTag):
1765
1906
  if object_size == 0 or object_size > self.filesize:
1766
1907
  break # invalid object, stop parsing.
1767
1908
  object_id = object_header[:16]
1768
- if object_id == self._ASF_CONTENT_DESC and self._parse_tags:
1909
+ if self._parse_tags and object_id == self._ASF_CONTENT_DESC:
1769
1910
  walker = BytesIO(fh.read(object_size - header_len))
1770
1911
  (title_length, author_length,
1771
1912
  copyright_length, description_length,
@@ -1782,7 +1923,7 @@ class _Wma(TinyTag):
1782
1923
  walker.read(length).decode('utf-16', 'replace'))
1783
1924
  if not i_field_name.startswith('_') and value:
1784
1925
  self._set_field(i_field_name, value)
1785
- elif object_id == self._ASF_EXT_CONTENT_DESC and self._parse_tags:
1926
+ elif self._parse_tags and object_id == self._ASF_EXT_CONTENT_DESC:
1786
1927
  # http://web.archive.org/web/20131203084402/http://msdn.microsoft.com/en-us/library/bb643323.aspx#_Toc509555195
1787
1928
  walker = BytesIO(fh.read(object_size - header_len))
1788
1929
  descriptor_count = unpack('<H', walker.read(2))[0]
@@ -1804,8 +1945,9 @@ class _Wma(TinyTag):
1804
1945
  walker.seek(value_len, SEEK_CUR) # skip other values
1805
1946
  continue
1806
1947
  # try to get normalized field name
1807
- field_name = self._ASF_MAPPING.get(name)
1808
- if field_name is None: # custom field
1948
+ if name in self._ASF_MAPPING:
1949
+ field_name = self._ASF_MAPPING[name]
1950
+ else: # custom field
1809
1951
  if name.startswith('WM/'):
1810
1952
  name = name[3:]
1811
1953
  field_name = self._OTHER_PREFIX + name.lower()
@@ -1814,13 +1956,13 @@ class _Wma(TinyTag):
1814
1956
  self._set_field(field_name, int(value))
1815
1957
  elif value:
1816
1958
  self._set_field(field_name, value)
1817
- elif object_id == self._ASF_FILE_PROP and self._parse_duration:
1959
+ elif self._parse_duration and object_id == self._ASF_FILE_PROP:
1818
1960
  data = fh.read(object_size - header_len)
1819
1961
  play_duration = unpack('<Q', data[40:48])[0] / 10000000
1820
1962
  preroll = unpack('<Q', data[56:64])[0] / 1000
1821
1963
  # subtract the preroll to get the actual duration
1822
1964
  self.duration = max(play_duration - preroll, 0.0)
1823
- elif object_id == self._ASF_STREAM_PROPS and self._parse_duration:
1965
+ elif self._parse_duration and object_id == self._ASF_STREAM_PROPS:
1824
1966
  data = fh.read(object_size - header_len)
1825
1967
  stream_type = data[:16]
1826
1968
  if stream_type == self._STREAM_TYPE_ASF_AUDIO_MEDIA:
@@ -1876,11 +2018,11 @@ class _Aiff(TinyTag):
1876
2018
  subchunk_size = unpack('>I', chunk_header[4:])[0]
1877
2019
  # IFF chunks are padded to an even number of bytes
1878
2020
  subchunk_size += subchunk_size % 2
1879
- if subchunk_id in self._AIFF_MAPPING and self._parse_tags:
2021
+ if self._parse_tags and subchunk_id in self._AIFF_MAPPING:
1880
2022
  value = self._unpad(
1881
2023
  fh.read(subchunk_size).decode('utf-8', 'replace'))
1882
2024
  self._set_field(self._AIFF_MAPPING[subchunk_id], value)
1883
- elif subchunk_id == b'COMM' and self._parse_duration:
2025
+ elif self._parse_duration and subchunk_id == b'COMM':
1884
2026
  chunk = fh.read(subchunk_size)
1885
2027
  channels, num_frames, bitdepth = unpack('>hLh', chunk[:8])
1886
2028
  self.channels, self.bitdepth = channels, bitdepth
@@ -1894,7 +2036,7 @@ class _Aiff(TinyTag):
1894
2036
  sr, duration, bitrate)
1895
2037
  except OverflowError:
1896
2038
  pass
1897
- elif subchunk_id in {b'id3 ', b'ID3 '} and self._parse_tags:
2039
+ elif self._parse_tags and subchunk_id in {b'id3 ', b'ID3 '}:
1898
2040
  # pylint: disable=protected-access
1899
2041
  id3 = _ID3()
1900
2042
  id3._filehandler = fh
File without changes
File without changes
File without changes