tensorcodec 0.1.2__tar.gz → 0.1.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/PKG-INFO +14 -5
  2. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/README.md +12 -4
  3. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/compatibility.md +15 -0
  4. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/package_size.md +1 -1
  5. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/releasing.md +6 -5
  6. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/system_ffmpeg.md +9 -1
  7. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/README.md +4 -2
  8. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/native/Cargo.lock +2 -1
  9. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/native/Cargo.toml +2 -1
  10. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/native/src/ffmpeg.rs +158 -102
  11. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/native/src/lib.rs +13 -3
  12. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/packaging/size-baseline.json +12 -12
  13. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/pyproject.toml +2 -1
  14. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/build_ffmpeg.sh +1 -1
  15. tensorcodec-0.1.3/scripts/build_macos_wheel.sh +16 -0
  16. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/check_wheel_runtime.py +38 -0
  17. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/check_wheel_size.py +2 -2
  18. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/__init__.py +1 -1
  19. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/_frame.py +5 -1
  20. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/decoders/_decoder.py +42 -7
  21. tensorcodec-0.1.3/tests/test_native_output.py +296 -0
  22. tensorcodec-0.1.3/tests/test_open_cost.py +93 -0
  23. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/LICENSE +0 -0
  24. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/playback_semantics.md +0 -0
  25. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/FFmpeg-GPL-3.0.txt +0 -0
  26. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/FFmpeg-LGPL-3.0.txt +0 -0
  27. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/FFmpeg-NOTICE.md +0 -0
  28. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/OpenSSL.txt +0 -0
  29. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/Zstandard.txt +0 -0
  30. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/packaging/size-policy.json +0 -0
  31. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/build_linux_wheel.sh +0 -0
  32. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/build_nasm.sh +0 -0
  33. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/build_openssl.sh +0 -0
  34. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/configure_oracle_ffmpeg.py +0 -0
  35. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/update_size_comparison.py +0 -0
  36. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/_metadata.py +0 -0
  37. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/decoders/__init__.py +0 -0
  38. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/py.typed +0 -0
  39. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/__init__.py +0 -0
  40. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/conftest.py +0 -0
  41. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_audio_contract.py +0 -0
  42. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_differential.py +0 -0
  43. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_runtime.py +0 -0
  44. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_size_comparison.py +0 -0
  45. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_video_contract.py +0 -0
  46. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_video_fidelity.py +0 -0
  47. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_wheel_size.py +0 -0
  48. {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/utils.py +0 -0
@@ -1,8 +1,9 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tensorcodec
3
- Version: 0.1.2
3
+ Version: 0.1.3
4
4
  Classifier: Development Status :: 3 - Alpha
5
5
  Classifier: Operating System :: POSIX :: Linux
6
+ Classifier: Operating System :: MacOS :: MacOS X
6
7
  Classifier: Programming Language :: Python :: 3
7
8
  Classifier: Programming Language :: Rust
8
9
  Classifier: Topic :: Multimedia :: Video
@@ -48,7 +49,7 @@ CPU video/audio decoding with TorchCodec-style APIs and NumPy output.
48
49
  ranges are checked against TorchCodec 0.17.0 and independently generated media.
49
50
  - **Efficient batch decoding.** Rust/PyO3 bindings to FFmpeg process frame batches
50
51
  in a single native call, avoiding per-frame Python calls.
51
- - **Lightweight installation.** Linux wheels are 10.2–10.4 MiB (v0.1.1), including
52
+ - **Lightweight installation.** Linux wheels are 10.2–10.4 MiB (v0.1.2), including
52
53
  FFmpeg shared libraries. NumPy is the only Python dependency.
53
54
 
54
55
  ## Quick start
@@ -78,7 +79,7 @@ Arrays keep their storage after the decoder closes. Paths, URLs, encoded bytes,
78
79
 
79
80
  ## Features
80
81
 
81
- TensorCodec 0.1.2 relative to TorchCodec 0.17.0.
82
+ TensorCodec 0.1.3 relative to TorchCodec 0.17.0.
82
83
  ✓ supported · △ partial support · — not implemented.
83
84
 
84
85
  | Component | TensorCodec | TorchCodec 0.17.0 |
@@ -105,6 +106,7 @@ FPS-based frame queries are supported; clip samplers are a separate API.
105
106
  | NCHW / NHWC RGB output | ✓ | ✓ |
106
107
  | uint8 / float32 / automatic dtype | ✓ SDR and high-bit-depth video | ✓ |
107
108
  | uint16 RGB output | ✓ Full-range RGB48 | — |
109
+ | Native grayscale/depth and packed RGB(A) | ✓ Values preserved | — |
108
110
  | PQ / HLG decoding | ✓ Transfer-encoded RGB | ✓ |
109
111
  | Right-angle display rotation | ✓ | ✓ |
110
112
  | Audio ranges / resampling / channel mixing | ✓ float32 | ✓ |
@@ -118,6 +120,11 @@ float32 above 8 bits, or `output_dtype="uint16"` for full-range 16-bit RGB.
118
120
  HDR output retains PQ/HLG encoding without SDR tone mapping. Rotation is applied
119
121
  automatically, and metadata dimensions match the output.
120
122
 
123
+ For unmodified samples, use `VideoDecoder(path, output_format="native")`.
124
+ Supported formats: `gray`, `gray12le`, `gray16le/be`, `rgb24`, `rgba`.
125
+ Native output preserves channel count, integer values and pixel coordinates;
126
+ `expected_pixel_format` optionally asserts the source format.
127
+
121
128
  ## Package size
122
129
 
123
130
  <!-- wheel-size:start -->
@@ -153,8 +160,8 @@ See the [compatibility contract](docs/compatibility.md) and
153
160
 
154
161
  - **Wheels:** Linux x86_64 and ARM64 (aarch64), glibc 2.17+, CPython 3.10+.
155
162
  NumPy must also provide a compatible wheel; newer Python versions may require
156
- a newer glibc. macOS, Windows, musl/Alpine and free-threaded Python wheels are
157
- not release targets yet.
163
+ a newer glibc. macOS 14+ wheels support ARM64 and x86_64. Windows, musl/Alpine
164
+ and free-threaded Python wheels are not provided.
158
165
  - **Exact seeking:** scans packet timestamps when opening the decoder. Incorrect
159
166
  container keyframe flags can produce corrupt frames; repaired input or corrected
160
167
  frame mappings are needed in that case.
@@ -174,6 +181,8 @@ Source builds require Rust, Clang/libclang, pkg-config and FFmpeg 7 development
174
181
  headers/libraries. Python handles API and playback selection; Rust + PyO3 handles
175
182
  FFmpeg. Native decoding releases the GIL, allowing separate decoder instances to
176
183
  run concurrently across Python threads. Calls on the same instance are serialized.
184
+ The default is one FFmpeg thread per decoder; use independent workers for concurrent
185
+ windows and tune the total thread count to avoid oversubscription.
177
186
 
178
187
  ```sh
179
188
  uv sync --group dev --group oracle
@@ -24,7 +24,7 @@ CPU video/audio decoding with TorchCodec-style APIs and NumPy output.
24
24
  ranges are checked against TorchCodec 0.17.0 and independently generated media.
25
25
  - **Efficient batch decoding.** Rust/PyO3 bindings to FFmpeg process frame batches
26
26
  in a single native call, avoiding per-frame Python calls.
27
- - **Lightweight installation.** Linux wheels are 10.2–10.4 MiB (v0.1.1), including
27
+ - **Lightweight installation.** Linux wheels are 10.2–10.4 MiB (v0.1.2), including
28
28
  FFmpeg shared libraries. NumPy is the only Python dependency.
29
29
 
30
30
  ## Quick start
@@ -54,7 +54,7 @@ Arrays keep their storage after the decoder closes. Paths, URLs, encoded bytes,
54
54
 
55
55
  ## Features
56
56
 
57
- TensorCodec 0.1.2 relative to TorchCodec 0.17.0.
57
+ TensorCodec 0.1.3 relative to TorchCodec 0.17.0.
58
58
  ✓ supported · △ partial support · — not implemented.
59
59
 
60
60
  | Component | TensorCodec | TorchCodec 0.17.0 |
@@ -81,6 +81,7 @@ FPS-based frame queries are supported; clip samplers are a separate API.
81
81
  | NCHW / NHWC RGB output | ✓ | ✓ |
82
82
  | uint8 / float32 / automatic dtype | ✓ SDR and high-bit-depth video | ✓ |
83
83
  | uint16 RGB output | ✓ Full-range RGB48 | — |
84
+ | Native grayscale/depth and packed RGB(A) | ✓ Values preserved | — |
84
85
  | PQ / HLG decoding | ✓ Transfer-encoded RGB | ✓ |
85
86
  | Right-angle display rotation | ✓ | ✓ |
86
87
  | Audio ranges / resampling / channel mixing | ✓ float32 | ✓ |
@@ -94,6 +95,11 @@ float32 above 8 bits, or `output_dtype="uint16"` for full-range 16-bit RGB.
94
95
  HDR output retains PQ/HLG encoding without SDR tone mapping. Rotation is applied
95
96
  automatically, and metadata dimensions match the output.
96
97
 
98
+ For unmodified samples, use `VideoDecoder(path, output_format="native")`.
99
+ Supported formats: `gray`, `gray12le`, `gray16le/be`, `rgb24`, `rgba`.
100
+ Native output preserves channel count, integer values and pixel coordinates;
101
+ `expected_pixel_format` optionally asserts the source format.
102
+
97
103
  ## Package size
98
104
 
99
105
  <!-- wheel-size:start -->
@@ -129,8 +135,8 @@ See the [compatibility contract](docs/compatibility.md) and
129
135
 
130
136
  - **Wheels:** Linux x86_64 and ARM64 (aarch64), glibc 2.17+, CPython 3.10+.
131
137
  NumPy must also provide a compatible wheel; newer Python versions may require
132
- a newer glibc. macOS, Windows, musl/Alpine and free-threaded Python wheels are
133
- not release targets yet.
138
+ a newer glibc. macOS 14+ wheels support ARM64 and x86_64. Windows, musl/Alpine
139
+ and free-threaded Python wheels are not provided.
134
140
  - **Exact seeking:** scans packet timestamps when opening the decoder. Incorrect
135
141
  container keyframe flags can produce corrupt frames; repaired input or corrected
136
142
  frame mappings are needed in that case.
@@ -150,6 +156,8 @@ Source builds require Rust, Clang/libclang, pkg-config and FFmpeg 7 development
150
156
  headers/libraries. Python handles API and playback selection; Rust + PyO3 handles
151
157
  FFmpeg. Native decoding releases the GIL, allowing separate decoder instances to
152
158
  run concurrently across Python threads. Calls on the same instance are serialized.
159
+ The default is one FFmpeg thread per decoder; use independent workers for concurrent
160
+ windows and tune the total thread count to avoid oversubscription.
153
161
 
154
162
  ```sh
155
163
  uv sync --group dev --group oracle
@@ -66,3 +66,18 @@ TensorCodec extensions. Right-angle display rotations are applied automatically;
66
66
  metadata dimensions describe the rotated output. Reflected and non-right-angle
67
67
  display matrices remain unsupported. Color metadata and pixel aspect ratio
68
68
  describe the source; HDR output is not linear light or sRGB.
69
+
70
+ ## Native video output
71
+
72
+ `output_format="native"` bypasses color conversion and display transforms for
73
+ `gray`, `gray12le`, `gray16le`, `gray16be`, `rgb24` and `rgba`. NCHW/NHWC keeps
74
+ 1, 3 or 4 channels. Output uses host-endian uint8/uint16 without range scaling;
75
+ `output_dtype` may be omitted, `"auto"`, or the matching integer dtype.
76
+ `expected_pixel_format` asserts the source layout. Unsupported formats, dtype
77
+ conversions and changes of pixel format or dimensions fail explicitly.
78
+
79
+ Frame/FrameBatch `pixel_format` records the native source format and survives
80
+ indexing and FPS resampling. RGB results retain their existing behavior. Native
81
+ mode preserves encoded pixel coordinates, including inputs with display matrices.
82
+ Playback selection follows the same TorchCodec contract as RGB, including the
83
+ frame overlapping a range's start; it does not copy PyAV's legacy PTS-only range rule.
@@ -51,7 +51,7 @@ download; candidate builds never update it.
51
51
  To reproduce or recover a documentation update after publication:
52
52
 
53
53
  ```sh
54
- uv run --no-project python scripts/update_size_comparison.py --version 0.1.2
54
+ uv run --no-project python scripts/update_size_comparison.py --version 0.1.3
55
55
  ```
56
56
 
57
57
  Review and commit `README.md` and `packaging/size-baseline.json` together. The script
@@ -1,10 +1,11 @@
1
1
  # Publishing TensorCodec
2
2
 
3
- Release version: `0.1.2`. Distribution and import name: `tensorcodec`.
3
+ Release version: `0.1.3`. Distribution and import name: `tensorcodec`.
4
4
  Binary wheels target Linux x86_64 and ARM64 (aarch64), glibc 2.17+, CPython 3.10+.
5
5
  NumPy must also provide a compatible wheel for the selected Python/glibc pair.
6
6
  The wheel bundles shared FFmpeg 7.1.5 and OpenSSL 3.5.9 LTS; its only Python
7
- runtime dependency is NumPy. macOS/Windows wheels are not yet provided.
7
+ runtime dependency is NumPy. macOS 14+ ARM64/x86_64 wheels bundle the same minimal
8
+ runtime. Windows wheels are not provided.
8
9
 
9
10
  ## Trusted publisher configuration
10
11
 
@@ -24,7 +25,7 @@ is needed. Repository visibility does not need to change for a release.
24
25
 
25
26
  ## Release
26
27
 
27
- Run the **Publish to PyPI** workflow on `main`. It builds the portable Linux wheels
28
+ Run the **Publish to PyPI** workflow on `main`. It builds the portable Linux/macOS wheels
28
29
  and source distribution, checks package metadata, validates the pinned oracle
29
30
  and compares playback before uploading through PyPI Trusted Publishing. It uses
30
31
  the existing GitHub `pypi` environment. Publication fails if authorization is
@@ -36,8 +37,8 @@ gh workflow run publish.yml --repo MilkClouds/tensorcodec --ref main
36
37
 
37
38
  For a build and full validation without uploading, pass `--field publish=false`.
38
39
 
39
- Check the workflow and https://pypi.org/project/tensorcodec/0.1.2/ before reporting
40
- success. Verify a fresh `uv pip install tensorcodec==0.1.2` and a decode without
40
+ Check the workflow and https://pypi.org/project/tensorcodec/0.1.3/ before reporting
41
+ success. Verify a fresh `uv pip install tensorcodec==0.1.3` and a decode without
41
42
  Torch/PyAV on both architectures. Update the version before subsequent releases;
42
43
  PyPI versions cannot be overwritten.
43
44
 
@@ -37,7 +37,7 @@ test -f "$FFMPEG_DIR/include/libavcodec/avcodec.h"
37
37
  test -f "$FFMPEG_DIR/lib/libavcodec.so.61"
38
38
 
39
39
  uv venv
40
- uv pip install --no-binary tensorcodec 'tensorcodec==0.1.2'
40
+ uv pip install --no-binary tensorcodec 'tensorcodec==0.1.3'
41
41
  ```
42
42
 
43
43
  The version/build constraint avoids silently selecting an incompatible FFmpeg
@@ -75,3 +75,11 @@ The native extension build and playback contracts run against prebuilt FFmpeg
75
75
  no external FFmpeg library installation. Configuration-specific outputs can differ
76
76
  within the color-conversion tolerances described in the
77
77
  [compatibility contract](compatibility.md).
78
+
79
+ ## macOS wheels
80
+
81
+ Since 0.1.3, macOS 14+ wheels support Apple Silicon and Intel, bundling
82
+ FFmpeg/OpenSSL with `delocate`. CI tests
83
+ the installed wheels and clean Python 3.10/3.13 environments. Developers can run
84
+ `scripts/build_macos_wheel.sh` with Rust, Xcode tools, NASM, pkg-config, coreutils,
85
+ maturin and delocate installed in their build environment.
@@ -1,6 +1,6 @@
1
1
  # Bundled native libraries
2
2
 
3
- TensorCodec's own code is MIT licensed. Linux wheels bundle shared FFmpeg
3
+ TensorCodec's own code is MIT licensed. Linux and macOS wheels bundle shared FFmpeg
4
4
  7.1.5 libraries, built without GPL codec libraries using
5
5
  `scripts/build_ffmpeg.sh`. This configuration is LGPL-3.0-or-later. Its notices
6
6
  and both the LGPLv3 and incorporated GPLv3 texts are included here. The exact
@@ -8,9 +8,11 @@ upstream source is https://ffmpeg.org/releases/ffmpeg-7.1.5.tar.xz; the build
8
8
  script records the configuration. FFmpeg libraries remain dynamically linked
9
9
  and can be rebuilt/replaced with an ABI-compatible build.
10
10
 
11
- Portable Linux release wheels also bundle shared OpenSSL 3.5.9 LTS, licensed
11
+ Release wheels also bundle shared OpenSSL 3.5.9 LTS, licensed
12
12
  under Apache-2.0. Its exact source is
13
13
  https://github.com/openssl/openssl/releases/download/openssl-3.5.9/openssl-3.5.9.tar.gz;
14
14
  `scripts/build_openssl.sh` records the checksum and build configuration.
15
15
  System-library builds can additionally depend on Zstandard; its notices are
16
16
  retained here. Inspect repaired wheels when changing the native build.
17
+
18
+ PNG/Deflate decoding uses the platform zlib library.
@@ -455,9 +455,10 @@ checksum = "61c41af27dd6d1e27b1b16b489db798443478cef1f06a660c96db617ba5de3b1"
455
455
 
456
456
  [[package]]
457
457
  name = "tensorcodec-native"
458
- version = "0.1.2"
458
+ version = "0.1.3"
459
459
  dependencies = [
460
460
  "ffmpeg-sys-next",
461
+ "libc",
461
462
  "numpy",
462
463
  "pyo3",
463
464
  ]
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "tensorcodec-native"
3
- version = "0.1.2"
3
+ version = "0.1.3"
4
4
  edition = "2021"
5
5
  license = "MIT"
6
6
 
@@ -9,6 +9,7 @@ name = "_native"
9
9
  crate-type = ["cdylib"]
10
10
 
11
11
  [dependencies]
12
+ libc = "0.2"
12
13
  pyo3 = { version = "0.23", features = ["abi3-py310"] }
13
14
  numpy = "0.23"
14
15
  ffmpeg-sys-next = { version = "7.1", default-features = false, features = ["avcodec", "avformat", "swscale", "swresample"] }
@@ -64,7 +64,7 @@ enum Reader {
64
64
  }
65
65
  unsafe extern "C" fn read_memory(opaque: *mut c_void, buffer: *mut u8, size: i32) -> i32 {
66
66
  if size <= 0 {
67
- return -22;
67
+ return av::AVERROR(libc::EINVAL);
68
68
  }
69
69
  let reader = &mut *(opaque as *mut Reader);
70
70
  let input = match reader {
@@ -121,13 +121,13 @@ unsafe extern "C" fn seek_memory(opaque: *mut c_void, offset: i64, whence: i32)
121
121
  0 => 0,
122
122
  1 => input.position as i64,
123
123
  2 => input.data.len() as i64,
124
- _ => return -22,
124
+ _ => return i64::from(av::AVERROR(libc::EINVAL)),
125
125
  };
126
126
  let Some(position) = base.checked_add(offset) else {
127
- return -22;
127
+ return i64::from(av::AVERROR(libc::EINVAL));
128
128
  };
129
129
  if position < 0 || position > input.data.len() as i64 {
130
- return -22;
130
+ return i64::from(av::AVERROR(libc::EINVAL));
131
131
  }
132
132
  input.position = position as usize;
133
133
  position
@@ -146,6 +146,7 @@ pub struct Decoder {
146
146
  index: i32,
147
147
  time_base: av::AVRational,
148
148
  audio: bool,
149
+ video_layout: (i32, i32, av::AVPixelFormat),
149
150
  draining: bool,
150
151
  }
151
152
  // SAFETY: all pointers are uniquely owned. PyO3's mutable borrow plus the Python
@@ -194,6 +195,7 @@ impl Decoder {
194
195
  index: 0,
195
196
  time_base: av::AVRational { num: 0, den: 1 },
196
197
  audio,
198
+ video_layout: (0, 0, av::AVPixelFormat::AV_PIX_FMT_NONE),
197
199
  draining: false,
198
200
  };
199
201
  unsafe {
@@ -276,6 +278,13 @@ impl Decoder {
276
278
  if (*params).codec_type != media_type {
277
279
  return Err(Error("stream has the wrong media type".into(), true));
278
280
  }
281
+ if !audio {
282
+ this.video_layout = (
283
+ (*params).width,
284
+ (*params).height,
285
+ std::mem::transmute::<i32, av::AVPixelFormat>((*params).format),
286
+ );
287
+ }
279
288
  this.time_base = (*stream).time_base;
280
289
  if this.time_base.den <= 0 || this.time_base.num <= 0 {
281
290
  return Err(failure("invalid stream time base"));
@@ -336,7 +345,11 @@ impl Decoder {
336
345
  Ok(())
337
346
  }
338
347
 
339
- pub fn metadata<'py>(&self, py: Python<'py>) -> PyResult<Bound<'py, PyDict>> {
348
+ pub fn metadata<'py>(
349
+ &self,
350
+ py: Python<'py>,
351
+ apply_rotation: bool,
352
+ ) -> PyResult<Bound<'py, PyDict>> {
340
353
  let data = PyDict::new(py);
341
354
  unsafe {
342
355
  let stream = &*self.stream();
@@ -448,7 +461,8 @@ impl Decoder {
448
461
  params.nb_coded_side_data,
449
462
  av::AVPacketSideDataType::AV_PKT_DATA_DISPLAYMATRIX,
450
463
  );
451
- let rotation = if !side_data.is_null() && (*side_data).size >= 36 {
464
+ let rotation = if apply_rotation && !side_data.is_null() && (*side_data).size >= 36
465
+ {
452
466
  let matrix = (*side_data).data as *const i32;
453
467
  let determinant = *matrix as f64 * *matrix.add(4) as f64
454
468
  - *matrix.add(1) as f64 * *matrix.add(3) as f64;
@@ -513,7 +527,7 @@ impl Decoder {
513
527
  if code == av::AVERROR_EOF {
514
528
  return Ok(false);
515
529
  }
516
- if code != -11 {
530
+ if code != av::AVERROR(libc::EAGAIN) {
517
531
  check(code, "receive decoded frame")?;
518
532
  }
519
533
  if self.draining {
@@ -567,12 +581,31 @@ impl Decoder {
567
581
  if self.audio {
568
582
  return Err(failure("cannot decode video from audio stream"));
569
583
  }
584
+ let native = matches!(dtype, OutputDtype::Native);
585
+ let (width, height, source_format) = self.video_layout;
586
+ let (channels, dtype, big_endian) = if native {
587
+ use av::AVPixelFormat::*;
588
+ match source_format {
589
+ AV_PIX_FMT_GRAY8 => (1, OutputDtype::U8, false),
590
+ AV_PIX_FMT_GRAY12LE | AV_PIX_FMT_GRAY16LE => (1, OutputDtype::U16, false),
591
+ AV_PIX_FMT_GRAY16BE => (1, OutputDtype::U16, true),
592
+ AV_PIX_FMT_RGB24 => (3, OutputDtype::U8, false),
593
+ AV_PIX_FMT_RGBA => (4, OutputDtype::U8, false),
594
+ _ => {
595
+ return Err(Error(
596
+ "native output does not support this pixel format".into(),
597
+ true,
598
+ ))
599
+ }
600
+ }
601
+ } else {
602
+ (3, dtype, false)
603
+ };
570
604
  let high_depth = !matches!(dtype, OutputDtype::U8);
571
- let width = unsafe { (*self.codec).width } as usize;
572
- let height = unsafe { (*self.codec).height } as usize;
605
+ let (width, height) = (width as usize, height as usize);
573
606
  let count = width
574
607
  .checked_mul(height)
575
- .and_then(|n| n.checked_mul(3))
608
+ .and_then(|n| n.checked_mul(channels))
576
609
  .ok_or_else(|| failure("frame is too large"))?;
577
610
  let total = count
578
611
  .checked_mul(targets.len())
@@ -627,106 +660,119 @@ impl Decoder {
627
660
  return Err(failure("dynamic frame dimensions are unsupported"));
628
661
  }
629
662
  let input_format: av::AVPixelFormat = std::mem::transmute(frame.format);
630
- let output_format = if high_depth {
631
- av::AVPixelFormat::AV_PIX_FMT_RGB48LE
663
+ let output_frame = if native {
664
+ if input_format != source_format {
665
+ return Err(Error("pixel format changed within stream".into(), true));
666
+ }
667
+ frame
632
668
  } else {
633
- av::AVPixelFormat::AV_PIX_FMT_RGB24
634
- };
635
- let config = (
636
- frame.width,
637
- frame.height,
638
- input_format as i32,
639
- output_format as i32,
640
- );
641
- if self.scale_config != Some(config) {
642
- av::sws_freeContext(self.scale);
643
- self.scale = av::sws_getContext(
669
+ let output_format = if high_depth {
670
+ av::AVPixelFormat::AV_PIX_FMT_RGB48LE
671
+ } else {
672
+ av::AVPixelFormat::AV_PIX_FMT_RGB24
673
+ };
674
+ let config = (
644
675
  frame.width,
645
676
  frame.height,
646
- input_format,
647
- frame.width,
648
- frame.height,
649
- output_format,
650
- 0,
651
- ptr::null_mut(),
652
- ptr::null_mut(),
653
- ptr::null(),
677
+ input_format as i32,
678
+ output_format as i32,
654
679
  );
655
- self.scale_config = Some(config);
656
- }
657
- if self.scale.is_null() {
658
- return Err(failure("cannot initialize color conversion"));
659
- }
660
- let mut inverse = ptr::null_mut();
661
- let mut table = ptr::null_mut();
662
- let (
663
- mut source_range,
664
- mut destination_range,
665
- mut brightness,
666
- mut contrast,
667
- mut saturation,
668
- ) = (0, 0, 0, 0, 0);
669
- check(
670
- av::sws_getColorspaceDetails(
671
- self.scale,
672
- &mut inverse,
673
- &mut source_range,
674
- &mut table,
675
- &mut destination_range,
676
- &mut brightness,
677
- &mut contrast,
678
- &mut saturation,
679
- ),
680
- "read color conversion settings",
681
- )?;
682
- if frame.color_range != av::AVColorRange::AVCOL_RANGE_UNSPECIFIED {
683
- source_range =
684
- i32::from(frame.color_range == av::AVColorRange::AVCOL_RANGE_JPEG);
685
- }
686
- let coefficients = av::sws_getCoefficients(frame.colorspace as i32);
687
- check(
688
- av::sws_setColorspaceDetails(
689
- self.scale,
690
- coefficients,
691
- source_range,
692
- coefficients,
693
- destination_range,
694
- brightness,
695
- contrast,
696
- saturation,
697
- ),
698
- "configure color conversion",
699
- )?;
700
- if (*self.rgb_frame).width != frame.width
701
- || (*self.rgb_frame).height != frame.height
702
- || (*self.rgb_frame).format != output_format as i32
703
- {
704
- av::av_frame_unref(self.rgb_frame);
705
- (*self.rgb_frame).width = frame.width;
706
- (*self.rgb_frame).height = frame.height;
707
- (*self.rgb_frame).format = output_format as i32;
680
+ if self.scale_config != Some(config) {
681
+ av::sws_freeContext(self.scale);
682
+ self.scale = av::sws_getContext(
683
+ frame.width,
684
+ frame.height,
685
+ input_format,
686
+ frame.width,
687
+ frame.height,
688
+ output_format,
689
+ 0,
690
+ ptr::null_mut(),
691
+ ptr::null_mut(),
692
+ ptr::null(),
693
+ );
694
+ self.scale_config = Some(config);
695
+ }
696
+ if self.scale.is_null() {
697
+ return Err(failure("cannot initialize color conversion"));
698
+ }
699
+ let mut inverse = ptr::null_mut();
700
+ let mut table = ptr::null_mut();
701
+ let (
702
+ mut source_range,
703
+ mut destination_range,
704
+ mut brightness,
705
+ mut contrast,
706
+ mut saturation,
707
+ ) = (0, 0, 0, 0, 0);
708
708
  check(
709
- av::av_frame_get_buffer(self.rgb_frame, 32),
710
- "allocate RGB frame",
709
+ av::sws_getColorspaceDetails(
710
+ self.scale,
711
+ &mut inverse,
712
+ &mut source_range,
713
+ &mut table,
714
+ &mut destination_range,
715
+ &mut brightness,
716
+ &mut contrast,
717
+ &mut saturation,
718
+ ),
719
+ "read color conversion settings",
711
720
  )?;
721
+ if frame.color_range != av::AVColorRange::AVCOL_RANGE_UNSPECIFIED {
722
+ source_range =
723
+ i32::from(frame.color_range == av::AVColorRange::AVCOL_RANGE_JPEG);
724
+ }
725
+ let coefficients = av::sws_getCoefficients(frame.colorspace as i32);
726
+ check(
727
+ av::sws_setColorspaceDetails(
728
+ self.scale,
729
+ coefficients,
730
+ source_range,
731
+ coefficients,
732
+ destination_range,
733
+ brightness,
734
+ contrast,
735
+ saturation,
736
+ ),
737
+ "configure color conversion",
738
+ )?;
739
+ if (*self.rgb_frame).width != frame.width
740
+ || (*self.rgb_frame).height != frame.height
741
+ || (*self.rgb_frame).format != output_format as i32
742
+ {
743
+ av::av_frame_unref(self.rgb_frame);
744
+ (*self.rgb_frame).width = frame.width;
745
+ (*self.rgb_frame).height = frame.height;
746
+ (*self.rgb_frame).format = output_format as i32;
747
+ check(
748
+ av::av_frame_get_buffer(self.rgb_frame, 32),
749
+ "allocate RGB frame",
750
+ )?;
751
+ }
752
+ let rows = av::sws_scale(
753
+ self.scale,
754
+ frame.data.as_ptr() as *const *const u8,
755
+ frame.linesize.as_ptr(),
756
+ 0,
757
+ frame.height,
758
+ (*self.rgb_frame).data.as_ptr(),
759
+ (*self.rgb_frame).linesize.as_ptr(),
760
+ );
761
+ if rows != frame.height {
762
+ return Err(failure("color conversion failed"));
763
+ }
764
+ &*self.rgb_frame
765
+ };
766
+ let row_bytes = width * channels * if high_depth { 2 } else { 1 };
767
+ if output_frame.data[0].is_null()
768
+ || (output_frame.linesize[0].unsigned_abs() as usize) < row_bytes
769
+ {
770
+ return Err(failure("invalid decoded frame stride"));
712
771
  }
713
- let rows = av::sws_scale(
714
- self.scale,
715
- frame.data.as_ptr() as *const *const u8,
716
- frame.linesize.as_ptr(),
717
- 0,
718
- frame.height,
719
- (*self.rgb_frame).data.as_ptr(),
720
- (*self.rgb_frame).linesize.as_ptr(),
721
- );
722
- if rows != frame.height {
723
- return Err(failure("color conversion failed"));
724
- }
725
- let row_bytes = width * 3 * if high_depth { 2 } else { 1 };
726
772
  for row in 0..height {
727
773
  ptr::copy_nonoverlapping(
728
- (*self.rgb_frame).data[0]
729
- .add(row * (*self.rgb_frame).linesize[0] as usize),
774
+ output_frame.data[0]
775
+ .offset(row as isize * output_frame.linesize[0] as isize),
730
776
  pixels.as_mut_ptr().add(first * stride + row * row_bytes),
731
777
  row_bytes,
732
778
  );
@@ -760,10 +806,17 @@ impl Decoder {
760
806
  .as_chunks::<2>()
761
807
  .0
762
808
  .iter()
763
- .map(|p| u16::from_le_bytes(*p))
809
+ .map(|p| {
810
+ if big_endian {
811
+ u16::from_be_bytes(*p)
812
+ } else {
813
+ u16::from_le_bytes(*p)
814
+ }
815
+ })
764
816
  .collect(),
765
817
  ),
766
818
  OutputDtype::U8 => Pixels::U8(pixels),
819
+ OutputDtype::Native => unreachable!(),
767
820
  };
768
821
  Ok(Video {
769
822
  pixels,
@@ -771,6 +824,7 @@ impl Decoder {
771
824
  durations,
772
825
  width,
773
826
  height,
827
+ channels,
774
828
  })
775
829
  }
776
830
 
@@ -881,6 +935,7 @@ pub enum Pixels {
881
935
  F32(Vec<f32>),
882
936
  }
883
937
  pub enum OutputDtype {
938
+ Native,
884
939
  U8,
885
940
  U16,
886
941
  F32,
@@ -891,6 +946,7 @@ pub struct Video {
891
946
  pub durations: Vec<f64>,
892
947
  pub width: usize,
893
948
  pub height: usize,
949
+ pub channels: usize,
894
950
  }
895
951
  pub struct Audio {
896
952
  pub data: Vec<f32>,