tensorcodec 0.1.2__tar.gz → 0.1.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/PKG-INFO +14 -5
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/README.md +12 -4
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/compatibility.md +15 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/package_size.md +1 -1
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/releasing.md +6 -5
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/system_ffmpeg.md +9 -1
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/README.md +4 -2
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/native/Cargo.lock +2 -1
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/native/Cargo.toml +2 -1
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/native/src/ffmpeg.rs +158 -102
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/native/src/lib.rs +13 -3
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/packaging/size-baseline.json +12 -12
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/pyproject.toml +2 -1
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/build_ffmpeg.sh +1 -1
- tensorcodec-0.1.3/scripts/build_macos_wheel.sh +16 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/check_wheel_runtime.py +38 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/check_wheel_size.py +2 -2
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/__init__.py +1 -1
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/_frame.py +5 -1
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/decoders/_decoder.py +42 -7
- tensorcodec-0.1.3/tests/test_native_output.py +296 -0
- tensorcodec-0.1.3/tests/test_open_cost.py +93 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/LICENSE +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/docs/playback_semantics.md +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/FFmpeg-GPL-3.0.txt +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/FFmpeg-LGPL-3.0.txt +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/FFmpeg-NOTICE.md +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/OpenSSL.txt +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/licenses/Zstandard.txt +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/packaging/size-policy.json +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/build_linux_wheel.sh +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/build_nasm.sh +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/build_openssl.sh +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/configure_oracle_ffmpeg.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/scripts/update_size_comparison.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/_metadata.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/decoders/__init__.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/src/tensorcodec/py.typed +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/__init__.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/conftest.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_audio_contract.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_differential.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_runtime.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_size_comparison.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_video_contract.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_video_fidelity.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/test_wheel_size.py +0 -0
- {tensorcodec-0.1.2 → tensorcodec-0.1.3}/tests/utils.py +0 -0
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tensorcodec
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.3
|
|
4
4
|
Classifier: Development Status :: 3 - Alpha
|
|
5
5
|
Classifier: Operating System :: POSIX :: Linux
|
|
6
|
+
Classifier: Operating System :: MacOS :: MacOS X
|
|
6
7
|
Classifier: Programming Language :: Python :: 3
|
|
7
8
|
Classifier: Programming Language :: Rust
|
|
8
9
|
Classifier: Topic :: Multimedia :: Video
|
|
@@ -48,7 +49,7 @@ CPU video/audio decoding with TorchCodec-style APIs and NumPy output.
|
|
|
48
49
|
ranges are checked against TorchCodec 0.17.0 and independently generated media.
|
|
49
50
|
- **Efficient batch decoding.** Rust/PyO3 bindings to FFmpeg process frame batches
|
|
50
51
|
in a single native call, avoiding per-frame Python calls.
|
|
51
|
-
- **Lightweight installation.** Linux wheels are 10.2–10.4 MiB (v0.1.
|
|
52
|
+
- **Lightweight installation.** Linux wheels are 10.2–10.4 MiB (v0.1.2), including
|
|
52
53
|
FFmpeg shared libraries. NumPy is the only Python dependency.
|
|
53
54
|
|
|
54
55
|
## Quick start
|
|
@@ -78,7 +79,7 @@ Arrays keep their storage after the decoder closes. Paths, URLs, encoded bytes,
|
|
|
78
79
|
|
|
79
80
|
## Features
|
|
80
81
|
|
|
81
|
-
TensorCodec 0.1.
|
|
82
|
+
TensorCodec 0.1.3 relative to TorchCodec 0.17.0.
|
|
82
83
|
✓ supported · △ partial support · — not implemented.
|
|
83
84
|
|
|
84
85
|
| Component | TensorCodec | TorchCodec 0.17.0 |
|
|
@@ -105,6 +106,7 @@ FPS-based frame queries are supported; clip samplers are a separate API.
|
|
|
105
106
|
| NCHW / NHWC RGB output | ✓ | ✓ |
|
|
106
107
|
| uint8 / float32 / automatic dtype | ✓ SDR and high-bit-depth video | ✓ |
|
|
107
108
|
| uint16 RGB output | ✓ Full-range RGB48 | — |
|
|
109
|
+
| Native grayscale/depth and packed RGB(A) | ✓ Values preserved | — |
|
|
108
110
|
| PQ / HLG decoding | ✓ Transfer-encoded RGB | ✓ |
|
|
109
111
|
| Right-angle display rotation | ✓ | ✓ |
|
|
110
112
|
| Audio ranges / resampling / channel mixing | ✓ float32 | ✓ |
|
|
@@ -118,6 +120,11 @@ float32 above 8 bits, or `output_dtype="uint16"` for full-range 16-bit RGB.
|
|
|
118
120
|
HDR output retains PQ/HLG encoding without SDR tone mapping. Rotation is applied
|
|
119
121
|
automatically, and metadata dimensions match the output.
|
|
120
122
|
|
|
123
|
+
For unmodified samples, use `VideoDecoder(path, output_format="native")`.
|
|
124
|
+
Supported formats: `gray`, `gray12le`, `gray16le/be`, `rgb24`, `rgba`.
|
|
125
|
+
Native output preserves channel count, integer values and pixel coordinates;
|
|
126
|
+
`expected_pixel_format` optionally asserts the source format.
|
|
127
|
+
|
|
121
128
|
## Package size
|
|
122
129
|
|
|
123
130
|
<!-- wheel-size:start -->
|
|
@@ -153,8 +160,8 @@ See the [compatibility contract](docs/compatibility.md) and
|
|
|
153
160
|
|
|
154
161
|
- **Wheels:** Linux x86_64 and ARM64 (aarch64), glibc 2.17+, CPython 3.10+.
|
|
155
162
|
NumPy must also provide a compatible wheel; newer Python versions may require
|
|
156
|
-
a newer glibc. macOS
|
|
157
|
-
|
|
163
|
+
a newer glibc. macOS 14+ wheels support ARM64 and x86_64. Windows, musl/Alpine
|
|
164
|
+
and free-threaded Python wheels are not provided.
|
|
158
165
|
- **Exact seeking:** scans packet timestamps when opening the decoder. Incorrect
|
|
159
166
|
container keyframe flags can produce corrupt frames; repaired input or corrected
|
|
160
167
|
frame mappings are needed in that case.
|
|
@@ -174,6 +181,8 @@ Source builds require Rust, Clang/libclang, pkg-config and FFmpeg 7 development
|
|
|
174
181
|
headers/libraries. Python handles API and playback selection; Rust + PyO3 handles
|
|
175
182
|
FFmpeg. Native decoding releases the GIL, allowing separate decoder instances to
|
|
176
183
|
run concurrently across Python threads. Calls on the same instance are serialized.
|
|
184
|
+
The default is one FFmpeg thread per decoder; use independent workers for concurrent
|
|
185
|
+
windows and tune the total thread count to avoid oversubscription.
|
|
177
186
|
|
|
178
187
|
```sh
|
|
179
188
|
uv sync --group dev --group oracle
|
|
@@ -24,7 +24,7 @@ CPU video/audio decoding with TorchCodec-style APIs and NumPy output.
|
|
|
24
24
|
ranges are checked against TorchCodec 0.17.0 and independently generated media.
|
|
25
25
|
- **Efficient batch decoding.** Rust/PyO3 bindings to FFmpeg process frame batches
|
|
26
26
|
in a single native call, avoiding per-frame Python calls.
|
|
27
|
-
- **Lightweight installation.** Linux wheels are 10.2–10.4 MiB (v0.1.
|
|
27
|
+
- **Lightweight installation.** Linux wheels are 10.2–10.4 MiB (v0.1.2), including
|
|
28
28
|
FFmpeg shared libraries. NumPy is the only Python dependency.
|
|
29
29
|
|
|
30
30
|
## Quick start
|
|
@@ -54,7 +54,7 @@ Arrays keep their storage after the decoder closes. Paths, URLs, encoded bytes,
|
|
|
54
54
|
|
|
55
55
|
## Features
|
|
56
56
|
|
|
57
|
-
TensorCodec 0.1.
|
|
57
|
+
TensorCodec 0.1.3 relative to TorchCodec 0.17.0.
|
|
58
58
|
✓ supported · △ partial support · — not implemented.
|
|
59
59
|
|
|
60
60
|
| Component | TensorCodec | TorchCodec 0.17.0 |
|
|
@@ -81,6 +81,7 @@ FPS-based frame queries are supported; clip samplers are a separate API.
|
|
|
81
81
|
| NCHW / NHWC RGB output | ✓ | ✓ |
|
|
82
82
|
| uint8 / float32 / automatic dtype | ✓ SDR and high-bit-depth video | ✓ |
|
|
83
83
|
| uint16 RGB output | ✓ Full-range RGB48 | — |
|
|
84
|
+
| Native grayscale/depth and packed RGB(A) | ✓ Values preserved | — |
|
|
84
85
|
| PQ / HLG decoding | ✓ Transfer-encoded RGB | ✓ |
|
|
85
86
|
| Right-angle display rotation | ✓ | ✓ |
|
|
86
87
|
| Audio ranges / resampling / channel mixing | ✓ float32 | ✓ |
|
|
@@ -94,6 +95,11 @@ float32 above 8 bits, or `output_dtype="uint16"` for full-range 16-bit RGB.
|
|
|
94
95
|
HDR output retains PQ/HLG encoding without SDR tone mapping. Rotation is applied
|
|
95
96
|
automatically, and metadata dimensions match the output.
|
|
96
97
|
|
|
98
|
+
For unmodified samples, use `VideoDecoder(path, output_format="native")`.
|
|
99
|
+
Supported formats: `gray`, `gray12le`, `gray16le/be`, `rgb24`, `rgba`.
|
|
100
|
+
Native output preserves channel count, integer values and pixel coordinates;
|
|
101
|
+
`expected_pixel_format` optionally asserts the source format.
|
|
102
|
+
|
|
97
103
|
## Package size
|
|
98
104
|
|
|
99
105
|
<!-- wheel-size:start -->
|
|
@@ -129,8 +135,8 @@ See the [compatibility contract](docs/compatibility.md) and
|
|
|
129
135
|
|
|
130
136
|
- **Wheels:** Linux x86_64 and ARM64 (aarch64), glibc 2.17+, CPython 3.10+.
|
|
131
137
|
NumPy must also provide a compatible wheel; newer Python versions may require
|
|
132
|
-
a newer glibc. macOS
|
|
133
|
-
|
|
138
|
+
a newer glibc. macOS 14+ wheels support ARM64 and x86_64. Windows, musl/Alpine
|
|
139
|
+
and free-threaded Python wheels are not provided.
|
|
134
140
|
- **Exact seeking:** scans packet timestamps when opening the decoder. Incorrect
|
|
135
141
|
container keyframe flags can produce corrupt frames; repaired input or corrected
|
|
136
142
|
frame mappings are needed in that case.
|
|
@@ -150,6 +156,8 @@ Source builds require Rust, Clang/libclang, pkg-config and FFmpeg 7 development
|
|
|
150
156
|
headers/libraries. Python handles API and playback selection; Rust + PyO3 handles
|
|
151
157
|
FFmpeg. Native decoding releases the GIL, allowing separate decoder instances to
|
|
152
158
|
run concurrently across Python threads. Calls on the same instance are serialized.
|
|
159
|
+
The default is one FFmpeg thread per decoder; use independent workers for concurrent
|
|
160
|
+
windows and tune the total thread count to avoid oversubscription.
|
|
153
161
|
|
|
154
162
|
```sh
|
|
155
163
|
uv sync --group dev --group oracle
|
|
@@ -66,3 +66,18 @@ TensorCodec extensions. Right-angle display rotations are applied automatically;
|
|
|
66
66
|
metadata dimensions describe the rotated output. Reflected and non-right-angle
|
|
67
67
|
display matrices remain unsupported. Color metadata and pixel aspect ratio
|
|
68
68
|
describe the source; HDR output is not linear light or sRGB.
|
|
69
|
+
|
|
70
|
+
## Native video output
|
|
71
|
+
|
|
72
|
+
`output_format="native"` bypasses color conversion and display transforms for
|
|
73
|
+
`gray`, `gray12le`, `gray16le`, `gray16be`, `rgb24` and `rgba`. NCHW/NHWC keeps
|
|
74
|
+
1, 3 or 4 channels. Output uses host-endian uint8/uint16 without range scaling;
|
|
75
|
+
`output_dtype` may be omitted, `"auto"`, or the matching integer dtype.
|
|
76
|
+
`expected_pixel_format` asserts the source layout. Unsupported formats, dtype
|
|
77
|
+
conversions and changes of pixel format or dimensions fail explicitly.
|
|
78
|
+
|
|
79
|
+
Frame/FrameBatch `pixel_format` records the native source format and survives
|
|
80
|
+
indexing and FPS resampling. RGB results retain their existing behavior. Native
|
|
81
|
+
mode preserves encoded pixel coordinates, including inputs with display matrices.
|
|
82
|
+
Playback selection follows the same TorchCodec contract as RGB, including the
|
|
83
|
+
frame overlapping a range's start; it does not copy PyAV's legacy PTS-only range rule.
|
|
@@ -51,7 +51,7 @@ download; candidate builds never update it.
|
|
|
51
51
|
To reproduce or recover a documentation update after publication:
|
|
52
52
|
|
|
53
53
|
```sh
|
|
54
|
-
uv run --no-project python scripts/update_size_comparison.py --version 0.1.
|
|
54
|
+
uv run --no-project python scripts/update_size_comparison.py --version 0.1.3
|
|
55
55
|
```
|
|
56
56
|
|
|
57
57
|
Review and commit `README.md` and `packaging/size-baseline.json` together. The script
|
|
@@ -1,10 +1,11 @@
|
|
|
1
1
|
# Publishing TensorCodec
|
|
2
2
|
|
|
3
|
-
Release version: `0.1.
|
|
3
|
+
Release version: `0.1.3`. Distribution and import name: `tensorcodec`.
|
|
4
4
|
Binary wheels target Linux x86_64 and ARM64 (aarch64), glibc 2.17+, CPython 3.10+.
|
|
5
5
|
NumPy must also provide a compatible wheel for the selected Python/glibc pair.
|
|
6
6
|
The wheel bundles shared FFmpeg 7.1.5 and OpenSSL 3.5.9 LTS; its only Python
|
|
7
|
-
runtime dependency is NumPy. macOS/
|
|
7
|
+
runtime dependency is NumPy. macOS 14+ ARM64/x86_64 wheels bundle the same minimal
|
|
8
|
+
runtime. Windows wheels are not provided.
|
|
8
9
|
|
|
9
10
|
## Trusted publisher configuration
|
|
10
11
|
|
|
@@ -24,7 +25,7 @@ is needed. Repository visibility does not need to change for a release.
|
|
|
24
25
|
|
|
25
26
|
## Release
|
|
26
27
|
|
|
27
|
-
Run the **Publish to PyPI** workflow on `main`. It builds the portable Linux wheels
|
|
28
|
+
Run the **Publish to PyPI** workflow on `main`. It builds the portable Linux/macOS wheels
|
|
28
29
|
and source distribution, checks package metadata, validates the pinned oracle
|
|
29
30
|
and compares playback before uploading through PyPI Trusted Publishing. It uses
|
|
30
31
|
the existing GitHub `pypi` environment. Publication fails if authorization is
|
|
@@ -36,8 +37,8 @@ gh workflow run publish.yml --repo MilkClouds/tensorcodec --ref main
|
|
|
36
37
|
|
|
37
38
|
For a build and full validation without uploading, pass `--field publish=false`.
|
|
38
39
|
|
|
39
|
-
Check the workflow and https://pypi.org/project/tensorcodec/0.1.
|
|
40
|
-
success. Verify a fresh `uv pip install tensorcodec==0.1.
|
|
40
|
+
Check the workflow and https://pypi.org/project/tensorcodec/0.1.3/ before reporting
|
|
41
|
+
success. Verify a fresh `uv pip install tensorcodec==0.1.3` and a decode without
|
|
41
42
|
Torch/PyAV on both architectures. Update the version before subsequent releases;
|
|
42
43
|
PyPI versions cannot be overwritten.
|
|
43
44
|
|
|
@@ -37,7 +37,7 @@ test -f "$FFMPEG_DIR/include/libavcodec/avcodec.h"
|
|
|
37
37
|
test -f "$FFMPEG_DIR/lib/libavcodec.so.61"
|
|
38
38
|
|
|
39
39
|
uv venv
|
|
40
|
-
uv pip install --no-binary tensorcodec 'tensorcodec==0.1.
|
|
40
|
+
uv pip install --no-binary tensorcodec 'tensorcodec==0.1.3'
|
|
41
41
|
```
|
|
42
42
|
|
|
43
43
|
The version/build constraint avoids silently selecting an incompatible FFmpeg
|
|
@@ -75,3 +75,11 @@ The native extension build and playback contracts run against prebuilt FFmpeg
|
|
|
75
75
|
no external FFmpeg library installation. Configuration-specific outputs can differ
|
|
76
76
|
within the color-conversion tolerances described in the
|
|
77
77
|
[compatibility contract](compatibility.md).
|
|
78
|
+
|
|
79
|
+
## macOS wheels
|
|
80
|
+
|
|
81
|
+
Since 0.1.3, macOS 14+ wheels support Apple Silicon and Intel, bundling
|
|
82
|
+
FFmpeg/OpenSSL with `delocate`. CI tests
|
|
83
|
+
the installed wheels and clean Python 3.10/3.13 environments. Developers can run
|
|
84
|
+
`scripts/build_macos_wheel.sh` with Rust, Xcode tools, NASM, pkg-config, coreutils,
|
|
85
|
+
maturin and delocate installed in their build environment.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# Bundled native libraries
|
|
2
2
|
|
|
3
|
-
TensorCodec's own code is MIT licensed. Linux wheels bundle shared FFmpeg
|
|
3
|
+
TensorCodec's own code is MIT licensed. Linux and macOS wheels bundle shared FFmpeg
|
|
4
4
|
7.1.5 libraries, built without GPL codec libraries using
|
|
5
5
|
`scripts/build_ffmpeg.sh`. This configuration is LGPL-3.0-or-later. Its notices
|
|
6
6
|
and both the LGPLv3 and incorporated GPLv3 texts are included here. The exact
|
|
@@ -8,9 +8,11 @@ upstream source is https://ffmpeg.org/releases/ffmpeg-7.1.5.tar.xz; the build
|
|
|
8
8
|
script records the configuration. FFmpeg libraries remain dynamically linked
|
|
9
9
|
and can be rebuilt/replaced with an ABI-compatible build.
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
Release wheels also bundle shared OpenSSL 3.5.9 LTS, licensed
|
|
12
12
|
under Apache-2.0. Its exact source is
|
|
13
13
|
https://github.com/openssl/openssl/releases/download/openssl-3.5.9/openssl-3.5.9.tar.gz;
|
|
14
14
|
`scripts/build_openssl.sh` records the checksum and build configuration.
|
|
15
15
|
System-library builds can additionally depend on Zstandard; its notices are
|
|
16
16
|
retained here. Inspect repaired wheels when changing the native build.
|
|
17
|
+
|
|
18
|
+
PNG/Deflate decoding uses the platform zlib library.
|
|
@@ -455,9 +455,10 @@ checksum = "61c41af27dd6d1e27b1b16b489db798443478cef1f06a660c96db617ba5de3b1"
|
|
|
455
455
|
|
|
456
456
|
[[package]]
|
|
457
457
|
name = "tensorcodec-native"
|
|
458
|
-
version = "0.1.
|
|
458
|
+
version = "0.1.3"
|
|
459
459
|
dependencies = [
|
|
460
460
|
"ffmpeg-sys-next",
|
|
461
|
+
"libc",
|
|
461
462
|
"numpy",
|
|
462
463
|
"pyo3",
|
|
463
464
|
]
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[package]
|
|
2
2
|
name = "tensorcodec-native"
|
|
3
|
-
version = "0.1.
|
|
3
|
+
version = "0.1.3"
|
|
4
4
|
edition = "2021"
|
|
5
5
|
license = "MIT"
|
|
6
6
|
|
|
@@ -9,6 +9,7 @@ name = "_native"
|
|
|
9
9
|
crate-type = ["cdylib"]
|
|
10
10
|
|
|
11
11
|
[dependencies]
|
|
12
|
+
libc = "0.2"
|
|
12
13
|
pyo3 = { version = "0.23", features = ["abi3-py310"] }
|
|
13
14
|
numpy = "0.23"
|
|
14
15
|
ffmpeg-sys-next = { version = "7.1", default-features = false, features = ["avcodec", "avformat", "swscale", "swresample"] }
|
|
@@ -64,7 +64,7 @@ enum Reader {
|
|
|
64
64
|
}
|
|
65
65
|
unsafe extern "C" fn read_memory(opaque: *mut c_void, buffer: *mut u8, size: i32) -> i32 {
|
|
66
66
|
if size <= 0 {
|
|
67
|
-
return
|
|
67
|
+
return av::AVERROR(libc::EINVAL);
|
|
68
68
|
}
|
|
69
69
|
let reader = &mut *(opaque as *mut Reader);
|
|
70
70
|
let input = match reader {
|
|
@@ -121,13 +121,13 @@ unsafe extern "C" fn seek_memory(opaque: *mut c_void, offset: i64, whence: i32)
|
|
|
121
121
|
0 => 0,
|
|
122
122
|
1 => input.position as i64,
|
|
123
123
|
2 => input.data.len() as i64,
|
|
124
|
-
_ => return
|
|
124
|
+
_ => return i64::from(av::AVERROR(libc::EINVAL)),
|
|
125
125
|
};
|
|
126
126
|
let Some(position) = base.checked_add(offset) else {
|
|
127
|
-
return
|
|
127
|
+
return i64::from(av::AVERROR(libc::EINVAL));
|
|
128
128
|
};
|
|
129
129
|
if position < 0 || position > input.data.len() as i64 {
|
|
130
|
-
return
|
|
130
|
+
return i64::from(av::AVERROR(libc::EINVAL));
|
|
131
131
|
}
|
|
132
132
|
input.position = position as usize;
|
|
133
133
|
position
|
|
@@ -146,6 +146,7 @@ pub struct Decoder {
|
|
|
146
146
|
index: i32,
|
|
147
147
|
time_base: av::AVRational,
|
|
148
148
|
audio: bool,
|
|
149
|
+
video_layout: (i32, i32, av::AVPixelFormat),
|
|
149
150
|
draining: bool,
|
|
150
151
|
}
|
|
151
152
|
// SAFETY: all pointers are uniquely owned. PyO3's mutable borrow plus the Python
|
|
@@ -194,6 +195,7 @@ impl Decoder {
|
|
|
194
195
|
index: 0,
|
|
195
196
|
time_base: av::AVRational { num: 0, den: 1 },
|
|
196
197
|
audio,
|
|
198
|
+
video_layout: (0, 0, av::AVPixelFormat::AV_PIX_FMT_NONE),
|
|
197
199
|
draining: false,
|
|
198
200
|
};
|
|
199
201
|
unsafe {
|
|
@@ -276,6 +278,13 @@ impl Decoder {
|
|
|
276
278
|
if (*params).codec_type != media_type {
|
|
277
279
|
return Err(Error("stream has the wrong media type".into(), true));
|
|
278
280
|
}
|
|
281
|
+
if !audio {
|
|
282
|
+
this.video_layout = (
|
|
283
|
+
(*params).width,
|
|
284
|
+
(*params).height,
|
|
285
|
+
std::mem::transmute::<i32, av::AVPixelFormat>((*params).format),
|
|
286
|
+
);
|
|
287
|
+
}
|
|
279
288
|
this.time_base = (*stream).time_base;
|
|
280
289
|
if this.time_base.den <= 0 || this.time_base.num <= 0 {
|
|
281
290
|
return Err(failure("invalid stream time base"));
|
|
@@ -336,7 +345,11 @@ impl Decoder {
|
|
|
336
345
|
Ok(())
|
|
337
346
|
}
|
|
338
347
|
|
|
339
|
-
pub fn metadata<'py>(
|
|
348
|
+
pub fn metadata<'py>(
|
|
349
|
+
&self,
|
|
350
|
+
py: Python<'py>,
|
|
351
|
+
apply_rotation: bool,
|
|
352
|
+
) -> PyResult<Bound<'py, PyDict>> {
|
|
340
353
|
let data = PyDict::new(py);
|
|
341
354
|
unsafe {
|
|
342
355
|
let stream = &*self.stream();
|
|
@@ -448,7 +461,8 @@ impl Decoder {
|
|
|
448
461
|
params.nb_coded_side_data,
|
|
449
462
|
av::AVPacketSideDataType::AV_PKT_DATA_DISPLAYMATRIX,
|
|
450
463
|
);
|
|
451
|
-
let rotation = if !side_data.is_null() && (*side_data).size >= 36
|
|
464
|
+
let rotation = if apply_rotation && !side_data.is_null() && (*side_data).size >= 36
|
|
465
|
+
{
|
|
452
466
|
let matrix = (*side_data).data as *const i32;
|
|
453
467
|
let determinant = *matrix as f64 * *matrix.add(4) as f64
|
|
454
468
|
- *matrix.add(1) as f64 * *matrix.add(3) as f64;
|
|
@@ -513,7 +527,7 @@ impl Decoder {
|
|
|
513
527
|
if code == av::AVERROR_EOF {
|
|
514
528
|
return Ok(false);
|
|
515
529
|
}
|
|
516
|
-
if code !=
|
|
530
|
+
if code != av::AVERROR(libc::EAGAIN) {
|
|
517
531
|
check(code, "receive decoded frame")?;
|
|
518
532
|
}
|
|
519
533
|
if self.draining {
|
|
@@ -567,12 +581,31 @@ impl Decoder {
|
|
|
567
581
|
if self.audio {
|
|
568
582
|
return Err(failure("cannot decode video from audio stream"));
|
|
569
583
|
}
|
|
584
|
+
let native = matches!(dtype, OutputDtype::Native);
|
|
585
|
+
let (width, height, source_format) = self.video_layout;
|
|
586
|
+
let (channels, dtype, big_endian) = if native {
|
|
587
|
+
use av::AVPixelFormat::*;
|
|
588
|
+
match source_format {
|
|
589
|
+
AV_PIX_FMT_GRAY8 => (1, OutputDtype::U8, false),
|
|
590
|
+
AV_PIX_FMT_GRAY12LE | AV_PIX_FMT_GRAY16LE => (1, OutputDtype::U16, false),
|
|
591
|
+
AV_PIX_FMT_GRAY16BE => (1, OutputDtype::U16, true),
|
|
592
|
+
AV_PIX_FMT_RGB24 => (3, OutputDtype::U8, false),
|
|
593
|
+
AV_PIX_FMT_RGBA => (4, OutputDtype::U8, false),
|
|
594
|
+
_ => {
|
|
595
|
+
return Err(Error(
|
|
596
|
+
"native output does not support this pixel format".into(),
|
|
597
|
+
true,
|
|
598
|
+
))
|
|
599
|
+
}
|
|
600
|
+
}
|
|
601
|
+
} else {
|
|
602
|
+
(3, dtype, false)
|
|
603
|
+
};
|
|
570
604
|
let high_depth = !matches!(dtype, OutputDtype::U8);
|
|
571
|
-
let width =
|
|
572
|
-
let height = unsafe { (*self.codec).height } as usize;
|
|
605
|
+
let (width, height) = (width as usize, height as usize);
|
|
573
606
|
let count = width
|
|
574
607
|
.checked_mul(height)
|
|
575
|
-
.and_then(|n| n.checked_mul(
|
|
608
|
+
.and_then(|n| n.checked_mul(channels))
|
|
576
609
|
.ok_or_else(|| failure("frame is too large"))?;
|
|
577
610
|
let total = count
|
|
578
611
|
.checked_mul(targets.len())
|
|
@@ -627,106 +660,119 @@ impl Decoder {
|
|
|
627
660
|
return Err(failure("dynamic frame dimensions are unsupported"));
|
|
628
661
|
}
|
|
629
662
|
let input_format: av::AVPixelFormat = std::mem::transmute(frame.format);
|
|
630
|
-
let
|
|
631
|
-
|
|
663
|
+
let output_frame = if native {
|
|
664
|
+
if input_format != source_format {
|
|
665
|
+
return Err(Error("pixel format changed within stream".into(), true));
|
|
666
|
+
}
|
|
667
|
+
frame
|
|
632
668
|
} else {
|
|
633
|
-
|
|
634
|
-
|
|
635
|
-
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
output_format as i32,
|
|
640
|
-
);
|
|
641
|
-
if self.scale_config != Some(config) {
|
|
642
|
-
av::sws_freeContext(self.scale);
|
|
643
|
-
self.scale = av::sws_getContext(
|
|
669
|
+
let output_format = if high_depth {
|
|
670
|
+
av::AVPixelFormat::AV_PIX_FMT_RGB48LE
|
|
671
|
+
} else {
|
|
672
|
+
av::AVPixelFormat::AV_PIX_FMT_RGB24
|
|
673
|
+
};
|
|
674
|
+
let config = (
|
|
644
675
|
frame.width,
|
|
645
676
|
frame.height,
|
|
646
|
-
input_format,
|
|
647
|
-
|
|
648
|
-
frame.height,
|
|
649
|
-
output_format,
|
|
650
|
-
0,
|
|
651
|
-
ptr::null_mut(),
|
|
652
|
-
ptr::null_mut(),
|
|
653
|
-
ptr::null(),
|
|
677
|
+
input_format as i32,
|
|
678
|
+
output_format as i32,
|
|
654
679
|
);
|
|
655
|
-
self.scale_config
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
677
|
-
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
source_range =
|
|
684
|
-
i32::from(frame.color_range == av::AVColorRange::AVCOL_RANGE_JPEG);
|
|
685
|
-
}
|
|
686
|
-
let coefficients = av::sws_getCoefficients(frame.colorspace as i32);
|
|
687
|
-
check(
|
|
688
|
-
av::sws_setColorspaceDetails(
|
|
689
|
-
self.scale,
|
|
690
|
-
coefficients,
|
|
691
|
-
source_range,
|
|
692
|
-
coefficients,
|
|
693
|
-
destination_range,
|
|
694
|
-
brightness,
|
|
695
|
-
contrast,
|
|
696
|
-
saturation,
|
|
697
|
-
),
|
|
698
|
-
"configure color conversion",
|
|
699
|
-
)?;
|
|
700
|
-
if (*self.rgb_frame).width != frame.width
|
|
701
|
-
|| (*self.rgb_frame).height != frame.height
|
|
702
|
-
|| (*self.rgb_frame).format != output_format as i32
|
|
703
|
-
{
|
|
704
|
-
av::av_frame_unref(self.rgb_frame);
|
|
705
|
-
(*self.rgb_frame).width = frame.width;
|
|
706
|
-
(*self.rgb_frame).height = frame.height;
|
|
707
|
-
(*self.rgb_frame).format = output_format as i32;
|
|
680
|
+
if self.scale_config != Some(config) {
|
|
681
|
+
av::sws_freeContext(self.scale);
|
|
682
|
+
self.scale = av::sws_getContext(
|
|
683
|
+
frame.width,
|
|
684
|
+
frame.height,
|
|
685
|
+
input_format,
|
|
686
|
+
frame.width,
|
|
687
|
+
frame.height,
|
|
688
|
+
output_format,
|
|
689
|
+
0,
|
|
690
|
+
ptr::null_mut(),
|
|
691
|
+
ptr::null_mut(),
|
|
692
|
+
ptr::null(),
|
|
693
|
+
);
|
|
694
|
+
self.scale_config = Some(config);
|
|
695
|
+
}
|
|
696
|
+
if self.scale.is_null() {
|
|
697
|
+
return Err(failure("cannot initialize color conversion"));
|
|
698
|
+
}
|
|
699
|
+
let mut inverse = ptr::null_mut();
|
|
700
|
+
let mut table = ptr::null_mut();
|
|
701
|
+
let (
|
|
702
|
+
mut source_range,
|
|
703
|
+
mut destination_range,
|
|
704
|
+
mut brightness,
|
|
705
|
+
mut contrast,
|
|
706
|
+
mut saturation,
|
|
707
|
+
) = (0, 0, 0, 0, 0);
|
|
708
708
|
check(
|
|
709
|
-
av::
|
|
710
|
-
|
|
709
|
+
av::sws_getColorspaceDetails(
|
|
710
|
+
self.scale,
|
|
711
|
+
&mut inverse,
|
|
712
|
+
&mut source_range,
|
|
713
|
+
&mut table,
|
|
714
|
+
&mut destination_range,
|
|
715
|
+
&mut brightness,
|
|
716
|
+
&mut contrast,
|
|
717
|
+
&mut saturation,
|
|
718
|
+
),
|
|
719
|
+
"read color conversion settings",
|
|
711
720
|
)?;
|
|
721
|
+
if frame.color_range != av::AVColorRange::AVCOL_RANGE_UNSPECIFIED {
|
|
722
|
+
source_range =
|
|
723
|
+
i32::from(frame.color_range == av::AVColorRange::AVCOL_RANGE_JPEG);
|
|
724
|
+
}
|
|
725
|
+
let coefficients = av::sws_getCoefficients(frame.colorspace as i32);
|
|
726
|
+
check(
|
|
727
|
+
av::sws_setColorspaceDetails(
|
|
728
|
+
self.scale,
|
|
729
|
+
coefficients,
|
|
730
|
+
source_range,
|
|
731
|
+
coefficients,
|
|
732
|
+
destination_range,
|
|
733
|
+
brightness,
|
|
734
|
+
contrast,
|
|
735
|
+
saturation,
|
|
736
|
+
),
|
|
737
|
+
"configure color conversion",
|
|
738
|
+
)?;
|
|
739
|
+
if (*self.rgb_frame).width != frame.width
|
|
740
|
+
|| (*self.rgb_frame).height != frame.height
|
|
741
|
+
|| (*self.rgb_frame).format != output_format as i32
|
|
742
|
+
{
|
|
743
|
+
av::av_frame_unref(self.rgb_frame);
|
|
744
|
+
(*self.rgb_frame).width = frame.width;
|
|
745
|
+
(*self.rgb_frame).height = frame.height;
|
|
746
|
+
(*self.rgb_frame).format = output_format as i32;
|
|
747
|
+
check(
|
|
748
|
+
av::av_frame_get_buffer(self.rgb_frame, 32),
|
|
749
|
+
"allocate RGB frame",
|
|
750
|
+
)?;
|
|
751
|
+
}
|
|
752
|
+
let rows = av::sws_scale(
|
|
753
|
+
self.scale,
|
|
754
|
+
frame.data.as_ptr() as *const *const u8,
|
|
755
|
+
frame.linesize.as_ptr(),
|
|
756
|
+
0,
|
|
757
|
+
frame.height,
|
|
758
|
+
(*self.rgb_frame).data.as_ptr(),
|
|
759
|
+
(*self.rgb_frame).linesize.as_ptr(),
|
|
760
|
+
);
|
|
761
|
+
if rows != frame.height {
|
|
762
|
+
return Err(failure("color conversion failed"));
|
|
763
|
+
}
|
|
764
|
+
&*self.rgb_frame
|
|
765
|
+
};
|
|
766
|
+
let row_bytes = width * channels * if high_depth { 2 } else { 1 };
|
|
767
|
+
if output_frame.data[0].is_null()
|
|
768
|
+
|| (output_frame.linesize[0].unsigned_abs() as usize) < row_bytes
|
|
769
|
+
{
|
|
770
|
+
return Err(failure("invalid decoded frame stride"));
|
|
712
771
|
}
|
|
713
|
-
let rows = av::sws_scale(
|
|
714
|
-
self.scale,
|
|
715
|
-
frame.data.as_ptr() as *const *const u8,
|
|
716
|
-
frame.linesize.as_ptr(),
|
|
717
|
-
0,
|
|
718
|
-
frame.height,
|
|
719
|
-
(*self.rgb_frame).data.as_ptr(),
|
|
720
|
-
(*self.rgb_frame).linesize.as_ptr(),
|
|
721
|
-
);
|
|
722
|
-
if rows != frame.height {
|
|
723
|
-
return Err(failure("color conversion failed"));
|
|
724
|
-
}
|
|
725
|
-
let row_bytes = width * 3 * if high_depth { 2 } else { 1 };
|
|
726
772
|
for row in 0..height {
|
|
727
773
|
ptr::copy_nonoverlapping(
|
|
728
|
-
|
|
729
|
-
.
|
|
774
|
+
output_frame.data[0]
|
|
775
|
+
.offset(row as isize * output_frame.linesize[0] as isize),
|
|
730
776
|
pixels.as_mut_ptr().add(first * stride + row * row_bytes),
|
|
731
777
|
row_bytes,
|
|
732
778
|
);
|
|
@@ -760,10 +806,17 @@ impl Decoder {
|
|
|
760
806
|
.as_chunks::<2>()
|
|
761
807
|
.0
|
|
762
808
|
.iter()
|
|
763
|
-
.map(|p|
|
|
809
|
+
.map(|p| {
|
|
810
|
+
if big_endian {
|
|
811
|
+
u16::from_be_bytes(*p)
|
|
812
|
+
} else {
|
|
813
|
+
u16::from_le_bytes(*p)
|
|
814
|
+
}
|
|
815
|
+
})
|
|
764
816
|
.collect(),
|
|
765
817
|
),
|
|
766
818
|
OutputDtype::U8 => Pixels::U8(pixels),
|
|
819
|
+
OutputDtype::Native => unreachable!(),
|
|
767
820
|
};
|
|
768
821
|
Ok(Video {
|
|
769
822
|
pixels,
|
|
@@ -771,6 +824,7 @@ impl Decoder {
|
|
|
771
824
|
durations,
|
|
772
825
|
width,
|
|
773
826
|
height,
|
|
827
|
+
channels,
|
|
774
828
|
})
|
|
775
829
|
}
|
|
776
830
|
|
|
@@ -881,6 +935,7 @@ pub enum Pixels {
|
|
|
881
935
|
F32(Vec<f32>),
|
|
882
936
|
}
|
|
883
937
|
pub enum OutputDtype {
|
|
938
|
+
Native,
|
|
884
939
|
U8,
|
|
885
940
|
U16,
|
|
886
941
|
F32,
|
|
@@ -891,6 +946,7 @@ pub struct Video {
|
|
|
891
946
|
pub durations: Vec<f64>,
|
|
892
947
|
pub width: usize,
|
|
893
948
|
pub height: usize,
|
|
949
|
+
pub channels: usize,
|
|
894
950
|
}
|
|
895
951
|
pub struct Audio {
|
|
896
952
|
pub data: Vec<f32>,
|