videopython 0.54.0__tar.gz → 0.55.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {videopython-0.54.0 → videopython-0.55.0}/PKG-INFO +5 -3
- {videopython-0.54.0 → videopython-0.55.0}/pyproject.toml +42 -12
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/_ollama.py +25 -1
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/understanding/faces.py +1 -1
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/audio/audio.py +5 -9
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/video.py +3 -2
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/effects.py +133 -32
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/operation.py +14 -3
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/streaming.py +5 -1
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/video_edit.py +19 -7
- {videopython-0.54.0 → videopython-0.55.0}/.gitignore +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/LICENSE +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/README.md +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/_device.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/_optional.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/_predictor.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/_revisions.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/auto_edit/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/auto_edit/backend.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/auto_edit/catalog.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/auto_edit/editor.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/auto_edit/local.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/auto_edit/models.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/auto_edit/resolve.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/_tts_backend.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/audio_ops.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/config.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/dubber.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/models.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/pipeline.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/quality.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/remux.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/separation.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/timing.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/translation.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/dubbing/voice_sample.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/effects.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/errors.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/generation/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/generation/audio.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/generation/image.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/generation/video.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/keyframe.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/ops.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/transforms.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/understanding/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/understanding/_detector.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/understanding/audio.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/understanding/classification.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/understanding/image.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/understanding/objects.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/understanding/temporal.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/video_analysis/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/video_analysis/analyzer.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/video_analysis/detectors.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/video_analysis/models.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/video_analysis/sampling.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/video_analysis/source_metadata.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/audio/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/audio/analysis.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/_dimensions.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/_ffmpeg.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/_video_io.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/description.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/draw_detections.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/exceptions.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/Anton-OFL.txt +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/Anton-Regular.ttf +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/BebasNeue-OFL.txt +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/BebasNeue-Regular.ttf +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/DejaVuSans.ttf +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/LICENSE_DEJAVU +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/Lato-Bold.ttf +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/Lato-OFL.txt +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/Poppins-Bold.ttf +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/Poppins-OFL.txt +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/fonts/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/base/transcription.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/_ass.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/_easing.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/_schema.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/audio_ops.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/transcription_overlay.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/editing/transforms.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/mcp/__init__.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/mcp/server.py +0 -0
- {videopython-0.54.0 → videopython-0.55.0}/src/videopython/py.typed +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: videopython
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.55.0
|
|
4
4
|
Summary: Minimal video generation and processing library.
|
|
5
5
|
Project-URL: Homepage, https://videopython.com
|
|
6
6
|
Project-URL: Repository, https://github.com/bartwojtowicz/videopython/
|
|
@@ -24,11 +24,11 @@ Requires-Dist: resvg-py>=0.3.2
|
|
|
24
24
|
Requires-Dist: tqdm>=4.66.3
|
|
25
25
|
Provides-Extra: ai
|
|
26
26
|
Requires-Dist: accelerate>=0.29.2; extra == 'ai'
|
|
27
|
-
Requires-Dist: chatterbox-tts>=0.1.7; extra == 'ai'
|
|
28
27
|
Requires-Dist: demucs>=4.0.0; extra == 'ai'
|
|
29
28
|
Requires-Dist: diffusers>=0.35.0; extra == 'ai'
|
|
30
29
|
Requires-Dist: ftfy>=6.1; extra == 'ai'
|
|
31
30
|
Requires-Dist: imagehash>=4.3; extra == 'ai'
|
|
31
|
+
Requires-Dist: numba>=0.62; extra == 'ai'
|
|
32
32
|
Requires-Dist: ollama>=0.5.0; extra == 'ai'
|
|
33
33
|
Requires-Dist: openai-whisper>=20240930; extra == 'ai'
|
|
34
34
|
Requires-Dist: pyannote-audio>=4.0.0; extra == 'ai'
|
|
@@ -36,9 +36,11 @@ Requires-Dist: pyloudnorm>=0.1.1; extra == 'ai'
|
|
|
36
36
|
Requires-Dist: silero-vad>=5.1; extra == 'ai'
|
|
37
37
|
Requires-Dist: torch>=2.8.0; extra == 'ai'
|
|
38
38
|
Requires-Dist: torchaudio>=2.8.0; extra == 'ai'
|
|
39
|
-
Requires-Dist:
|
|
39
|
+
Requires-Dist: torchcodec>=0.7.0; extra == 'ai'
|
|
40
|
+
Requires-Dist: torchvision>=0.23.0; extra == 'ai'
|
|
40
41
|
Requires-Dist: transformers>=5.2.0; extra == 'ai'
|
|
41
42
|
Requires-Dist: transnetv2-pytorch>=1.0.5; extra == 'ai'
|
|
43
|
+
Requires-Dist: videopython-chatterbox>=0.1.7.post1; extra == 'ai'
|
|
42
44
|
Provides-Extra: mcp
|
|
43
45
|
Requires-Dist: mcp<2,>=1.27; extra == 'mcp'
|
|
44
46
|
Description-Content-Type: text/markdown
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "videopython"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.55.0"
|
|
4
4
|
description = "Minimal video generation and processing library."
|
|
5
5
|
authors = [
|
|
6
6
|
{ name = "Bartosz Wójtowicz", email = "bartoszwojtowicz@outlook.com" },
|
|
@@ -66,6 +66,13 @@ ai = ["videopython[ai]"]
|
|
|
66
66
|
ai = [
|
|
67
67
|
# Speech recognition / diarization (understanding/audio.py)
|
|
68
68
|
"openai-whisper>=20240930",
|
|
69
|
+
# whisper imports numba (timing.py), and numba is version-coupled to numpy:
|
|
70
|
+
# each release declares a `numpy<X` cap that moves forward (0.62 -> <2.4,
|
|
71
|
+
# 0.66 -> <2.5). Without a floor a resolver can satisfy a newer numpy by
|
|
72
|
+
# backtracking numba to an ancient release instead, which then fails at
|
|
73
|
+
# `import whisper`. Floor it so numba's own cap does the work; 0.62 is the
|
|
74
|
+
# oldest release covering our full Python range (3.11-3.13).
|
|
75
|
+
"numba>=0.62",
|
|
69
76
|
"pyannote-audio>=4.0.0",
|
|
70
77
|
"silero-vad>=5.1",
|
|
71
78
|
# Visual understanding: detection (D-FINE/YuNet via transformers + opencv core
|
|
@@ -75,9 +82,14 @@ ai = [
|
|
|
75
82
|
"imagehash>=4.3",
|
|
76
83
|
# Audio source separation (understanding/separation.py)
|
|
77
84
|
"demucs>=4.0.0",
|
|
78
|
-
# Voice cloning TTS (generation/audio.py — Chatterbox)
|
|
79
|
-
#
|
|
80
|
-
|
|
85
|
+
# Voice cloning TTS (generation/audio.py — Chatterbox). Upstream chatterbox-tts
|
|
86
|
+
# 0.1.7 pins diffusers==0.29.0 / torch==2.6.0 / transformers==5.2.0 with `==`,
|
|
87
|
+
# which is unsatisfiable against pyannote-audio (torch>=2.8) and our
|
|
88
|
+
# diffusers>=0.35 — it made `pip install "videopython[ai]"` impossible for
|
|
89
|
+
# consumers. videopython-chatterbox is upstream's source with corrected
|
|
90
|
+
# metadata (identical apart from its self-version lookup); the import name is
|
|
91
|
+
# still `chatterbox`. See https://github.com/BartWojtowicz/videopython-chatterbox
|
|
92
|
+
"videopython-chatterbox>=0.1.7.post1",
|
|
81
93
|
# Local media generation: Qwen-Image/Wan2.2 + MusicGen (generation/*).
|
|
82
94
|
# >=0.35 is the floor that ships QwenImagePipeline + Wan2.2 support (tested on 0.37.1).
|
|
83
95
|
"diffusers>=0.35.0",
|
|
@@ -94,9 +106,17 @@ ai = [
|
|
|
94
106
|
# processor used by the D-FINE object detector (understanding/objects.py).
|
|
95
107
|
"torch>=2.8.0",
|
|
96
108
|
"torchaudio>=2.8.0",
|
|
97
|
-
#
|
|
98
|
-
#
|
|
99
|
-
|
|
109
|
+
# No upper bound: each torchvision release declares an exact `torch==X.Y.Z`
|
|
110
|
+
# pin, so resolvers co-select a valid pair unaided. Capping torchvision alone
|
|
111
|
+
# is actively harmful — it transitively pinned torch to 2.8 (cu12) while
|
|
112
|
+
# torchaudio floated to 2.11 (cu13), resolving cleanly and then dying at
|
|
113
|
+
# import with `OSError: libcudart.so.13`. Cap every member or none.
|
|
114
|
+
"torchvision>=0.23.0",
|
|
115
|
+
# torchaudio 2.11 routes save()/load() through torchcodec, and torchcodec is a
|
|
116
|
+
# native extension linked against libtorch that declares no torch requirement
|
|
117
|
+
# of its own. It arrives transitively via pyannote-audio; declare it so it is
|
|
118
|
+
# visible rather than incidental.
|
|
119
|
+
"torchcodec>=0.7.0",
|
|
100
120
|
]
|
|
101
121
|
# MCP server (videopython/mcp/). Pin <2 — v2 is pre-release/breaking.
|
|
102
122
|
mcp = ["mcp>=1.27,<2"]
|
|
@@ -136,12 +156,22 @@ module = [
|
|
|
136
156
|
ignore_missing_imports = true
|
|
137
157
|
|
|
138
158
|
[tool.uv]
|
|
139
|
-
#
|
|
140
|
-
#
|
|
141
|
-
#
|
|
142
|
-
#
|
|
159
|
+
# NOTE: overrides are a uv workspace feature — they do NOT ship in the built
|
|
160
|
+
# wheel. Anything reconciled only here is invisible to `pip install videopython`,
|
|
161
|
+
# so [ai] must be resolvable without this block. Keep it minimal.
|
|
162
|
+
#
|
|
163
|
+
# The torch/torchaudio/torchvision/diffusers overrides that used to live here
|
|
164
|
+
# existed solely to paper over chatterbox-tts's `==` pins. They masked the
|
|
165
|
+
# breakage from CI while every downstream consumer hit it, and an override
|
|
166
|
+
# replaces a requirement *everywhere* — including torchvision's own exact
|
|
167
|
+
# `torch==2.8.0` pin, which would have produced an ABI-invalid pair on the next
|
|
168
|
+
# `uv lock --upgrade`. [ai] now depends on videopython-chatterbox, whose metadata
|
|
169
|
+
# is correct, so none of them are needed.
|
|
170
|
+
# The numpy>=2.0.0 override is gone for the same reason: it existed to counter
|
|
171
|
+
# chatterbox-tts's `numpy<2.0.0` pin, and it also replaced numba's `numpy<2.5`
|
|
172
|
+
# cap, resolving numpy 2.5 into an environment where `import whisper` dies with
|
|
173
|
+
# "Numba needs NumPy 2.4 or less". numba IS version-coupled to numpy.
|
|
143
174
|
override-dependencies = [
|
|
144
|
-
"torch>=2.8.0", "torchaudio>=2.8.0", "torchvision>=0.23.0,<0.24.0", "numpy>=2.0.0", "diffusers>=0.35.0",
|
|
145
175
|
# Some transitive deps pull opencv-python, which conflicts with our
|
|
146
176
|
# opencv-python-headless (both provide cv2). Exclude opencv-python so
|
|
147
177
|
# only the headless variant is installed.
|
|
@@ -23,6 +23,13 @@ class OllamaStructuredClient:
|
|
|
23
23
|
must be served by a local Ollama daemon and support structured-output
|
|
24
24
|
``format`` (and vision, when images are passed); ``options`` are extra Ollama
|
|
25
25
|
generation options merged over ``temperature=0``.
|
|
26
|
+
|
|
27
|
+
Reasoning models emit their chain-of-thought *before* the schema-constrained
|
|
28
|
+
answer, and that thinking counts against ``num_predict``. On a reasoning model
|
|
29
|
+
(the default ``qwen3.6:27b`` is one) a translation call spends its entire token
|
|
30
|
+
budget thinking, stops on ``length``, and returns empty content. None of these
|
|
31
|
+
callers want the chain-of-thought, so thinking is disabled on models that
|
|
32
|
+
support it.
|
|
26
33
|
"""
|
|
27
34
|
|
|
28
35
|
def __init__(self, model: str, *, host: str | None = None, options: dict[str, Any] | None = None) -> None:
|
|
@@ -30,6 +37,7 @@ class OllamaStructuredClient:
|
|
|
30
37
|
self.host = host
|
|
31
38
|
self.options: dict[str, Any] = {"temperature": 0.0, **(options or {})}
|
|
32
39
|
self._client: Any = None
|
|
40
|
+
self._thinking_capable: bool | None = None
|
|
33
41
|
|
|
34
42
|
def _get_client(self) -> Any:
|
|
35
43
|
if self._client is None:
|
|
@@ -37,6 +45,17 @@ class OllamaStructuredClient:
|
|
|
37
45
|
self._client = ollama.Client(host=self.host)
|
|
38
46
|
return self._client
|
|
39
47
|
|
|
48
|
+
def _supports_thinking(self) -> bool:
|
|
49
|
+
"""Whether the model advertises Ollama's ``thinking`` capability (cached).
|
|
50
|
+
|
|
51
|
+
Passing ``think`` to a model that has no thinking capability is an error, so
|
|
52
|
+
this is checked rather than assumed.
|
|
53
|
+
"""
|
|
54
|
+
if self._thinking_capable is None:
|
|
55
|
+
capabilities = self._get_client().show(self.model).capabilities or []
|
|
56
|
+
self._thinking_capable = "thinking" in capabilities
|
|
57
|
+
return self._thinking_capable
|
|
58
|
+
|
|
40
59
|
def generate_json(
|
|
41
60
|
self,
|
|
42
61
|
*,
|
|
@@ -50,7 +69,12 @@ class OllamaStructuredClient:
|
|
|
50
69
|
if images:
|
|
51
70
|
user["images"] = [encode_png_b64(image) for image in images]
|
|
52
71
|
messages = [{"role": "system", "content": system}, user]
|
|
53
|
-
|
|
72
|
+
kwargs: dict[str, Any] = {}
|
|
73
|
+
if self._supports_thinking():
|
|
74
|
+
kwargs["think"] = False
|
|
75
|
+
response = self._get_client().chat(
|
|
76
|
+
model=self.model, messages=messages, format=schema, options=self.options, **kwargs
|
|
77
|
+
)
|
|
54
78
|
content = response.message.content
|
|
55
79
|
try:
|
|
56
80
|
data = json.loads(content)
|
|
@@ -236,7 +236,7 @@ class FaceSmoothingTracker(_FaceTrackerBase):
|
|
|
236
236
|
frame_center = (0.5, 0.5)
|
|
237
237
|
_, bbox = min(
|
|
238
238
|
faces_with_box,
|
|
239
|
-
key=lambda fb: (
|
|
239
|
+
key=lambda fb: (fb[1].center[0] - frame_center[0]) ** 2 + (fb[1].center[1] - frame_center[1]) ** 2,
|
|
240
240
|
)
|
|
241
241
|
elif self.selection_strategy == "index":
|
|
242
242
|
idx = self.face_index if self.face_index < len(faces_with_box) else 0
|
|
@@ -224,16 +224,12 @@ class Audio:
|
|
|
224
224
|
if dtype is None:
|
|
225
225
|
raise AudioLoadError(f"Unsupported sample width: {sample_width}")
|
|
226
226
|
|
|
227
|
-
|
|
227
|
+
# Explicitly annotated: numpy>=2.5 shape-types ndarray, so the
|
|
228
|
+
# 1-D frombuffer result cannot be rebound to a 2-D view below.
|
|
229
|
+
data: np.ndarray[Any, np.dtype[np.float32]]
|
|
230
|
+
data = np.frombuffer(raw_data, dtype=dtype).astype(np.float32)
|
|
228
231
|
|
|
229
|
-
# Reshape if stereo
|
|
230
|
-
if channels == 2:
|
|
231
|
-
data = data.reshape(-1, 2)
|
|
232
|
-
|
|
233
|
-
# Convert to float32
|
|
234
|
-
data = data.astype(np.float32)
|
|
235
|
-
|
|
236
|
-
# Reshape before normalization if stereo
|
|
232
|
+
# Reshape to (frames, channels) if stereo
|
|
237
233
|
if channels == 2:
|
|
238
234
|
data = data.reshape(-1, 2)
|
|
239
235
|
|
|
@@ -427,8 +427,9 @@ def extract_frames_at_indices(
|
|
|
427
427
|
# Truncate to complete frames only
|
|
428
428
|
raw_data = raw_data[: actual_frames * frame_size]
|
|
429
429
|
|
|
430
|
-
|
|
431
|
-
|
|
430
|
+
# Chained rather than reshaped in place: numpy>=2.5 shape-types ndarray, so
|
|
431
|
+
# rebinding a 1-D frombuffer result to a 4-D view is an assignment error.
|
|
432
|
+
frames = np.frombuffer(raw_data, dtype=np.uint8).copy().reshape(-1, metadata.height, metadata.width, 3)
|
|
432
433
|
|
|
433
434
|
# Reorder to match original frame_indices order if needed
|
|
434
435
|
if unique_sorted_indices != frame_indices:
|
|
@@ -38,6 +38,15 @@ if TYPE_CHECKING:
|
|
|
38
38
|
|
|
39
39
|
logger = logging.getLogger(__name__)
|
|
40
40
|
|
|
41
|
+
GRAIN_POOL_PAD = 128
|
|
42
|
+
"""Padding on each axis of :class:`FilmGrain`'s precomputed noise plane.
|
|
43
|
+
|
|
44
|
+
Bounds both the per-frame offset range (``pad**2`` = 16384 distinct windows,
|
|
45
|
+
far more than any realistic frame count reuses noticeably) and the pool's
|
|
46
|
+
memory: ``(h + pad) * (w + pad) * 3 * 2`` bytes, ~15 MB at 1080p. Constant in
|
|
47
|
+
clip duration, so the effect stays O(1)-memory.
|
|
48
|
+
"""
|
|
49
|
+
|
|
41
50
|
__all__ = [
|
|
42
51
|
"Effect",
|
|
43
52
|
"FullImageOverlay",
|
|
@@ -255,23 +264,66 @@ class ColorGrading(Effect):
|
|
|
255
264
|
description="Shift color temperature. -1.0 = cool/blue tint, 0 = neutral, 1.0 = warm/orange tint.",
|
|
256
265
|
)
|
|
257
266
|
|
|
258
|
-
|
|
259
|
-
|
|
267
|
+
_lut_tone: np.ndarray | None = PrivateAttr(default=None)
|
|
268
|
+
_lut_temp: np.ndarray | None = PrivateAttr(default=None)
|
|
260
269
|
|
|
261
|
-
|
|
262
|
-
|
|
270
|
+
@staticmethod
|
|
271
|
+
def _as_lut(channels: list[np.ndarray]) -> np.ndarray:
|
|
272
|
+
"""Pack three 256-entry ramps into the ``(1, 256, 3)`` array cv2.LUT wants."""
|
|
273
|
+
return np.ascontiguousarray(np.stack(channels, axis=1).reshape(1, 256, 3).astype(np.uint8))
|
|
274
|
+
|
|
275
|
+
def _build_luts(self) -> None:
|
|
276
|
+
"""Precompute the two point-operation lookup tables.
|
|
277
|
+
|
|
278
|
+
Brightness, contrast and temperature are all per-channel point
|
|
279
|
+
operations -- the output for an input byte depends on nothing else --
|
|
280
|
+
so they collapse into 256-entry tables built once instead of float
|
|
281
|
+
arithmetic over every pixel of every frame.
|
|
282
|
+
|
|
283
|
+
Two tables rather than one because saturation sits between them in the
|
|
284
|
+
documented order (brightness -> contrast -> saturation -> temperature)
|
|
285
|
+
and is *not* a point operation. Folding temperature into the first table
|
|
286
|
+
would silently reorder it ahead of saturation, and the two do not
|
|
287
|
+
commute.
|
|
288
|
+
"""
|
|
289
|
+
x = np.arange(256, dtype=np.float32) / 255.0
|
|
290
|
+
tone = x + self.brightness if self.brightness != 0 else x
|
|
263
291
|
if self.contrast != 1.0:
|
|
264
|
-
|
|
292
|
+
tone = (tone - 0.5) * self.contrast + 0.5
|
|
293
|
+
tone_u8 = np.clip(tone * 255.0, 0, 255)
|
|
294
|
+
self._lut_tone = self._as_lut([tone_u8, tone_u8, tone_u8])
|
|
295
|
+
|
|
296
|
+
shift = self.temperature * 0.1
|
|
297
|
+
self._lut_temp = self._as_lut(
|
|
298
|
+
[
|
|
299
|
+
np.clip((x + shift) * 255.0, 0, 255),
|
|
300
|
+
np.clip(x * 255.0, 0, 255),
|
|
301
|
+
np.clip((x - shift) * 255.0, 0, 255),
|
|
302
|
+
]
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
def _grade_frame(self, frame: np.ndarray) -> np.ndarray:
|
|
306
|
+
if self._lut_tone is None or self._lut_temp is None:
|
|
307
|
+
self._build_luts()
|
|
308
|
+
assert self._lut_tone is not None and self._lut_temp is not None
|
|
309
|
+
|
|
310
|
+
out = frame
|
|
311
|
+
if self.brightness != 0 or self.contrast != 1.0:
|
|
312
|
+
out = cv2.LUT(out, self._lut_tone)
|
|
265
313
|
if self.saturation != 1.0:
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
314
|
+
# Saturation as a blend toward the luma-weighted greyscale of the
|
|
315
|
+
# frame. Replaces an RGB->HSV->RGB float32 round trip that cost more
|
|
316
|
+
# than every other stage of the grade combined (~42 of 56 ms/frame).
|
|
317
|
+
grey = cv2.cvtColor(out, cv2.COLOR_RGB2GRAY)
|
|
318
|
+
out = cv2.addWeighted(
|
|
319
|
+
out, self.saturation, cv2.cvtColor(grey, cv2.COLOR_GRAY2RGB), 1.0 - self.saturation, 0
|
|
320
|
+
)
|
|
269
321
|
if self.temperature != 0:
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
img[:, :, 2] = img[:, :, 2] - temp_shift
|
|
322
|
+
out = cv2.LUT(out, self._lut_temp)
|
|
323
|
+
return out if out is not frame else frame.copy()
|
|
273
324
|
|
|
274
|
-
|
|
325
|
+
def streaming_init(self, total_frames: int, fps: float, width: int, height: int, **_context: Any) -> None:
|
|
326
|
+
self._build_luts()
|
|
275
327
|
|
|
276
328
|
def process_frame(self, frame: np.ndarray, frame_index: int) -> np.ndarray:
|
|
277
329
|
return self._grade_frame(frame)
|
|
@@ -299,7 +351,7 @@ class Vignette(Effect):
|
|
|
299
351
|
)
|
|
300
352
|
|
|
301
353
|
_mask: np.ndarray | None = PrivateAttr(default=None)
|
|
302
|
-
|
|
354
|
+
_stream_mask_u8: np.ndarray | None = PrivateAttr(default=None)
|
|
303
355
|
|
|
304
356
|
def _create_mask(self, height: int, width: int) -> np.ndarray:
|
|
305
357
|
y = np.linspace(-1, 1, height)
|
|
@@ -310,13 +362,28 @@ class Vignette(Effect):
|
|
|
310
362
|
return mask.astype(np.float32)
|
|
311
363
|
|
|
312
364
|
def streaming_init(self, total_frames: int, fps: float, width: int, height: int, **_context: Any) -> None:
|
|
365
|
+
"""Bake the gain mask into a 3-channel uint8 lookup, once per stream.
|
|
366
|
+
|
|
367
|
+
The mask is static, so the only per-frame work should be one multiply.
|
|
368
|
+
Storing it as uint8 replicated across channels lets ``cv2.multiply`` run
|
|
369
|
+
the whole frame in a single SIMD pass with no float conversion --
|
|
370
|
+
~10x faster than promoting every frame to float32, and the mask itself
|
|
371
|
+
is *smaller* than the float32 original (6.2 MB vs 8.3 MB at 1080p).
|
|
372
|
+
|
|
373
|
+
The clip to [0, 1] also fixes a real defect: ``_create_mask`` goes
|
|
374
|
+
NEGATIVE once ``strength`` is high enough (down to -1.0 at
|
|
375
|
+
``strength=1.0``), and ``(frame * -1.0).astype(np.uint8)`` wraps around,
|
|
376
|
+
so the darkest corners rendered as mid-grey (200 -> 56) and the vignette
|
|
377
|
+
got *brighter* past the zero crossing instead of saturating to black.
|
|
378
|
+
"""
|
|
313
379
|
if self._mask is None or self._mask.shape != (height, width):
|
|
314
380
|
self._mask = self._create_mask(height, width)
|
|
315
|
-
|
|
381
|
+
scaled = (np.clip(self._mask, 0.0, 1.0) * 255.0).astype(np.uint8)
|
|
382
|
+
self._stream_mask_u8 = np.ascontiguousarray(np.repeat(scaled[:, :, np.newaxis], 3, axis=2))
|
|
316
383
|
|
|
317
384
|
def process_frame(self, frame: np.ndarray, frame_index: int) -> np.ndarray:
|
|
318
|
-
assert self.
|
|
319
|
-
return (frame.
|
|
385
|
+
assert self._stream_mask_u8 is not None
|
|
386
|
+
return cv2.multiply(frame, self._stream_mask_u8, scale=1.0 / 255.0, dtype=cv2.CV_8U)
|
|
320
387
|
|
|
321
388
|
|
|
322
389
|
class KenBurns(Effect):
|
|
@@ -407,7 +474,6 @@ class Fade(Effect):
|
|
|
407
474
|
"""Fades video and audio to or from black."""
|
|
408
475
|
|
|
409
476
|
op: Literal["fade"] = "fade"
|
|
410
|
-
audio_coupled: ClassVar[bool] = True
|
|
411
477
|
|
|
412
478
|
mode: Literal["in", "out", "in_out"] = Field(
|
|
413
479
|
description=('"in" fades from black at the start, "out" fades to black at the end, "in_out" does both.'),
|
|
@@ -505,10 +571,18 @@ class Fade(Effect):
|
|
|
505
571
|
|
|
506
572
|
|
|
507
573
|
class VolumeAdjust(Effect):
|
|
508
|
-
"""Changes audio volume within a time range without affecting video frames.
|
|
574
|
+
"""Changes audio volume within a time range without affecting video frames.
|
|
575
|
+
|
|
576
|
+
Pixel-passthrough by construction: the effect exists entirely on the audio
|
|
577
|
+
graph (:meth:`to_ffmpeg_audio_filter` -> ``volume``). It declares
|
|
578
|
+
:attr:`video_passthrough` so the plan builder places that audio filter and
|
|
579
|
+
leaves the video chain alone -- without it, the op would schedule per-frame
|
|
580
|
+
Python and drag the whole segment through a rawvideo decode/encode
|
|
581
|
+
round-trip to run an identity function over every pixel.
|
|
582
|
+
"""
|
|
509
583
|
|
|
510
584
|
op: Literal["volume_adjust"] = "volume_adjust"
|
|
511
|
-
|
|
585
|
+
video_passthrough: ClassVar[bool] = True
|
|
512
586
|
|
|
513
587
|
volume: float = Field(
|
|
514
588
|
1.0,
|
|
@@ -521,8 +595,15 @@ class VolumeAdjust(Effect):
|
|
|
521
595
|
description="Seconds to smoothly ramp volume at the start and end of the window, preventing audible clicks.",
|
|
522
596
|
)
|
|
523
597
|
|
|
524
|
-
|
|
525
|
-
|
|
598
|
+
@property
|
|
599
|
+
def compiles_to_filter(self) -> bool:
|
|
600
|
+
"""Always filter-class: the whole effect is the audio twin.
|
|
601
|
+
|
|
602
|
+
Paired with :attr:`video_passthrough`, this routes the op down the
|
|
603
|
+
filter path, where :meth:`to_ffmpeg_filter` returning ``None`` is read
|
|
604
|
+
as "no video filter by design" rather than "failed to compile".
|
|
605
|
+
"""
|
|
606
|
+
return True
|
|
526
607
|
|
|
527
608
|
def to_ffmpeg_audio_filter(self, ctx: FilterCtx) -> str | None:
|
|
528
609
|
"""Apply the volume change over the window via the ``volume`` filter.
|
|
@@ -1343,21 +1424,41 @@ class FilmGrain(Effect):
|
|
|
1343
1424
|
)
|
|
1344
1425
|
seed: int = Field(0, description="Seed for the noise RNG. Same seed = same grain pattern.")
|
|
1345
1426
|
|
|
1346
|
-
|
|
1347
|
-
|
|
1348
|
-
|
|
1349
|
-
amp = self.intensity * 255.0
|
|
1350
|
-
if self.monochrome:
|
|
1351
|
-
noise = rng.standard_normal((h, w, 1), dtype=np.float32) * amp
|
|
1352
|
-
else:
|
|
1353
|
-
noise = rng.standard_normal((h, w, 3), dtype=np.float32) * amp
|
|
1354
|
-
return np.clip(frame.astype(np.float32) + noise, 0, 255).astype(np.uint8)
|
|
1427
|
+
_pool: np.ndarray | None = PrivateAttr(default=None)
|
|
1428
|
+
_offsets: np.ndarray | None = PrivateAttr(default=None)
|
|
1429
|
+
_geometry: tuple[int, int] = PrivateAttr(default=(0, 0))
|
|
1355
1430
|
|
|
1356
1431
|
def streaming_init(self, total_frames: int, fps: float, width: int, height: int, **_context: Any) -> None:
|
|
1357
|
-
|
|
1432
|
+
"""Draw one oversized noise plane up front; each frame reads a random window of it.
|
|
1433
|
+
|
|
1434
|
+
Generating fresh Gaussian noise per frame meant ~2M ``standard_normal``
|
|
1435
|
+
samples every frame, which dominated the effect (~31 ms/frame, more than
|
|
1436
|
+
twice the encoder's whole per-frame budget). Sampling a
|
|
1437
|
+
``GRAIN_POOL_PAD``-padded plane once and taking a randomly offset window
|
|
1438
|
+
per frame gives grain that still changes every frame -- offsets jump
|
|
1439
|
+
rather than drift, so it scintillates like film rather than sliding --
|
|
1440
|
+
for one saturating integer add.
|
|
1441
|
+
|
|
1442
|
+
Reproducibility is unchanged in contract (same ``seed`` -> same grain),
|
|
1443
|
+
though the pattern itself differs from the per-frame-RNG version.
|
|
1444
|
+
"""
|
|
1445
|
+
amp = self.intensity * 255.0
|
|
1446
|
+
pad = GRAIN_POOL_PAD
|
|
1447
|
+
shape: tuple[int, ...] = (height + pad, width + pad) if self.monochrome else (height + pad, width + pad, 3)
|
|
1448
|
+
noise = (np.random.default_rng(self.seed).standard_normal(shape, dtype=np.float32) * amp).astype(np.int16)
|
|
1449
|
+
if self.monochrome:
|
|
1450
|
+
# Luma-only grain: the same sample in all three channels.
|
|
1451
|
+
noise = np.repeat(noise[:, :, np.newaxis], 3, axis=2)
|
|
1452
|
+
self._pool = np.ascontiguousarray(noise)
|
|
1453
|
+
self._offsets = np.random.default_rng(self.seed + 1).integers(0, pad, size=(max(total_frames, 1), 2))
|
|
1454
|
+
self._geometry = (height, width)
|
|
1358
1455
|
|
|
1359
1456
|
def process_frame(self, frame: np.ndarray, frame_index: int) -> np.ndarray:
|
|
1360
|
-
|
|
1457
|
+
assert self._pool is not None and self._offsets is not None
|
|
1458
|
+
height, width = self._geometry
|
|
1459
|
+
oy, ox = self._offsets[frame_index % len(self._offsets)]
|
|
1460
|
+
# cv2.add saturates at 0/255, so no separate clip and no float promotion.
|
|
1461
|
+
return cv2.add(frame, self._pool[oy : oy + height, ox : ox + width], dtype=cv2.CV_8U)
|
|
1361
1462
|
|
|
1362
1463
|
|
|
1363
1464
|
class Sharpen(Effect):
|
|
@@ -474,12 +474,23 @@ class Effect(Operation):
|
|
|
474
474
|
:attr:`compiles_to_filter` and implement :meth:`to_ffmpeg_filter` (and, for
|
|
475
475
|
audio-coupled effects like ``Fade``/``VolumeAdjust``,
|
|
476
476
|
:meth:`to_ffmpeg_audio_filter`) so the window stays coherent across the
|
|
477
|
-
decode/encode graph.
|
|
477
|
+
decode/encode graph. An effect may implement BOTH contracts: the filter is
|
|
478
|
+
the fast path and ``process_frame`` stays as the reference implementation,
|
|
479
|
+
with ``src/tests/editing/test_filter_parity.py`` pinning them together.
|
|
478
480
|
"""
|
|
479
481
|
|
|
480
482
|
category: ClassVar[OpCategory] = OpCategory.EFFECT
|
|
481
|
-
|
|
482
|
-
"""Whether
|
|
483
|
+
video_passthrough: ClassVar[bool] = False
|
|
484
|
+
"""Whether a filter-class effect leaves pixels untouched (audio-only, e.g. ``volume_adjust``).
|
|
485
|
+
|
|
486
|
+
Only consulted when :attr:`compiles_to_filter` is True. It distinguishes the
|
|
487
|
+
two reasons :meth:`to_ffmpeg_filter` returns ``None``: "this op has no video
|
|
488
|
+
filter *by design*" (passthrough -- place the audio twin and move on) from
|
|
489
|
+
"this op failed to compile at this position" (fall through to the per-frame
|
|
490
|
+
path). Without it an audio-only effect is indistinguishable from a failed
|
|
491
|
+
compile and is forced onto the framewise pipeline, paying a full rawvideo
|
|
492
|
+
round-trip to run a no-op over every pixel.
|
|
493
|
+
"""
|
|
483
494
|
|
|
484
495
|
window: TimeRange | None = Field(
|
|
485
496
|
None,
|
|
@@ -148,7 +148,11 @@ def _classify_op_list(ops: Sequence[Operation], location_prefix: str) -> list[Op
|
|
|
148
148
|
# order either way and does not block later transforms the way
|
|
149
149
|
# a scheduled frame effect does. The builder mirrors this.
|
|
150
150
|
entries.append(OpStreamability(location, op.op, StreamingClass.FILTER))
|
|
151
|
-
if seen_effect:
|
|
151
|
+
if seen_effect and not op.video_passthrough:
|
|
152
|
+
# A video_passthrough effect (volume_adjust) places nothing
|
|
153
|
+
# on the video chain, so it cannot open the encode stage --
|
|
154
|
+
# a frame effect after it is still perfectly orderable.
|
|
155
|
+
# The builder mirrors this by skipping post_vf_filters.
|
|
152
156
|
seen_encode_stage = True
|
|
153
157
|
continue
|
|
154
158
|
if not op.streams():
|
|
@@ -2078,13 +2078,14 @@ class VideoEdit(BaseModel):
|
|
|
2078
2078
|
abandon()
|
|
2079
2079
|
return None
|
|
2080
2080
|
if op.compiles_to_filter:
|
|
2081
|
-
# Filter-class effect (add_subtitles):
|
|
2082
|
-
# context at compile time and joins the
|
|
2083
|
-
# this op's plan position -- the decode
|
|
2084
|
-
# frame effect precedes it, else the
|
|
2085
|
-
# (FrameEncoder -vf), which runs after
|
|
2086
|
-
# process_frame. Either way plan order is
|
|
2087
|
-
# None compile falls through to the
|
|
2081
|
+
# Filter-class effect (add_subtitles, vignette, ...):
|
|
2082
|
+
# consumes its context at compile time and joins the
|
|
2083
|
+
# filter chain at this op's plan position -- the decode
|
|
2084
|
+
# chain when no frame effect precedes it, else the
|
|
2085
|
+
# encode chain (FrameEncoder -vf), which runs after
|
|
2086
|
+
# every process_frame. Either way plan order is
|
|
2087
|
+
# preserved. A None compile falls through to the
|
|
2088
|
+
# frame-effect path UNLESS the op is video_passthrough.
|
|
2088
2089
|
encode_stage_effect = bool(effect_schedule or post_vf_filters)
|
|
2089
2090
|
ctx = make_ctx(decode_filters=None if encode_stage_effect else tuple(vf_filters))
|
|
2090
2091
|
filter_expr = op.to_ffmpeg_filter(ctx)
|
|
@@ -2099,6 +2100,17 @@ class VideoEdit(BaseModel):
|
|
|
2099
2100
|
# none today; kept coupled for extensibility).
|
|
2100
2101
|
compile_audio_twin(op, ctx, encode_stage_effect)
|
|
2101
2102
|
continue
|
|
2103
|
+
if op.video_passthrough:
|
|
2104
|
+
# Audio-only filter effect (volume_adjust): there is
|
|
2105
|
+
# no video filter to place by design, but the audio
|
|
2106
|
+
# twin still lands -- at the stage a video filter
|
|
2107
|
+
# WOULD have landed, so audio/video stage placement
|
|
2108
|
+
# stays coupled. Deliberately does not touch
|
|
2109
|
+
# vf_filters/post_vf_filters/pipe_meta: the op is
|
|
2110
|
+
# invisible to the video chain, so it neither opens
|
|
2111
|
+
# the encode stage nor blocks a later frame effect.
|
|
2112
|
+
compile_audio_twin(op, ctx, encode_stage_effect)
|
|
2113
|
+
continue
|
|
2102
2114
|
if post_vf_filters:
|
|
2103
2115
|
# A frame effect after an encode-stage filter would run
|
|
2104
2116
|
# before it (process_frame precedes the encoder), so
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/understanding/classification.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{videopython-0.54.0 → videopython-0.55.0}/src/videopython/ai/video_analysis/source_metadata.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|