videopython 0.55.1__tar.gz → 0.55.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {videopython-0.55.1 → videopython-0.55.3}/PKG-INFO +1 -1
- {videopython-0.55.1 → videopython-0.55.3}/pyproject.toml +1 -1
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/_ollama.py +58 -3
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/effects.py +5 -1
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/understanding/audio.py +10 -1
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/understanding/image.py +5 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/understanding/objects.py +64 -7
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/audio/audio.py +106 -65
- {videopython-0.55.1 → videopython-0.55.3}/.gitignore +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/LICENSE +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/README.md +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/_device.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/_optional.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/_predictor.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/_revisions.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/auto_edit/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/auto_edit/backend.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/auto_edit/catalog.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/auto_edit/editor.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/auto_edit/local.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/auto_edit/models.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/auto_edit/resolve.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/_tts_backend.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/audio_ops.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/config.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/dubber.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/models.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/pipeline.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/quality.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/remux.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/separation.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/timing.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/translation.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/dubbing/voice_sample.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/errors.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/generation/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/generation/audio.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/generation/image.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/generation/video.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/keyframe.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/ops.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/transforms.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/understanding/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/understanding/_detector.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/understanding/classification.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/understanding/faces.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/understanding/temporal.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/video_analysis/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/video_analysis/analyzer.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/video_analysis/detectors.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/video_analysis/models.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/video_analysis/sampling.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/video_analysis/source_metadata.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/audio/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/audio/analysis.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/_dimensions.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/_ffmpeg.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/_video_io.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/description.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/draw_detections.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/exceptions.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/Anton-OFL.txt +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/Anton-Regular.ttf +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/BebasNeue-OFL.txt +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/BebasNeue-Regular.ttf +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/DejaVuSans.ttf +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/LICENSE_DEJAVU +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/Lato-Bold.ttf +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/Lato-OFL.txt +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/Poppins-Bold.ttf +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/Poppins-OFL.txt +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/fonts/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/transcription.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/base/video.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/_ass.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/_easing.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/_schema.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/audio_ops.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/effects.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/operation.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/streaming.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/transcription_overlay.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/transforms.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/editing/video_edit.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/mcp/__init__.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/mcp/server.py +0 -0
- {videopython-0.55.1 → videopython-0.55.3}/src/videopython/py.typed +0 -0
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import json
|
|
6
|
+
import re
|
|
6
7
|
from typing import Any
|
|
7
8
|
|
|
8
9
|
import numpy as np
|
|
@@ -16,6 +17,45 @@ class OllamaError(AiError, RuntimeError):
|
|
|
16
17
|
"""Ollama returned unusable output (non-JSON or an unexpected shape)."""
|
|
17
18
|
|
|
18
19
|
|
|
20
|
+
# Ollama's own default context window is 4096 tokens, and an oversized request
|
|
21
|
+
# *fails* (``exceed_context_size_error``) instead of being truncated -- so a call
|
|
22
|
+
# carrying images has to size the window before sending. A vision model spends
|
|
23
|
+
# roughly this many tokens per image. Deliberately generous: underestimating
|
|
24
|
+
# fails the request outright, overestimating only costs KV-cache memory.
|
|
25
|
+
_TOKENS_PER_IMAGE = 1024
|
|
26
|
+
# Headroom for the system prompt, the user text, and the generated answer.
|
|
27
|
+
_TEXT_TOKEN_ALLOWANCE = 2048
|
|
28
|
+
# Never ask for less than Ollama's own default.
|
|
29
|
+
_MIN_NUM_CTX = 4096
|
|
30
|
+
_CONTEXT_OVERFLOW_RE = re.compile(r"request \((\d+) tokens\) exceeds the available context size")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _round_num_ctx(token_count: int) -> int:
|
|
34
|
+
return max(_MIN_NUM_CTX, 1 << (token_count - 1).bit_length())
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _num_ctx_for_images(image_count: int) -> int:
|
|
38
|
+
"""Estimate the initial context window for ``image_count`` images.
|
|
39
|
+
|
|
40
|
+
Rounded up to a power of two so a run over scenes with differing frame counts
|
|
41
|
+
reuses a handful of window sizes instead of a new one per call -- ``num_ctx``
|
|
42
|
+
is a runner-level setting, so varying it every call risks reloading the model
|
|
43
|
+
between scenes.
|
|
44
|
+
"""
|
|
45
|
+
needed = image_count * _TOKENS_PER_IMAGE + _TEXT_TOKEN_ALLOWANCE
|
|
46
|
+
return _round_num_ctx(needed)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _num_ctx_after_overflow(error: str, current: int) -> int | None:
|
|
50
|
+
"""Return a larger context after Ollama reports a context overflow."""
|
|
51
|
+
match = _CONTEXT_OVERFLOW_RE.search(error)
|
|
52
|
+
if match:
|
|
53
|
+
return _round_num_ctx(int(match.group(1)) + _TEXT_TOKEN_ALLOWANCE)
|
|
54
|
+
if "exceed_context_size_error" in error:
|
|
55
|
+
return current * 2
|
|
56
|
+
return None
|
|
57
|
+
|
|
58
|
+
|
|
19
59
|
class OllamaStructuredClient:
|
|
20
60
|
"""Generate schema-constrained JSON from text + optional images via Ollama.
|
|
21
61
|
|
|
@@ -30,6 +70,10 @@ class OllamaStructuredClient:
|
|
|
30
70
|
budget thinking, stops on ``length``, and returns empty content. None of these
|
|
31
71
|
callers want the chain-of-thought, so thinking is disabled on models that
|
|
32
72
|
support it.
|
|
73
|
+
|
|
74
|
+
Image requests start with a context estimate based on the image count. If a
|
|
75
|
+
model uses more visual tokens, the request retries once with the token count
|
|
76
|
+
reported by Ollama. An explicit ``num_ctx`` in ``options`` always wins.
|
|
33
77
|
"""
|
|
34
78
|
|
|
35
79
|
def __init__(self, model: str, *, host: str | None = None, options: dict[str, Any] | None = None) -> None:
|
|
@@ -69,12 +113,23 @@ class OllamaStructuredClient:
|
|
|
69
113
|
if images:
|
|
70
114
|
user["images"] = [encode_png_b64(image) for image in images]
|
|
71
115
|
messages = [{"role": "system", "content": system}, user]
|
|
116
|
+
options = self.options
|
|
117
|
+
image_count = len(images) if images else 0
|
|
118
|
+
auto_num_ctx = image_count > 0 and "num_ctx" not in options
|
|
119
|
+
if auto_num_ctx:
|
|
120
|
+
options = {**options, "num_ctx": _num_ctx_for_images(image_count)}
|
|
72
121
|
kwargs: dict[str, Any] = {}
|
|
73
122
|
if self._supports_thinking():
|
|
74
123
|
kwargs["think"] = False
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
124
|
+
client = self._get_client()
|
|
125
|
+
try:
|
|
126
|
+
response = client.chat(model=self.model, messages=messages, format=schema, options=options, **kwargs)
|
|
127
|
+
except Exception as exc:
|
|
128
|
+
retry_num_ctx = _num_ctx_after_overflow(str(exc), options["num_ctx"]) if auto_num_ctx else None
|
|
129
|
+
if retry_num_ctx is None:
|
|
130
|
+
raise
|
|
131
|
+
options = {**options, "num_ctx": retry_num_ctx}
|
|
132
|
+
response = client.chat(model=self.model, messages=messages, format=schema, options=options, **kwargs)
|
|
78
133
|
content = response.message.content
|
|
79
134
|
try:
|
|
80
135
|
data = json.loads(content)
|
|
@@ -43,7 +43,11 @@ class ObjectDetectionOverlay(Effect):
|
|
|
43
43
|
confidence_threshold: float = Field(0.5, ge=0, le=1, description="Minimum detection confidence to draw a box, 0-1.")
|
|
44
44
|
class_filter: list[str] | None = Field(
|
|
45
45
|
None,
|
|
46
|
-
description=
|
|
46
|
+
description=(
|
|
47
|
+
'Only draw these COCO class names, e.g. ["person", "car", "motorcycle"]. '
|
|
48
|
+
"Null draws all classes. Standard COCO names and D-FINE's alternate spellings "
|
|
49
|
+
"(motorbike, aeroplane, sofa, pottedplant, diningtable, tvmonitor) are accepted."
|
|
50
|
+
),
|
|
47
51
|
)
|
|
48
52
|
show_confidence: bool = Field(True, description="Append the detection confidence as a percentage to each label.")
|
|
49
53
|
box_color: tuple[int, int, int] | None = Field(
|
|
@@ -333,7 +333,16 @@ class AudioToText(ManagedPredictor):
|
|
|
333
333
|
)
|
|
334
334
|
|
|
335
335
|
all_words = self._assign_speakers_to_words(all_words, diarization_result)
|
|
336
|
-
|
|
336
|
+
|
|
337
|
+
# Rebuilding from words regroups by speaker and drops the per-segment
|
|
338
|
+
# confidence the supplied transcription carried, exactly as it does on the
|
|
339
|
+
# combined path -- so re-attach it the same way. Without this, splitting
|
|
340
|
+
# transcription and diarization into two calls silently loses confidence
|
|
341
|
+
# that running them as one keeps.
|
|
342
|
+
source_segments = transcription.segments
|
|
343
|
+
rebuilt = Transcription(words=all_words, language=transcription.language)
|
|
344
|
+
_attach_confidence_by_overlap(rebuilt.segments, source_segments)
|
|
345
|
+
return rebuilt
|
|
337
346
|
|
|
338
347
|
def _run_vad(self, audio_mono: Audio) -> list[tuple[float, float]]:
|
|
339
348
|
"""Return voiced spans in seconds using Silero VAD.
|
|
@@ -41,6 +41,11 @@ class SceneVLM(ManagedPredictor):
|
|
|
41
41
|
The model must be vision-capable and support Ollama's structured-output
|
|
42
42
|
``format``; ``ollama pull <model>`` first. ``options`` are extra Ollama
|
|
43
43
|
generation options merged over ``temperature=0``.
|
|
44
|
+
|
|
45
|
+
A scene's frames are sent as one multi-image request, so the context window
|
|
46
|
+
is sized to the frame count automatically -- Ollama's 4096-token default
|
|
47
|
+
fits only one or two frames and *fails* anything larger. Pass an explicit
|
|
48
|
+
``num_ctx`` in ``options`` to override that sizing.
|
|
44
49
|
"""
|
|
45
50
|
|
|
46
51
|
def __init__(
|
|
@@ -10,8 +10,12 @@ two stay one mental model. Consumed by
|
|
|
10
10
|
per-frame object analysis.
|
|
11
11
|
|
|
12
12
|
D-FINE (Apache-2.0) replaced the AGPL-licensed Ultralytics YOLO weights. Its COCO
|
|
13
|
-
labels use VOC-style spellings (``motorbike``, ``aeroplane``,
|
|
14
|
-
``diningtable``, ``tvmonitor``)
|
|
13
|
+
labels use VOC-style spellings for six classes (``motorbike``, ``aeroplane``,
|
|
14
|
+
``sofa``, ``pottedplant``, ``diningtable``, ``tvmonitor``) where the standard COCO
|
|
15
|
+
names are ``motorcycle``, ``airplane``, ``couch``, ``potted plant``,
|
|
16
|
+
``dining table`` and ``tv``. ``class_filter`` accepts either spelling: names are
|
|
17
|
+
normalized through :data:`CLASS_ALIASES` before matching, and any name the model
|
|
18
|
+
does not emit is logged once the class list is known.
|
|
15
19
|
"""
|
|
16
20
|
|
|
17
21
|
from __future__ import annotations
|
|
@@ -19,6 +23,9 @@ from __future__ import annotations
|
|
|
19
23
|
import logging
|
|
20
24
|
from typing import TYPE_CHECKING, Any
|
|
21
25
|
|
|
26
|
+
if TYPE_CHECKING:
|
|
27
|
+
from collections.abc import Iterable
|
|
28
|
+
|
|
22
29
|
from videopython.ai._revisions import pinned
|
|
23
30
|
from videopython.ai.understanding._detector import Backend, DetectorBase
|
|
24
31
|
from videopython.base.description import BoundingBox, DetectedObject
|
|
@@ -28,7 +35,31 @@ if TYPE_CHECKING:
|
|
|
28
35
|
|
|
29
36
|
logger = logging.getLogger(__name__)
|
|
30
37
|
|
|
31
|
-
__all__ = ["ObjectDetector", "MODEL_SIZES"]
|
|
38
|
+
__all__ = ["ObjectDetector", "MODEL_SIZES", "CLASS_ALIASES", "normalize_class_names"]
|
|
39
|
+
|
|
40
|
+
# D-FINE emits VOC-style spellings for six COCO classes. Callers -- and LLMs asked
|
|
41
|
+
# to name a class -- reach for the standard COCO spelling, which would match
|
|
42
|
+
# nothing and draw nothing, with no error to explain the silence. Accept both.
|
|
43
|
+
CLASS_ALIASES: dict[str, str] = {
|
|
44
|
+
"motorcycle": "motorbike",
|
|
45
|
+
"airplane": "aeroplane",
|
|
46
|
+
"couch": "sofa",
|
|
47
|
+
"potted plant": "pottedplant",
|
|
48
|
+
"dining table": "diningtable",
|
|
49
|
+
"tv": "tvmonitor",
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def normalize_class_names(names: "Iterable[str]") -> tuple[str, ...]:
|
|
54
|
+
"""Map COCO-standard class names onto the VOC-style spellings D-FINE emits.
|
|
55
|
+
|
|
56
|
+
Case and surrounding whitespace are normalized first, so ``"Potted Plant"``
|
|
57
|
+
and ``"potted plant"`` both reach ``pottedplant``. A name with no alias is
|
|
58
|
+
passed through unchanged (lowercased) -- this translates spellings, it does
|
|
59
|
+
not validate them; see :meth:`ObjectDetector.unknown_filter_classes`.
|
|
60
|
+
"""
|
|
61
|
+
return tuple(CLASS_ALIASES.get(key, key) for key in (" ".join(n.lower().split()) for n in names))
|
|
62
|
+
|
|
32
63
|
|
|
33
64
|
# D-FINE COCO checkpoints (Apache-2.0), in ascending size/quality. ``model_size``
|
|
34
65
|
# on ``ObjectDetectionOverlay`` maps onto these; pinned in ``_revisions.py``.
|
|
@@ -46,7 +77,9 @@ class ObjectDetector(DetectorBase[DetectedObject]):
|
|
|
46
77
|
The D-FINE weights (default ``ustc-community/dfine-nano-coco``) download from
|
|
47
78
|
HuggingFace on first real use; class names come from the model config.
|
|
48
79
|
Detection is gated by ``confidence_threshold`` and optionally restricted to
|
|
49
|
-
``class_filter
|
|
80
|
+
``class_filter``, which accepts either D-FINE's VOC-style spellings or the
|
|
81
|
+
standard COCO ones (``motorcycle``, ``airplane``, ``couch``, ``potted plant``,
|
|
82
|
+
``dining table``, ``tv``) -- see :func:`normalize_class_names`.
|
|
50
83
|
"""
|
|
51
84
|
|
|
52
85
|
DEFAULT_CONFIDENCE_THRESHOLD = 0.5
|
|
@@ -68,14 +101,18 @@ class ObjectDetector(DetectorBase[DetectedObject]):
|
|
|
68
101
|
``ustc-community/dfine-nano-coco``, ``...-small-coco``,
|
|
69
102
|
``...-medium-coco``, ``...-large-coco``). Downloaded on first use.
|
|
70
103
|
confidence_threshold: Minimum detection confidence in ``[0, 1]``.
|
|
71
|
-
class_filter: If non-empty, only these COCO class names are kept
|
|
72
|
-
|
|
104
|
+
class_filter: If non-empty, only these COCO class names are kept.
|
|
105
|
+
Either spelling works -- ``motorcycle`` and ``motorbike`` both
|
|
106
|
+
match -- and names are normalized via
|
|
107
|
+
:func:`normalize_class_names`. A name the model never emits is
|
|
108
|
+
logged as a warning once the model loads, since it would
|
|
109
|
+
otherwise just silently match nothing.
|
|
73
110
|
backend: Detection device - ``"cpu"``, ``"gpu"``, or ``"auto"``.
|
|
74
111
|
"""
|
|
75
112
|
super().__init__(backend=backend)
|
|
76
113
|
self.model_name = model_name
|
|
77
114
|
self.confidence_threshold = confidence_threshold
|
|
78
|
-
self.class_filter =
|
|
115
|
+
self.class_filter = normalize_class_names(class_filter)
|
|
79
116
|
self._model: Any = None
|
|
80
117
|
self._processor: Any = None
|
|
81
118
|
self._class_names: dict[int, str] = {}
|
|
@@ -93,6 +130,26 @@ class ObjectDetector(DetectorBase[DetectedObject]):
|
|
|
93
130
|
model = model.to("cuda")
|
|
94
131
|
self._model = model
|
|
95
132
|
self._class_names = {int(k): v for k, v in model.config.id2label.items()}
|
|
133
|
+
self._warn_unknown_filter_classes()
|
|
134
|
+
|
|
135
|
+
def unknown_filter_classes(self) -> tuple[str, ...]:
|
|
136
|
+
"""``class_filter`` names this model never emits (empty until it loads)."""
|
|
137
|
+
if not self._class_names:
|
|
138
|
+
return ()
|
|
139
|
+
known = set(self._class_names.values())
|
|
140
|
+
return tuple(name for name in self.class_filter if name not in known)
|
|
141
|
+
|
|
142
|
+
def _warn_unknown_filter_classes(self) -> None:
|
|
143
|
+
"""Log filter names that cannot match, which would otherwise draw nothing."""
|
|
144
|
+
unknown = self.unknown_filter_classes()
|
|
145
|
+
if unknown:
|
|
146
|
+
logger.warning(
|
|
147
|
+
"class_filter names not emitted by %s: %s. Nothing will be detected for them; "
|
|
148
|
+
"the model's classes are %s.",
|
|
149
|
+
self.model_name,
|
|
150
|
+
", ".join(unknown),
|
|
151
|
+
", ".join(sorted(set(self._class_names.values()))),
|
|
152
|
+
)
|
|
96
153
|
|
|
97
154
|
def _infer(self, images: list[np.ndarray]) -> list[list[DetectedObject]]:
|
|
98
155
|
import torch
|
|
@@ -2,6 +2,7 @@ from __future__ import annotations
|
|
|
2
2
|
|
|
3
3
|
import io
|
|
4
4
|
import subprocess
|
|
5
|
+
import tempfile
|
|
5
6
|
import wave
|
|
6
7
|
from dataclasses import dataclass
|
|
7
8
|
from pathlib import Path
|
|
@@ -16,6 +17,17 @@ from videopython.base.exceptions import AudioLoadError, FFmpegProbeError
|
|
|
16
17
|
if TYPE_CHECKING:
|
|
17
18
|
from videopython.audio.analysis import AudioLevels, AudioSegment, AudioSegmentType, SilentSegment
|
|
18
19
|
|
|
20
|
+
# What `Audio.from_path` asks ffmpeg for. Piped WAV comes back as pcm_s16le whatever
|
|
21
|
+
# the source's bit depth, so requesting raw s16le loses no fidelity that the WAV path
|
|
22
|
+
# was preserving, and there is no header to parse back.
|
|
23
|
+
_PCM_FORMAT = "s16le"
|
|
24
|
+
_PCM_DTYPE = np.int16
|
|
25
|
+
_PCM_SAMPLE_WIDTH = 2
|
|
26
|
+
|
|
27
|
+
# Read size when draining ffmpeg's stdout: big enough that a multi-GB decode is not
|
|
28
|
+
# millions of round trips, small enough to be irrelevant for a short clip.
|
|
29
|
+
_DECODE_CHUNK_BYTES = 8 << 20
|
|
30
|
+
|
|
19
31
|
|
|
20
32
|
def atempo_chain(speed: float) -> list[str]:
|
|
21
33
|
"""Build the ``atempo`` filter chain that time-stretches audio by ``speed``.
|
|
@@ -163,100 +175,131 @@ class Audio:
|
|
|
163
175
|
return cls(data, metadata)
|
|
164
176
|
|
|
165
177
|
@classmethod
|
|
166
|
-
def from_path(
|
|
178
|
+
def from_path(
|
|
179
|
+
cls,
|
|
180
|
+
file_path: str | Path,
|
|
181
|
+
*,
|
|
182
|
+
sample_rate: int | None = None,
|
|
183
|
+
channels: int | None = None,
|
|
184
|
+
) -> Audio:
|
|
167
185
|
"""
|
|
168
|
-
Load audio from a file using ffmpeg
|
|
186
|
+
Load audio from a file using ffmpeg.
|
|
187
|
+
|
|
188
|
+
``sample_rate`` and ``channels`` ask ffmpeg to convert *while decoding*
|
|
189
|
+
rather than loading the source in full and converting afterwards. For a
|
|
190
|
+
caller that only wants 16kHz mono -- speech recognition, diarization,
|
|
191
|
+
speaker embeddings -- that is the difference between holding the whole
|
|
192
|
+
source in memory and holding a twelfth of it: a 12-hour 48kHz stereo
|
|
193
|
+
recording is 16.5GB of float32 at source rate and 1.4GB at 16kHz mono.
|
|
194
|
+
Resampling uses soxr, the engine :meth:`resample` uses, so the result
|
|
195
|
+
tracks ``Audio.from_path(p).to_mono().resample(r)`` sample for sample at
|
|
196
|
+
an error RMS around one 16-bit LSB -- the two quantize at different points
|
|
197
|
+
in the chain, and neither is the more faithful for it.
|
|
169
198
|
|
|
170
199
|
Args:
|
|
171
200
|
file_path: Path to the audio file
|
|
201
|
+
sample_rate: Decode at this rate instead of the source's.
|
|
202
|
+
channels: Decode to this many channels instead of the source's.
|
|
203
|
+
``1`` downmixes to mono.
|
|
172
204
|
|
|
173
205
|
Returns:
|
|
174
206
|
Audio: New Audio instance
|
|
175
207
|
|
|
176
208
|
Raises:
|
|
177
209
|
FileNotFoundError: If the file doesn't exist
|
|
210
|
+
ValueError: If ``sample_rate`` or ``channels`` is not positive
|
|
178
211
|
AudioLoadError: If there's an error loading the audio
|
|
179
212
|
"""
|
|
180
213
|
file_path = Path(file_path)
|
|
181
214
|
if not file_path.exists():
|
|
182
215
|
raise FileNotFoundError(f"File not found: {file_path}")
|
|
216
|
+
if sample_rate is not None and sample_rate <= 0:
|
|
217
|
+
raise ValueError("Sample rate must be positive")
|
|
218
|
+
if channels is not None and channels <= 0:
|
|
219
|
+
raise ValueError("Channel count must be positive")
|
|
183
220
|
|
|
184
|
-
# Get audio info
|
|
185
221
|
info = cls._get_ffmpeg_info(file_path)
|
|
186
|
-
|
|
187
|
-
|
|
222
|
+
target_rate = info["sample_rate"] if sample_rate is None else sample_rate
|
|
223
|
+
target_channels = info["channels"] if channels is None else channels
|
|
224
|
+
|
|
225
|
+
# Raw PCM rather than a WAV round-trip. Piped WAV comes back as pcm_s16le
|
|
226
|
+
# whatever the source's bit depth -- `-bits_per_raw_sample` is a hint the
|
|
227
|
+
# WAV muxer does not act on -- and its header carries a placeholder length
|
|
228
|
+
# because a pipe is not seekable. So parsing it back told us only what we
|
|
229
|
+
# had already asked for, and cost two more full copies of the audio on the
|
|
230
|
+
# way: one for `BytesIO`, one for `readframes`.
|
|
188
231
|
cmd = [
|
|
189
232
|
"ffmpeg",
|
|
233
|
+
"-v",
|
|
234
|
+
"error",
|
|
190
235
|
"-i",
|
|
191
236
|
str(file_path),
|
|
192
237
|
"-f",
|
|
193
|
-
|
|
238
|
+
_PCM_FORMAT,
|
|
194
239
|
"-ar",
|
|
195
|
-
str(
|
|
240
|
+
str(target_rate),
|
|
196
241
|
"-ac",
|
|
197
|
-
str(
|
|
198
|
-
|
|
199
|
-
|
|
242
|
+
str(target_channels),
|
|
243
|
+
# soxr rather than ffmpeg's default resampler, so that decoding at a
|
|
244
|
+
# rate and resampling to it afterwards agree.
|
|
245
|
+
"-af",
|
|
246
|
+
"aresample=resampler=soxr",
|
|
200
247
|
"-", # Output to stdout
|
|
201
248
|
]
|
|
202
249
|
|
|
203
250
|
try:
|
|
204
|
-
|
|
205
|
-
|
|
251
|
+
# stderr to a file, not a pipe: stdout is drained to completion before
|
|
252
|
+
# stderr is read, and a full stderr pipe would deadlock ffmpeg partway
|
|
253
|
+
# through the audio.
|
|
254
|
+
with tempfile.TemporaryFile() as errors:
|
|
255
|
+
process = subprocess.Popen(cmd, stdout=subprocess.PIPE, stderr=errors)
|
|
256
|
+
assert process.stdout is not None
|
|
257
|
+
# A bytearray rather than `communicate()`, which accumulates chunks
|
|
258
|
+
# in a list and joins them at the end -- holding the whole of a long
|
|
259
|
+
# decode twice at the moment it completes.
|
|
260
|
+
raw = bytearray()
|
|
261
|
+
try:
|
|
262
|
+
while chunk := process.stdout.read(_DECODE_CHUNK_BYTES):
|
|
263
|
+
raw += chunk
|
|
264
|
+
finally:
|
|
265
|
+
process.stdout.close()
|
|
266
|
+
if process.wait() != 0:
|
|
267
|
+
errors.seek(0)
|
|
268
|
+
raise AudioLoadError(f"FFmpeg error: {errors.read().decode(errors='replace')}")
|
|
269
|
+
except subprocess.CalledProcessError as e:
|
|
270
|
+
raise AudioLoadError(f"Error running ffmpeg: {e}")
|
|
206
271
|
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
with io.BytesIO(wav_data) as wav_io:
|
|
212
|
-
with wave.open(wav_io, "rb") as wav_file:
|
|
213
|
-
# Get WAV metadata
|
|
214
|
-
sample_width = wav_file.getsampwidth()
|
|
215
|
-
channels = wav_file.getnchannels()
|
|
216
|
-
sample_rate = wav_file.getframerate()
|
|
217
|
-
n_frames = wav_file.getnframes()
|
|
218
|
-
|
|
219
|
-
# Read raw audio data
|
|
220
|
-
raw_data = wav_file.readframes(n_frames)
|
|
221
|
-
|
|
222
|
-
# Convert bytes to numpy array based on sample width
|
|
223
|
-
dtype_map = {1: np.int8, 2: np.int16, 4: np.int32}
|
|
224
|
-
dtype = dtype_map.get(sample_width)
|
|
225
|
-
if dtype is None:
|
|
226
|
-
raise AudioLoadError(f"Unsupported sample width: {sample_width}")
|
|
227
|
-
|
|
228
|
-
# Explicitly annotated: numpy>=2.5 shape-types ndarray, so the
|
|
229
|
-
# 1-D frombuffer result cannot be rebound to a 2-D view below.
|
|
230
|
-
data: np.ndarray[Any, np.dtype[np.float32]]
|
|
231
|
-
data = np.frombuffer(raw_data, dtype=dtype).astype(np.float32)
|
|
232
|
-
|
|
233
|
-
# Reshape to (frames, channels) if stereo
|
|
234
|
-
if channels == 2:
|
|
235
|
-
data = data.reshape(-1, 2)
|
|
236
|
-
|
|
237
|
-
# Normalize to float between -1 and 1
|
|
238
|
-
max_value = float(np.iinfo(dtype).max) # type: ignore
|
|
239
|
-
data = data / max_value
|
|
240
|
-
|
|
241
|
-
# Ensure normalization is within bounds due to floating point precision
|
|
242
|
-
data = np.clip(data, -1.0, 1.0)
|
|
243
|
-
|
|
244
|
-
# Calculate frame count from actual data length
|
|
245
|
-
# For stereo, len(data) is already correct after reshape
|
|
246
|
-
frame_count = len(data)
|
|
247
|
-
|
|
248
|
-
metadata = AudioMetadata(
|
|
249
|
-
sample_rate=sample_rate,
|
|
250
|
-
channels=channels,
|
|
251
|
-
sample_width=sample_width,
|
|
252
|
-
duration_seconds=info["duration"],
|
|
253
|
-
frame_count=frame_count,
|
|
254
|
-
)
|
|
272
|
+
# A truncated final frame would otherwise make `frombuffer` raise on a file
|
|
273
|
+
# that is entirely usable up to that point.
|
|
274
|
+
frame_bytes = _PCM_SAMPLE_WIDTH * target_channels
|
|
275
|
+
usable = len(raw) - (len(raw) % frame_bytes)
|
|
255
276
|
|
|
256
|
-
|
|
277
|
+
# Explicitly annotated: numpy>=2.5 shape-types ndarray, so the
|
|
278
|
+
# 1-D frombuffer result cannot be rebound to a 2-D view below.
|
|
279
|
+
data: np.ndarray[Any, np.dtype[np.float32]]
|
|
280
|
+
data = np.frombuffer(memoryview(raw)[:usable], dtype=_PCM_DTYPE).astype(np.float32)
|
|
257
281
|
|
|
258
|
-
|
|
259
|
-
|
|
282
|
+
# Reshape to (frames, channels) if stereo
|
|
283
|
+
if target_channels == 2:
|
|
284
|
+
data = data.reshape(-1, 2)
|
|
285
|
+
|
|
286
|
+
# Normalize to float between -1 and 1, and clamp for floating-point
|
|
287
|
+
# precision. Both in place: at these sizes a copy per step is most of what
|
|
288
|
+
# makes decoding a long file expensive.
|
|
289
|
+
data /= float(np.iinfo(_PCM_DTYPE).max)
|
|
290
|
+
np.clip(data, -1.0, 1.0, out=data)
|
|
291
|
+
|
|
292
|
+
# Calculate frame count from actual data length
|
|
293
|
+
# For stereo, len(data) is already correct after reshape
|
|
294
|
+
metadata = AudioMetadata(
|
|
295
|
+
sample_rate=target_rate,
|
|
296
|
+
channels=target_channels,
|
|
297
|
+
sample_width=_PCM_SAMPLE_WIDTH,
|
|
298
|
+
duration_seconds=info["duration"],
|
|
299
|
+
frame_count=len(data),
|
|
300
|
+
)
|
|
301
|
+
|
|
302
|
+
return cls(data, metadata)
|
|
260
303
|
|
|
261
304
|
@classmethod
|
|
262
305
|
def from_file(cls, file_path: str | Path) -> Audio:
|
|
@@ -653,8 +696,6 @@ class Audio:
|
|
|
653
696
|
filter_str = ",".join(filters) if filters else "anull"
|
|
654
697
|
|
|
655
698
|
# Save current audio to temp WAV, process with ffmpeg, read back
|
|
656
|
-
import tempfile
|
|
657
|
-
|
|
658
699
|
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as input_file:
|
|
659
700
|
input_path = input_file.name
|
|
660
701
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/understanding/classification.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{videopython-0.55.1 → videopython-0.55.3}/src/videopython/ai/video_analysis/source_metadata.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|