syncnet-python 0.2.1__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/PKG-INFO +1 -2
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/README.md +1 -2
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/pyproject.toml +1 -1
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/__init__.py +12 -1
- syncnet_python-0.2.2/syncnet_python/safe_syncnet_utils.py +154 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/syncnet_pipeline.py +68 -20
- syncnet_python-0.2.2/syncnet_python/test_error_handling.py +141 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python.egg-info/PKG-INFO +1 -2
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python.egg-info/SOURCES.txt +2 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/CLAUDE.md +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/INSTALL.md +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/LICENSE +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/MANIFEST.in +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/example/speech.wav +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/example/video.avi +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/requirements.txt +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/scripts/run_batch.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/scripts/run_example.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/setup.cfg +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/setup.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/__init__.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/cli.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/__init__.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/audio.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/base.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/compat.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/config.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/exceptions.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/inference.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/logging.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/models.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/sync_analyzer.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/types.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/utils.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/core/video.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/detectors/__init__.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/detectors/s3fd/__init__.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/detectors/s3fd/detector.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/detectors/s3fd/utils.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/pipeline/__init__.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/pipeline/config.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/pipeline/pipeline.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/utils/__init__.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/utils/exceptions.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/utils/face_detection.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet/utils/video.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/SyncNetInstance.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/SyncNetModel.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/cli.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/detectors/__init__.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/detectors/s3fd/__init__.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/detectors/s3fd/box_utils.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/detectors/s3fd/nets.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/run_syncnet_pipeline_on_1example.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/test_lse_metrics.py +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python.egg-info/dependency_links.txt +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python.egg-info/entry_points.txt +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python.egg-info/requires.txt +0 -0
- {syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: syncnet-python
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: SyncNet: Audio-visual synchronization detection using deep learning. Updated version of https://github.com/joonson/syncnet_python for modern Python versions.
|
|
5
5
|
Author: SyncNet Python Contributors
|
|
6
6
|
Maintainer: SyncNet Python Contributors
|
|
@@ -49,7 +49,6 @@ Dynamic: license-file
|
|
|
49
49
|
|
|
50
50
|
[](https://badge.fury.io/py/syncnet-python)
|
|
51
51
|
[](https://pypi.org/project/syncnet-python/)
|
|
52
|
-
[](https://pepy.tech/project/syncnet-python)
|
|
53
52
|
[](https://opensource.org/licenses/MIT)
|
|
54
53
|
|
|
55
54
|
Audio-visual synchronization detection using deep learning with modern Python architecture.
|
|
@@ -2,7 +2,6 @@
|
|
|
2
2
|
|
|
3
3
|
[](https://badge.fury.io/py/syncnet-python)
|
|
4
4
|
[](https://pypi.org/project/syncnet-python/)
|
|
5
|
-
[](https://pepy.tech/project/syncnet-python)
|
|
6
5
|
[](https://opensource.org/licenses/MIT)
|
|
7
6
|
|
|
8
7
|
Audio-visual synchronization detection using deep learning with modern Python architecture.
|
|
@@ -167,4 +166,4 @@ MIT License - see LICENSE file for details.
|
|
|
167
166
|
|
|
168
167
|
- GitHub: https://github.com/yourusername/syncnet-python
|
|
169
168
|
- Documentation: https://syncnet-python.readthedocs.io
|
|
170
|
-
- Issues: https://github.com/yourusername/syncnet-python/issues
|
|
169
|
+
- Issues: https://github.com/yourusername/syncnet-python/issues
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "syncnet-python"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.2"
|
|
8
8
|
description = "SyncNet: Audio-visual synchronization detection using deep learning. Updated version of https://github.com/joonson/syncnet_python for modern Python versions."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
@@ -4,23 +4,34 @@ This package provides a PyTorch implementation of SyncNet for detecting
|
|
|
4
4
|
synchronization between audio and video in multimedia content.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
__version__ = "0.
|
|
7
|
+
__version__ = "0.2.2"
|
|
8
8
|
|
|
9
9
|
# Import main components
|
|
10
10
|
try:
|
|
11
11
|
from .syncnet_pipeline import SyncNetPipeline
|
|
12
12
|
from .SyncNetModel import S as SyncNetModel
|
|
13
13
|
from .SyncNetInstance import SyncNetInstance
|
|
14
|
+
from .safe_syncnet_utils import (
|
|
15
|
+
safe_syncnet_inference,
|
|
16
|
+
extract_audio_from_video,
|
|
17
|
+
calculate_lse_metrics
|
|
18
|
+
)
|
|
14
19
|
except ImportError:
|
|
15
20
|
# Fallback for development
|
|
16
21
|
SyncNetPipeline = None
|
|
17
22
|
SyncNetModel = None
|
|
18
23
|
SyncNetInstance = None
|
|
24
|
+
safe_syncnet_inference = None
|
|
25
|
+
extract_audio_from_video = None
|
|
26
|
+
calculate_lse_metrics = None
|
|
19
27
|
|
|
20
28
|
__all__ = [
|
|
21
29
|
"SyncNetPipeline",
|
|
22
30
|
"SyncNetModel",
|
|
23
31
|
"SyncNetInstance",
|
|
32
|
+
"safe_syncnet_inference",
|
|
33
|
+
"extract_audio_from_video",
|
|
34
|
+
"calculate_lse_metrics",
|
|
24
35
|
"__version__"
|
|
25
36
|
]
|
|
26
37
|
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""Safe SyncNet utilities with comprehensive error handling."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import tempfile
|
|
5
|
+
import os
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Optional, Tuple, List
|
|
8
|
+
try:
|
|
9
|
+
from .syncnet_pipeline import SyncNetPipeline
|
|
10
|
+
except ImportError:
|
|
11
|
+
from syncnet_pipeline import SyncNetPipeline
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def safe_syncnet_inference(
|
|
15
|
+
pipeline: SyncNetPipeline,
|
|
16
|
+
video_path: str,
|
|
17
|
+
audio_path: Optional[str] = None,
|
|
18
|
+
cache_dir: Optional[str] = None
|
|
19
|
+
) -> Tuple[List[int], List[float], List[float], float, float, str, bool]:
|
|
20
|
+
"""
|
|
21
|
+
Safe SyncNet inference with automatic audio extraction if needed.
|
|
22
|
+
|
|
23
|
+
This function provides a wrapper around SyncNetPipeline.inference() that:
|
|
24
|
+
1. Handles audio_path=None by automatically extracting audio from video
|
|
25
|
+
2. Provides comprehensive error handling and logging
|
|
26
|
+
3. Ensures proper cleanup of temporary files
|
|
27
|
+
|
|
28
|
+
Args:
|
|
29
|
+
pipeline: Initialized SyncNetPipeline instance
|
|
30
|
+
video_path: Path to input video file
|
|
31
|
+
audio_path: Path to audio file (None for auto-extraction from video)
|
|
32
|
+
cache_dir: Directory for temporary files (None for auto-cleanup)
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
Tuple of (offsets, confidences, distances, max_conf, min_dist, s3fd_json, has_face)
|
|
36
|
+
|
|
37
|
+
Raises:
|
|
38
|
+
RuntimeError: If processing fails at any stage
|
|
39
|
+
FileNotFoundError: If video file doesn't exist
|
|
40
|
+
ValueError: If video format is not supported
|
|
41
|
+
"""
|
|
42
|
+
if not os.path.exists(video_path):
|
|
43
|
+
raise FileNotFoundError(f"Video file not found: {video_path}")
|
|
44
|
+
|
|
45
|
+
if audio_path and not os.path.exists(audio_path):
|
|
46
|
+
raise FileNotFoundError(f"Audio file not found: {audio_path}")
|
|
47
|
+
|
|
48
|
+
logging.info(f"Starting safe SyncNet inference for video: {video_path}")
|
|
49
|
+
|
|
50
|
+
try:
|
|
51
|
+
# Use the updated inference method that handles audio_path=None
|
|
52
|
+
results = pipeline.inference(
|
|
53
|
+
video_path=video_path,
|
|
54
|
+
audio_path=audio_path,
|
|
55
|
+
cache_dir=cache_dir
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
offsets, confs, dists, max_conf, min_dist, s3fd_json, has_face = results
|
|
59
|
+
|
|
60
|
+
# Validate results
|
|
61
|
+
if not has_face:
|
|
62
|
+
logging.warning("No faces detected in the video")
|
|
63
|
+
elif not offsets:
|
|
64
|
+
logging.warning("No valid face tracks found")
|
|
65
|
+
else:
|
|
66
|
+
logging.info(f"Successfully processed {len(offsets)} face tracks")
|
|
67
|
+
logging.info(f"LSE-C (max confidence): {max_conf:.3f}")
|
|
68
|
+
logging.info(f"LSE-D (min distance): {min_dist:.3f}")
|
|
69
|
+
|
|
70
|
+
return results
|
|
71
|
+
|
|
72
|
+
except Exception as e:
|
|
73
|
+
logging.error(f"SyncNet inference failed: {str(e)}")
|
|
74
|
+
raise RuntimeError(f"SyncNet processing failed: {str(e)}") from e
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def extract_audio_from_video(video_path: str, audio_path: str) -> None:
|
|
78
|
+
"""
|
|
79
|
+
Extract audio from video file using ffmpeg.
|
|
80
|
+
|
|
81
|
+
Args:
|
|
82
|
+
video_path: Path to input video file
|
|
83
|
+
audio_path: Path for output audio file
|
|
84
|
+
|
|
85
|
+
Raises:
|
|
86
|
+
RuntimeError: If audio extraction fails
|
|
87
|
+
FileNotFoundError: If video file doesn't exist
|
|
88
|
+
"""
|
|
89
|
+
if not os.path.exists(video_path):
|
|
90
|
+
raise FileNotFoundError(f"Video file not found: {video_path}")
|
|
91
|
+
|
|
92
|
+
# Create a temporary pipeline instance just for audio extraction
|
|
93
|
+
pipeline = SyncNetPipeline()
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
pipeline._extract_audio_from_video(video_path, audio_path)
|
|
97
|
+
logging.info(f"Successfully extracted audio from {video_path} to {audio_path}")
|
|
98
|
+
except Exception as e:
|
|
99
|
+
logging.error(f"Audio extraction failed: {str(e)}")
|
|
100
|
+
raise RuntimeError(f"Audio extraction failed: {str(e)}") from e
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def calculate_lse_metrics(
|
|
104
|
+
pipeline: SyncNetPipeline,
|
|
105
|
+
video_path: str,
|
|
106
|
+
audio_path: Optional[str] = None,
|
|
107
|
+
cache_dir: Optional[str] = None
|
|
108
|
+
) -> Tuple[float, float, str]:
|
|
109
|
+
"""
|
|
110
|
+
Calculate LSE-C and LSE-D metrics for a video.
|
|
111
|
+
|
|
112
|
+
Args:
|
|
113
|
+
pipeline: Initialized SyncNetPipeline instance
|
|
114
|
+
video_path: Path to input video file
|
|
115
|
+
audio_path: Path to audio file (None for auto-extraction)
|
|
116
|
+
cache_dir: Directory for temporary files
|
|
117
|
+
|
|
118
|
+
Returns:
|
|
119
|
+
Tuple of (lse_c, lse_d, quality_assessment)
|
|
120
|
+
|
|
121
|
+
Raises:
|
|
122
|
+
RuntimeError: If metric calculation fails
|
|
123
|
+
"""
|
|
124
|
+
try:
|
|
125
|
+
results = safe_syncnet_inference(pipeline, video_path, audio_path, cache_dir)
|
|
126
|
+
offsets, confs, dists, max_conf, min_dist, s3fd_json, has_face = results
|
|
127
|
+
|
|
128
|
+
if not has_face:
|
|
129
|
+
return 0.0, float('inf'), "NO_FACE"
|
|
130
|
+
|
|
131
|
+
if not confs or not dists:
|
|
132
|
+
return 0.0, float('inf'), "NO_TRACKS"
|
|
133
|
+
|
|
134
|
+
# LSE-C is the maximum confidence across all tracks
|
|
135
|
+
lse_c = max_conf
|
|
136
|
+
|
|
137
|
+
# LSE-D is the minimum distance across all tracks
|
|
138
|
+
lse_d = min_dist
|
|
139
|
+
|
|
140
|
+
# Quality assessment based on typical thresholds
|
|
141
|
+
if lse_c > 3.5 and lse_d < 7.0:
|
|
142
|
+
quality = "GOOD"
|
|
143
|
+
elif lse_c > 2.0 and lse_d < 10.0:
|
|
144
|
+
quality = "FAIR"
|
|
145
|
+
else:
|
|
146
|
+
quality = "POOR"
|
|
147
|
+
|
|
148
|
+
logging.info(f"LSE Metrics - C: {lse_c:.3f}, D: {lse_d:.3f}, Quality: {quality}")
|
|
149
|
+
|
|
150
|
+
return lse_c, lse_d, quality
|
|
151
|
+
|
|
152
|
+
except Exception as e:
|
|
153
|
+
logging.error(f"LSE metric calculation failed: {str(e)}")
|
|
154
|
+
raise RuntimeError(f"LSE metric calculation failed: {str(e)}") from e
|
|
@@ -215,11 +215,34 @@ class SyncNetPipeline:
|
|
|
215
215
|
os.remove(tmp_avi)
|
|
216
216
|
return final_avi
|
|
217
217
|
|
|
218
|
+
# ---------------------------- audio extraction helper ----------------- #
|
|
219
|
+
def _extract_audio_from_video(self, video_path: str, output_path: str) -> None:
|
|
220
|
+
"""Extract audio from video file using ffmpeg."""
|
|
221
|
+
cfg = self.cfg
|
|
222
|
+
ffmpeg_bin = cfg.ffmpeg_bin if cfg.ffmpeg_bin is not None else "ffmpeg"
|
|
223
|
+
|
|
224
|
+
cmd = [
|
|
225
|
+
ffmpeg_bin, "-y", "-i", str(video_path),
|
|
226
|
+
"-ac", "1", "-ar", str(cfg.audio_sample_rate),
|
|
227
|
+
"-acodec", "pcm_s16le", "-f", "wav",
|
|
228
|
+
str(output_path)
|
|
229
|
+
]
|
|
230
|
+
|
|
231
|
+
try:
|
|
232
|
+
result = subprocess.run(cmd, capture_output=True, text=True, check=True)
|
|
233
|
+
logging.info(f"Successfully extracted audio from {video_path} to {output_path}")
|
|
234
|
+
except subprocess.CalledProcessError as e:
|
|
235
|
+
logging.error(f"FFmpeg audio extraction failed: {e.stderr}")
|
|
236
|
+
raise RuntimeError(f"FFmpeg audio extraction failed: {e.stderr}")
|
|
237
|
+
except FileNotFoundError:
|
|
238
|
+
logging.error(f"FFmpeg not found at: {ffmpeg_bin}")
|
|
239
|
+
raise RuntimeError(f"FFmpeg not found. Please ensure ffmpeg is installed and in PATH.")
|
|
240
|
+
|
|
218
241
|
# ---------------------------- inference -------------------------------- #
|
|
219
242
|
def inference(
|
|
220
243
|
self,
|
|
221
|
-
video_path: str,
|
|
222
|
-
audio_path: str,
|
|
244
|
+
video_path: str,
|
|
245
|
+
audio_path: Optional[str] = None, # Now supports None for auto-extraction
|
|
223
246
|
*,
|
|
224
247
|
cache_dir: Optional[str] = None,
|
|
225
248
|
) -> Tuple[List[int], List[float], List[float], float, float, str, bool]:
|
|
@@ -229,34 +252,59 @@ class SyncNetPipeline:
|
|
|
229
252
|
work.mkdir(parents=True, exist_ok=True)
|
|
230
253
|
|
|
231
254
|
try:
|
|
255
|
+
# Handle audio_path=None case - extract audio from video
|
|
256
|
+
if audio_path is None:
|
|
257
|
+
logging.info("audio_path is None, extracting audio from video")
|
|
258
|
+
extracted_audio_path = work / "extracted_audio.wav"
|
|
259
|
+
self._extract_audio_from_video(video_path, str(extracted_audio_path))
|
|
260
|
+
actual_audio_path = str(extracted_audio_path)
|
|
261
|
+
else:
|
|
262
|
+
actual_audio_path = audio_path
|
|
263
|
+
logging.info(f"Using provided audio path: {actual_audio_path}")
|
|
264
|
+
|
|
232
265
|
# 1) Convert video to constant-fps AVI
|
|
233
266
|
avi = work / "video.avi"
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
267
|
+
try:
|
|
268
|
+
(
|
|
269
|
+
ffmpeg.input(video_path)
|
|
270
|
+
.output(str(avi), **{"q:v": 2}, r=cfg.frame_rate, **{"async": 1})
|
|
271
|
+
.overwrite_output()
|
|
272
|
+
.run()
|
|
273
|
+
)
|
|
274
|
+
except ffmpeg.Error as e:
|
|
275
|
+
logging.error(f"FFmpeg video conversion failed: {e}")
|
|
276
|
+
raise RuntimeError(f"FFmpeg video conversion failed: {e}")
|
|
240
277
|
|
|
241
278
|
# 2) Extract frames
|
|
242
279
|
frames_dir = work / "frames"
|
|
243
280
|
frames_dir.mkdir(exist_ok=True)
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
281
|
+
try:
|
|
282
|
+
(
|
|
283
|
+
ffmpeg.input(str(avi))
|
|
284
|
+
.output(str(frames_dir / "%06d.jpg"), **{"q:v": 2}, f="image2", threads=1)
|
|
285
|
+
.overwrite_output()
|
|
286
|
+
.run()
|
|
287
|
+
)
|
|
288
|
+
except ffmpeg.Error as e:
|
|
289
|
+
logging.error(f"FFmpeg frame extraction failed: {e}")
|
|
290
|
+
raise RuntimeError(f"FFmpeg frame extraction failed: {e}")
|
|
291
|
+
|
|
250
292
|
frames = sorted(glob(str(frames_dir / "*.jpg")))
|
|
293
|
+
if not frames:
|
|
294
|
+
raise RuntimeError("No frames were extracted from the video")
|
|
251
295
|
|
|
252
296
|
# 3) Resample speech
|
|
253
297
|
audio_wav = work / "speech.wav"
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
298
|
+
try:
|
|
299
|
+
(
|
|
300
|
+
ffmpeg.input(actual_audio_path)
|
|
301
|
+
.output(str(audio_wav), ac=1, ar=cfg.audio_sample_rate, format="wav")
|
|
302
|
+
.overwrite_output()
|
|
303
|
+
.run()
|
|
304
|
+
)
|
|
305
|
+
except ffmpeg.Error as e:
|
|
306
|
+
logging.error(f"FFmpeg audio resampling failed: {e}")
|
|
307
|
+
raise RuntimeError(f"FFmpeg audio resampling failed: {e}")
|
|
260
308
|
|
|
261
309
|
# 4) Face detection
|
|
262
310
|
detections = []
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Test script for comprehensive error handling in SyncNet v0.2.2."""
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
import tempfile
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
import sys
|
|
8
|
+
import os
|
|
9
|
+
sys.path.insert(0, os.path.dirname(__file__))
|
|
10
|
+
|
|
11
|
+
from syncnet_pipeline import SyncNetPipeline
|
|
12
|
+
|
|
13
|
+
# Import safe utils with fallback
|
|
14
|
+
try:
|
|
15
|
+
from safe_syncnet_utils import safe_syncnet_inference, calculate_lse_metrics
|
|
16
|
+
except ImportError:
|
|
17
|
+
# Fallback implementation for testing
|
|
18
|
+
def safe_syncnet_inference(pipeline, video_path, audio_path=None, cache_dir=None):
|
|
19
|
+
return pipeline.inference(video_path=video_path, audio_path=audio_path, cache_dir=cache_dir)
|
|
20
|
+
|
|
21
|
+
def calculate_lse_metrics(pipeline, video_path, audio_path=None, cache_dir=None):
|
|
22
|
+
results = pipeline.inference(video_path=video_path, audio_path=audio_path, cache_dir=cache_dir)
|
|
23
|
+
offsets, confs, dists, max_conf, min_dist, _, has_face = results
|
|
24
|
+
|
|
25
|
+
if not has_face or not confs:
|
|
26
|
+
return 0.0, float('inf'), "NO_FACE"
|
|
27
|
+
|
|
28
|
+
quality = "GOOD" if max_conf > 3.5 and min_dist < 7.0 else "FAIR" if max_conf > 2.0 else "POOR"
|
|
29
|
+
return max_conf, min_dist, quality
|
|
30
|
+
|
|
31
|
+
# Configure logging
|
|
32
|
+
logging.basicConfig(
|
|
33
|
+
level=logging.INFO,
|
|
34
|
+
format="%(asctime)s [%(levelname)s] %(message)s"
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
def test_audio_none_handling():
|
|
38
|
+
"""Test automatic audio extraction when audio_path=None."""
|
|
39
|
+
print("\n=== Testing audio_path=None handling ===")
|
|
40
|
+
|
|
41
|
+
# Initialize pipeline
|
|
42
|
+
pipeline = SyncNetPipeline(
|
|
43
|
+
{'s3fd_weights': '../weights/sfd_face.pth', 'syncnet_weights': '../weights/syncnet_v2.model'},
|
|
44
|
+
device='cuda'
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
# Test with audio_path=None (should automatically extract audio)
|
|
48
|
+
video_path = '../example/pair_0000_lipsynced.mp4'
|
|
49
|
+
|
|
50
|
+
try:
|
|
51
|
+
print(f"Testing inference with audio_path=None for: {video_path}")
|
|
52
|
+
results = pipeline.inference(
|
|
53
|
+
video_path=video_path,
|
|
54
|
+
audio_path=None, # This should now work!
|
|
55
|
+
cache_dir='../example/cache_test_none'
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
offsets, confs, dists, max_conf, min_dist, _, has_face = results
|
|
59
|
+
|
|
60
|
+
if has_face:
|
|
61
|
+
print(f"✅ SUCCESS: Auto audio extraction worked!")
|
|
62
|
+
print(f" LSE-C: {max_conf:.3f}")
|
|
63
|
+
print(f" LSE-D: {min_dist:.3f}")
|
|
64
|
+
else:
|
|
65
|
+
print("⚠️ No face detected, but no error occurred")
|
|
66
|
+
|
|
67
|
+
except Exception as e:
|
|
68
|
+
print(f"❌ FAILED: {str(e)}")
|
|
69
|
+
|
|
70
|
+
def test_safe_wrapper_functions():
|
|
71
|
+
"""Test the safe wrapper functions."""
|
|
72
|
+
print("\n=== Testing Safe Wrapper Functions ===")
|
|
73
|
+
|
|
74
|
+
pipeline = SyncNetPipeline(
|
|
75
|
+
{'s3fd_weights': '../weights/sfd_face.pth', 'syncnet_weights': '../weights/syncnet_v2.model'},
|
|
76
|
+
device='cuda'
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
video_path = '../example/pair_0000_lipsynced.mp4'
|
|
80
|
+
|
|
81
|
+
try:
|
|
82
|
+
print("Testing safe_syncnet_inference...")
|
|
83
|
+
results = safe_syncnet_inference(
|
|
84
|
+
pipeline,
|
|
85
|
+
video_path,
|
|
86
|
+
audio_path=None,
|
|
87
|
+
cache_dir='../example/cache_test_safe'
|
|
88
|
+
)
|
|
89
|
+
print("✅ safe_syncnet_inference: SUCCESS")
|
|
90
|
+
|
|
91
|
+
print("Testing calculate_lse_metrics...")
|
|
92
|
+
lse_c, lse_d, quality = calculate_lse_metrics(
|
|
93
|
+
pipeline,
|
|
94
|
+
video_path,
|
|
95
|
+
audio_path=None,
|
|
96
|
+
cache_dir='../example/cache_test_metrics'
|
|
97
|
+
)
|
|
98
|
+
print(f"✅ calculate_lse_metrics: SUCCESS")
|
|
99
|
+
print(f" LSE-C: {lse_c:.3f}")
|
|
100
|
+
print(f" LSE-D: {lse_d:.3f}")
|
|
101
|
+
print(f" Quality: {quality}")
|
|
102
|
+
|
|
103
|
+
except Exception as e:
|
|
104
|
+
print(f"❌ Safe wrapper test FAILED: {str(e)}")
|
|
105
|
+
|
|
106
|
+
def test_error_cases():
|
|
107
|
+
"""Test various error cases."""
|
|
108
|
+
print("\n=== Testing Error Cases ===")
|
|
109
|
+
|
|
110
|
+
pipeline = SyncNetPipeline(
|
|
111
|
+
{'s3fd_weights': '../weights/sfd_face.pth', 'syncnet_weights': '../weights/syncnet_v2.model'},
|
|
112
|
+
device='cuda'
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
# Test with non-existent file
|
|
116
|
+
try:
|
|
117
|
+
results = safe_syncnet_inference(
|
|
118
|
+
pipeline,
|
|
119
|
+
"non_existent_video.mp4",
|
|
120
|
+
audio_path=None
|
|
121
|
+
)
|
|
122
|
+
print("❌ Should have failed for non-existent file")
|
|
123
|
+
except FileNotFoundError:
|
|
124
|
+
print("✅ Correctly caught FileNotFoundError for non-existent video")
|
|
125
|
+
except Exception as e:
|
|
126
|
+
print(f"⚠️ Unexpected error: {str(e)}")
|
|
127
|
+
|
|
128
|
+
def main():
|
|
129
|
+
"""Run all tests."""
|
|
130
|
+
print("SyncNet v0.2.2 Error Handling Test Suite")
|
|
131
|
+
print("=" * 50)
|
|
132
|
+
|
|
133
|
+
test_audio_none_handling()
|
|
134
|
+
test_safe_wrapper_functions()
|
|
135
|
+
test_error_cases()
|
|
136
|
+
|
|
137
|
+
print("\n" + "=" * 50)
|
|
138
|
+
print("Test suite completed!")
|
|
139
|
+
|
|
140
|
+
if __name__ == "__main__":
|
|
141
|
+
main()
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: syncnet-python
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: SyncNet: Audio-visual synchronization detection using deep learning. Updated version of https://github.com/joonson/syncnet_python for modern Python versions.
|
|
5
5
|
Author: SyncNet Python Contributors
|
|
6
6
|
Maintainer: SyncNet Python Contributors
|
|
@@ -49,7 +49,6 @@ Dynamic: license-file
|
|
|
49
49
|
|
|
50
50
|
[](https://badge.fury.io/py/syncnet-python)
|
|
51
51
|
[](https://pypi.org/project/syncnet-python/)
|
|
52
|
-
[](https://pepy.tech/project/syncnet-python)
|
|
53
52
|
[](https://opensource.org/licenses/MIT)
|
|
54
53
|
|
|
55
54
|
Audio-visual synchronization detection using deep learning with modern Python architecture.
|
|
@@ -43,7 +43,9 @@ syncnet_python/cli.py
|
|
|
43
43
|
syncnet_python/run_syncnet_pipeline_on_1example.py
|
|
44
44
|
syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py
|
|
45
45
|
syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py
|
|
46
|
+
syncnet_python/safe_syncnet_utils.py
|
|
46
47
|
syncnet_python/syncnet_pipeline.py
|
|
48
|
+
syncnet_python/test_error_handling.py
|
|
47
49
|
syncnet_python/test_lse_metrics.py
|
|
48
50
|
syncnet_python.egg-info/PKG-INFO
|
|
49
51
|
syncnet_python.egg-info/SOURCES.txt
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{syncnet_python-0.2.1 → syncnet_python-0.2.2}/syncnet_python/run_syncnet_pipeline_on_1example.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|