syncnet-python 0.2.0__py3-none-any.whl → 0.2.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- syncnet_python/__init__.py +12 -1
- syncnet_python/safe_syncnet_utils.py +154 -0
- syncnet_python/syncnet_pipeline.py +117 -32
- syncnet_python/test_error_handling.py +141 -0
- syncnet_python/test_lse_metrics.py +95 -0
- {syncnet_python-0.2.0.dist-info → syncnet_python-0.2.2.dist-info}/METADATA +5 -1
- {syncnet_python-0.2.0.dist-info → syncnet_python-0.2.2.dist-info}/RECORD +11 -8
- {syncnet_python-0.2.0.dist-info → syncnet_python-0.2.2.dist-info}/WHEEL +0 -0
- {syncnet_python-0.2.0.dist-info → syncnet_python-0.2.2.dist-info}/entry_points.txt +0 -0
- {syncnet_python-0.2.0.dist-info → syncnet_python-0.2.2.dist-info}/licenses/LICENSE +0 -0
- {syncnet_python-0.2.0.dist-info → syncnet_python-0.2.2.dist-info}/top_level.txt +0 -0
syncnet_python/__init__.py
CHANGED
|
@@ -4,23 +4,34 @@ This package provides a PyTorch implementation of SyncNet for detecting
|
|
|
4
4
|
synchronization between audio and video in multimedia content.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
__version__ = "0.
|
|
7
|
+
__version__ = "0.2.2"
|
|
8
8
|
|
|
9
9
|
# Import main components
|
|
10
10
|
try:
|
|
11
11
|
from .syncnet_pipeline import SyncNetPipeline
|
|
12
12
|
from .SyncNetModel import S as SyncNetModel
|
|
13
13
|
from .SyncNetInstance import SyncNetInstance
|
|
14
|
+
from .safe_syncnet_utils import (
|
|
15
|
+
safe_syncnet_inference,
|
|
16
|
+
extract_audio_from_video,
|
|
17
|
+
calculate_lse_metrics
|
|
18
|
+
)
|
|
14
19
|
except ImportError:
|
|
15
20
|
# Fallback for development
|
|
16
21
|
SyncNetPipeline = None
|
|
17
22
|
SyncNetModel = None
|
|
18
23
|
SyncNetInstance = None
|
|
24
|
+
safe_syncnet_inference = None
|
|
25
|
+
extract_audio_from_video = None
|
|
26
|
+
calculate_lse_metrics = None
|
|
19
27
|
|
|
20
28
|
__all__ = [
|
|
21
29
|
"SyncNetPipeline",
|
|
22
30
|
"SyncNetModel",
|
|
23
31
|
"SyncNetInstance",
|
|
32
|
+
"safe_syncnet_inference",
|
|
33
|
+
"extract_audio_from_video",
|
|
34
|
+
"calculate_lse_metrics",
|
|
24
35
|
"__version__"
|
|
25
36
|
]
|
|
26
37
|
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""Safe SyncNet utilities with comprehensive error handling."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import tempfile
|
|
5
|
+
import os
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Optional, Tuple, List
|
|
8
|
+
try:
|
|
9
|
+
from .syncnet_pipeline import SyncNetPipeline
|
|
10
|
+
except ImportError:
|
|
11
|
+
from syncnet_pipeline import SyncNetPipeline
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def safe_syncnet_inference(
|
|
15
|
+
pipeline: SyncNetPipeline,
|
|
16
|
+
video_path: str,
|
|
17
|
+
audio_path: Optional[str] = None,
|
|
18
|
+
cache_dir: Optional[str] = None
|
|
19
|
+
) -> Tuple[List[int], List[float], List[float], float, float, str, bool]:
|
|
20
|
+
"""
|
|
21
|
+
Safe SyncNet inference with automatic audio extraction if needed.
|
|
22
|
+
|
|
23
|
+
This function provides a wrapper around SyncNetPipeline.inference() that:
|
|
24
|
+
1. Handles audio_path=None by automatically extracting audio from video
|
|
25
|
+
2. Provides comprehensive error handling and logging
|
|
26
|
+
3. Ensures proper cleanup of temporary files
|
|
27
|
+
|
|
28
|
+
Args:
|
|
29
|
+
pipeline: Initialized SyncNetPipeline instance
|
|
30
|
+
video_path: Path to input video file
|
|
31
|
+
audio_path: Path to audio file (None for auto-extraction from video)
|
|
32
|
+
cache_dir: Directory for temporary files (None for auto-cleanup)
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
Tuple of (offsets, confidences, distances, max_conf, min_dist, s3fd_json, has_face)
|
|
36
|
+
|
|
37
|
+
Raises:
|
|
38
|
+
RuntimeError: If processing fails at any stage
|
|
39
|
+
FileNotFoundError: If video file doesn't exist
|
|
40
|
+
ValueError: If video format is not supported
|
|
41
|
+
"""
|
|
42
|
+
if not os.path.exists(video_path):
|
|
43
|
+
raise FileNotFoundError(f"Video file not found: {video_path}")
|
|
44
|
+
|
|
45
|
+
if audio_path and not os.path.exists(audio_path):
|
|
46
|
+
raise FileNotFoundError(f"Audio file not found: {audio_path}")
|
|
47
|
+
|
|
48
|
+
logging.info(f"Starting safe SyncNet inference for video: {video_path}")
|
|
49
|
+
|
|
50
|
+
try:
|
|
51
|
+
# Use the updated inference method that handles audio_path=None
|
|
52
|
+
results = pipeline.inference(
|
|
53
|
+
video_path=video_path,
|
|
54
|
+
audio_path=audio_path,
|
|
55
|
+
cache_dir=cache_dir
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
offsets, confs, dists, max_conf, min_dist, s3fd_json, has_face = results
|
|
59
|
+
|
|
60
|
+
# Validate results
|
|
61
|
+
if not has_face:
|
|
62
|
+
logging.warning("No faces detected in the video")
|
|
63
|
+
elif not offsets:
|
|
64
|
+
logging.warning("No valid face tracks found")
|
|
65
|
+
else:
|
|
66
|
+
logging.info(f"Successfully processed {len(offsets)} face tracks")
|
|
67
|
+
logging.info(f"LSE-C (max confidence): {max_conf:.3f}")
|
|
68
|
+
logging.info(f"LSE-D (min distance): {min_dist:.3f}")
|
|
69
|
+
|
|
70
|
+
return results
|
|
71
|
+
|
|
72
|
+
except Exception as e:
|
|
73
|
+
logging.error(f"SyncNet inference failed: {str(e)}")
|
|
74
|
+
raise RuntimeError(f"SyncNet processing failed: {str(e)}") from e
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def extract_audio_from_video(video_path: str, audio_path: str) -> None:
|
|
78
|
+
"""
|
|
79
|
+
Extract audio from video file using ffmpeg.
|
|
80
|
+
|
|
81
|
+
Args:
|
|
82
|
+
video_path: Path to input video file
|
|
83
|
+
audio_path: Path for output audio file
|
|
84
|
+
|
|
85
|
+
Raises:
|
|
86
|
+
RuntimeError: If audio extraction fails
|
|
87
|
+
FileNotFoundError: If video file doesn't exist
|
|
88
|
+
"""
|
|
89
|
+
if not os.path.exists(video_path):
|
|
90
|
+
raise FileNotFoundError(f"Video file not found: {video_path}")
|
|
91
|
+
|
|
92
|
+
# Create a temporary pipeline instance just for audio extraction
|
|
93
|
+
pipeline = SyncNetPipeline()
|
|
94
|
+
|
|
95
|
+
try:
|
|
96
|
+
pipeline._extract_audio_from_video(video_path, audio_path)
|
|
97
|
+
logging.info(f"Successfully extracted audio from {video_path} to {audio_path}")
|
|
98
|
+
except Exception as e:
|
|
99
|
+
logging.error(f"Audio extraction failed: {str(e)}")
|
|
100
|
+
raise RuntimeError(f"Audio extraction failed: {str(e)}") from e
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def calculate_lse_metrics(
|
|
104
|
+
pipeline: SyncNetPipeline,
|
|
105
|
+
video_path: str,
|
|
106
|
+
audio_path: Optional[str] = None,
|
|
107
|
+
cache_dir: Optional[str] = None
|
|
108
|
+
) -> Tuple[float, float, str]:
|
|
109
|
+
"""
|
|
110
|
+
Calculate LSE-C and LSE-D metrics for a video.
|
|
111
|
+
|
|
112
|
+
Args:
|
|
113
|
+
pipeline: Initialized SyncNetPipeline instance
|
|
114
|
+
video_path: Path to input video file
|
|
115
|
+
audio_path: Path to audio file (None for auto-extraction)
|
|
116
|
+
cache_dir: Directory for temporary files
|
|
117
|
+
|
|
118
|
+
Returns:
|
|
119
|
+
Tuple of (lse_c, lse_d, quality_assessment)
|
|
120
|
+
|
|
121
|
+
Raises:
|
|
122
|
+
RuntimeError: If metric calculation fails
|
|
123
|
+
"""
|
|
124
|
+
try:
|
|
125
|
+
results = safe_syncnet_inference(pipeline, video_path, audio_path, cache_dir)
|
|
126
|
+
offsets, confs, dists, max_conf, min_dist, s3fd_json, has_face = results
|
|
127
|
+
|
|
128
|
+
if not has_face:
|
|
129
|
+
return 0.0, float('inf'), "NO_FACE"
|
|
130
|
+
|
|
131
|
+
if not confs or not dists:
|
|
132
|
+
return 0.0, float('inf'), "NO_TRACKS"
|
|
133
|
+
|
|
134
|
+
# LSE-C is the maximum confidence across all tracks
|
|
135
|
+
lse_c = max_conf
|
|
136
|
+
|
|
137
|
+
# LSE-D is the minimum distance across all tracks
|
|
138
|
+
lse_d = min_dist
|
|
139
|
+
|
|
140
|
+
# Quality assessment based on typical thresholds
|
|
141
|
+
if lse_c > 3.5 and lse_d < 7.0:
|
|
142
|
+
quality = "GOOD"
|
|
143
|
+
elif lse_c > 2.0 and lse_d < 10.0:
|
|
144
|
+
quality = "FAIR"
|
|
145
|
+
else:
|
|
146
|
+
quality = "POOR"
|
|
147
|
+
|
|
148
|
+
logging.info(f"LSE Metrics - C: {lse_c:.3f}, D: {lse_d:.3f}, Quality: {quality}")
|
|
149
|
+
|
|
150
|
+
return lse_c, lse_d, quality
|
|
151
|
+
|
|
152
|
+
except Exception as e:
|
|
153
|
+
logging.error(f"LSE metric calculation failed: {str(e)}")
|
|
154
|
+
raise RuntimeError(f"LSE metric calculation failed: {str(e)}") from e
|
|
@@ -18,10 +18,17 @@ from scipy.interpolate import interp1d
|
|
|
18
18
|
from scenedetect import ContentDetector, SceneManager, StatsManager
|
|
19
19
|
from scenedetect.video_manager import VideoManager
|
|
20
20
|
|
|
21
|
-
|
|
22
|
-
from .detectors.s3fd
|
|
23
|
-
from .
|
|
24
|
-
from .
|
|
21
|
+
try:
|
|
22
|
+
from .detectors.s3fd import S3FD
|
|
23
|
+
from .detectors.s3fd.nets import S3FDNet
|
|
24
|
+
from .SyncNetInstance import SyncNetInstance
|
|
25
|
+
from .SyncNetModel import S
|
|
26
|
+
except ImportError:
|
|
27
|
+
# Fallback for direct script execution
|
|
28
|
+
from detectors.s3fd import S3FD
|
|
29
|
+
from detectors.s3fd.nets import S3FDNet
|
|
30
|
+
from SyncNetInstance import SyncNetInstance
|
|
31
|
+
from SyncNetModel import S
|
|
25
32
|
|
|
26
33
|
# ---------------------------------------------------------------------- #
|
|
27
34
|
# Configuration #
|
|
@@ -47,6 +54,11 @@ class PipelineConfig:
|
|
|
47
54
|
# Tools
|
|
48
55
|
ffmpeg_bin: str = "ffmpeg" # assumes ffmpeg in $PATH
|
|
49
56
|
audio_sample_rate: int = 16000 # resample rate for speech
|
|
57
|
+
|
|
58
|
+
def __post_init__(self):
|
|
59
|
+
"""Validate configuration after initialization."""
|
|
60
|
+
if self.ffmpeg_bin is None:
|
|
61
|
+
self.ffmpeg_bin = "ffmpeg"
|
|
50
62
|
|
|
51
63
|
@classmethod
|
|
52
64
|
def from_dict(cls, d: Dict[str, Any]):
|
|
@@ -165,24 +177,72 @@ class SyncNetPipeline:
|
|
|
165
177
|
slice_wav = f"{base}.wav"
|
|
166
178
|
ss = track["frame"][0] / cfg.frame_rate
|
|
167
179
|
to = (track["frame"][-1] + 1) / cfg.frame_rate
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
180
|
+
|
|
181
|
+
# Ensure ffmpeg_bin is not None
|
|
182
|
+
ffmpeg_bin = cfg.ffmpeg_bin if cfg.ffmpeg_bin is not None else "ffmpeg"
|
|
183
|
+
|
|
184
|
+
cmd = [
|
|
185
|
+
ffmpeg_bin, "-y", "-i", str(audio_wav),
|
|
186
|
+
"-ss", f"{ss:.3f}", "-to", f"{to:.3f}",
|
|
187
|
+
str(slice_wav)
|
|
188
|
+
]
|
|
189
|
+
|
|
190
|
+
try:
|
|
191
|
+
result = subprocess.run(cmd, capture_output=True, text=True, check=True)
|
|
192
|
+
except subprocess.CalledProcessError as e:
|
|
193
|
+
logging.error(f"FFmpeg audio slicing failed: {e.stderr}")
|
|
194
|
+
raise RuntimeError(f"FFmpeg audio slicing failed: {e.stderr}")
|
|
195
|
+
except FileNotFoundError:
|
|
196
|
+
logging.error(f"FFmpeg not found at: {ffmpeg_bin}")
|
|
197
|
+
raise RuntimeError(f"FFmpeg not found. Please ensure ffmpeg is installed and in PATH.")
|
|
172
198
|
|
|
173
199
|
final_avi = f"{base}.avi"
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
200
|
+
|
|
201
|
+
cmd = [
|
|
202
|
+
ffmpeg_bin, "-y", "-i", str(tmp_avi), "-i", str(slice_wav),
|
|
203
|
+
"-c:v", "copy", "-c:a", "copy", str(final_avi)
|
|
204
|
+
]
|
|
205
|
+
|
|
206
|
+
try:
|
|
207
|
+
result = subprocess.run(cmd, capture_output=True, text=True, check=True)
|
|
208
|
+
except subprocess.CalledProcessError as e:
|
|
209
|
+
logging.error(f"FFmpeg video/audio merge failed: {e.stderr}")
|
|
210
|
+
raise RuntimeError(f"FFmpeg video/audio merge failed: {e.stderr}")
|
|
211
|
+
except FileNotFoundError:
|
|
212
|
+
logging.error(f"FFmpeg not found at: {ffmpeg_bin}")
|
|
213
|
+
raise RuntimeError(f"FFmpeg not found. Please ensure ffmpeg is installed and in PATH.")
|
|
214
|
+
|
|
178
215
|
os.remove(tmp_avi)
|
|
179
216
|
return final_avi
|
|
180
217
|
|
|
218
|
+
# ---------------------------- audio extraction helper ----------------- #
|
|
219
|
+
def _extract_audio_from_video(self, video_path: str, output_path: str) -> None:
|
|
220
|
+
"""Extract audio from video file using ffmpeg."""
|
|
221
|
+
cfg = self.cfg
|
|
222
|
+
ffmpeg_bin = cfg.ffmpeg_bin if cfg.ffmpeg_bin is not None else "ffmpeg"
|
|
223
|
+
|
|
224
|
+
cmd = [
|
|
225
|
+
ffmpeg_bin, "-y", "-i", str(video_path),
|
|
226
|
+
"-ac", "1", "-ar", str(cfg.audio_sample_rate),
|
|
227
|
+
"-acodec", "pcm_s16le", "-f", "wav",
|
|
228
|
+
str(output_path)
|
|
229
|
+
]
|
|
230
|
+
|
|
231
|
+
try:
|
|
232
|
+
result = subprocess.run(cmd, capture_output=True, text=True, check=True)
|
|
233
|
+
logging.info(f"Successfully extracted audio from {video_path} to {output_path}")
|
|
234
|
+
except subprocess.CalledProcessError as e:
|
|
235
|
+
logging.error(f"FFmpeg audio extraction failed: {e.stderr}")
|
|
236
|
+
raise RuntimeError(f"FFmpeg audio extraction failed: {e.stderr}")
|
|
237
|
+
except FileNotFoundError:
|
|
238
|
+
logging.error(f"FFmpeg not found at: {ffmpeg_bin}")
|
|
239
|
+
raise RuntimeError(f"FFmpeg not found. Please ensure ffmpeg is installed and in PATH.")
|
|
240
|
+
|
|
181
241
|
# ---------------------------- inference -------------------------------- #
|
|
182
242
|
def inference(
|
|
183
243
|
self,
|
|
184
|
-
video_path: str,
|
|
185
|
-
audio_path: str,
|
|
244
|
+
video_path: str,
|
|
245
|
+
audio_path: Optional[str] = None, # Now supports None for auto-extraction
|
|
186
246
|
*,
|
|
187
247
|
cache_dir: Optional[str] = None,
|
|
188
248
|
) -> Tuple[List[int], List[float], List[float], float, float, str, bool]:
|
|
@@ -192,34 +252,59 @@ class SyncNetPipeline:
|
|
|
192
252
|
work.mkdir(parents=True, exist_ok=True)
|
|
193
253
|
|
|
194
254
|
try:
|
|
255
|
+
# Handle audio_path=None case - extract audio from video
|
|
256
|
+
if audio_path is None:
|
|
257
|
+
logging.info("audio_path is None, extracting audio from video")
|
|
258
|
+
extracted_audio_path = work / "extracted_audio.wav"
|
|
259
|
+
self._extract_audio_from_video(video_path, str(extracted_audio_path))
|
|
260
|
+
actual_audio_path = str(extracted_audio_path)
|
|
261
|
+
else:
|
|
262
|
+
actual_audio_path = audio_path
|
|
263
|
+
logging.info(f"Using provided audio path: {actual_audio_path}")
|
|
264
|
+
|
|
195
265
|
# 1) Convert video to constant-fps AVI
|
|
196
266
|
avi = work / "video.avi"
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
267
|
+
try:
|
|
268
|
+
(
|
|
269
|
+
ffmpeg.input(video_path)
|
|
270
|
+
.output(str(avi), **{"q:v": 2}, r=cfg.frame_rate, **{"async": 1})
|
|
271
|
+
.overwrite_output()
|
|
272
|
+
.run()
|
|
273
|
+
)
|
|
274
|
+
except ffmpeg.Error as e:
|
|
275
|
+
logging.error(f"FFmpeg video conversion failed: {e}")
|
|
276
|
+
raise RuntimeError(f"FFmpeg video conversion failed: {e}")
|
|
203
277
|
|
|
204
278
|
# 2) Extract frames
|
|
205
279
|
frames_dir = work / "frames"
|
|
206
280
|
frames_dir.mkdir(exist_ok=True)
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
281
|
+
try:
|
|
282
|
+
(
|
|
283
|
+
ffmpeg.input(str(avi))
|
|
284
|
+
.output(str(frames_dir / "%06d.jpg"), **{"q:v": 2}, f="image2", threads=1)
|
|
285
|
+
.overwrite_output()
|
|
286
|
+
.run()
|
|
287
|
+
)
|
|
288
|
+
except ffmpeg.Error as e:
|
|
289
|
+
logging.error(f"FFmpeg frame extraction failed: {e}")
|
|
290
|
+
raise RuntimeError(f"FFmpeg frame extraction failed: {e}")
|
|
291
|
+
|
|
213
292
|
frames = sorted(glob(str(frames_dir / "*.jpg")))
|
|
293
|
+
if not frames:
|
|
294
|
+
raise RuntimeError("No frames were extracted from the video")
|
|
214
295
|
|
|
215
296
|
# 3) Resample speech
|
|
216
297
|
audio_wav = work / "speech.wav"
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
298
|
+
try:
|
|
299
|
+
(
|
|
300
|
+
ffmpeg.input(actual_audio_path)
|
|
301
|
+
.output(str(audio_wav), ac=1, ar=cfg.audio_sample_rate, format="wav")
|
|
302
|
+
.overwrite_output()
|
|
303
|
+
.run()
|
|
304
|
+
)
|
|
305
|
+
except ffmpeg.Error as e:
|
|
306
|
+
logging.error(f"FFmpeg audio resampling failed: {e}")
|
|
307
|
+
raise RuntimeError(f"FFmpeg audio resampling failed: {e}")
|
|
223
308
|
|
|
224
309
|
# 4) Face detection
|
|
225
310
|
detections = []
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Test script for comprehensive error handling in SyncNet v0.2.2."""
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
import tempfile
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
import sys
|
|
8
|
+
import os
|
|
9
|
+
sys.path.insert(0, os.path.dirname(__file__))
|
|
10
|
+
|
|
11
|
+
from syncnet_pipeline import SyncNetPipeline
|
|
12
|
+
|
|
13
|
+
# Import safe utils with fallback
|
|
14
|
+
try:
|
|
15
|
+
from safe_syncnet_utils import safe_syncnet_inference, calculate_lse_metrics
|
|
16
|
+
except ImportError:
|
|
17
|
+
# Fallback implementation for testing
|
|
18
|
+
def safe_syncnet_inference(pipeline, video_path, audio_path=None, cache_dir=None):
|
|
19
|
+
return pipeline.inference(video_path=video_path, audio_path=audio_path, cache_dir=cache_dir)
|
|
20
|
+
|
|
21
|
+
def calculate_lse_metrics(pipeline, video_path, audio_path=None, cache_dir=None):
|
|
22
|
+
results = pipeline.inference(video_path=video_path, audio_path=audio_path, cache_dir=cache_dir)
|
|
23
|
+
offsets, confs, dists, max_conf, min_dist, _, has_face = results
|
|
24
|
+
|
|
25
|
+
if not has_face or not confs:
|
|
26
|
+
return 0.0, float('inf'), "NO_FACE"
|
|
27
|
+
|
|
28
|
+
quality = "GOOD" if max_conf > 3.5 and min_dist < 7.0 else "FAIR" if max_conf > 2.0 else "POOR"
|
|
29
|
+
return max_conf, min_dist, quality
|
|
30
|
+
|
|
31
|
+
# Configure logging
|
|
32
|
+
logging.basicConfig(
|
|
33
|
+
level=logging.INFO,
|
|
34
|
+
format="%(asctime)s [%(levelname)s] %(message)s"
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
def test_audio_none_handling():
|
|
38
|
+
"""Test automatic audio extraction when audio_path=None."""
|
|
39
|
+
print("\n=== Testing audio_path=None handling ===")
|
|
40
|
+
|
|
41
|
+
# Initialize pipeline
|
|
42
|
+
pipeline = SyncNetPipeline(
|
|
43
|
+
{'s3fd_weights': '../weights/sfd_face.pth', 'syncnet_weights': '../weights/syncnet_v2.model'},
|
|
44
|
+
device='cuda'
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
# Test with audio_path=None (should automatically extract audio)
|
|
48
|
+
video_path = '../example/pair_0000_lipsynced.mp4'
|
|
49
|
+
|
|
50
|
+
try:
|
|
51
|
+
print(f"Testing inference with audio_path=None for: {video_path}")
|
|
52
|
+
results = pipeline.inference(
|
|
53
|
+
video_path=video_path,
|
|
54
|
+
audio_path=None, # This should now work!
|
|
55
|
+
cache_dir='../example/cache_test_none'
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
offsets, confs, dists, max_conf, min_dist, _, has_face = results
|
|
59
|
+
|
|
60
|
+
if has_face:
|
|
61
|
+
print(f"✅ SUCCESS: Auto audio extraction worked!")
|
|
62
|
+
print(f" LSE-C: {max_conf:.3f}")
|
|
63
|
+
print(f" LSE-D: {min_dist:.3f}")
|
|
64
|
+
else:
|
|
65
|
+
print("⚠️ No face detected, but no error occurred")
|
|
66
|
+
|
|
67
|
+
except Exception as e:
|
|
68
|
+
print(f"❌ FAILED: {str(e)}")
|
|
69
|
+
|
|
70
|
+
def test_safe_wrapper_functions():
|
|
71
|
+
"""Test the safe wrapper functions."""
|
|
72
|
+
print("\n=== Testing Safe Wrapper Functions ===")
|
|
73
|
+
|
|
74
|
+
pipeline = SyncNetPipeline(
|
|
75
|
+
{'s3fd_weights': '../weights/sfd_face.pth', 'syncnet_weights': '../weights/syncnet_v2.model'},
|
|
76
|
+
device='cuda'
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
video_path = '../example/pair_0000_lipsynced.mp4'
|
|
80
|
+
|
|
81
|
+
try:
|
|
82
|
+
print("Testing safe_syncnet_inference...")
|
|
83
|
+
results = safe_syncnet_inference(
|
|
84
|
+
pipeline,
|
|
85
|
+
video_path,
|
|
86
|
+
audio_path=None,
|
|
87
|
+
cache_dir='../example/cache_test_safe'
|
|
88
|
+
)
|
|
89
|
+
print("✅ safe_syncnet_inference: SUCCESS")
|
|
90
|
+
|
|
91
|
+
print("Testing calculate_lse_metrics...")
|
|
92
|
+
lse_c, lse_d, quality = calculate_lse_metrics(
|
|
93
|
+
pipeline,
|
|
94
|
+
video_path,
|
|
95
|
+
audio_path=None,
|
|
96
|
+
cache_dir='../example/cache_test_metrics'
|
|
97
|
+
)
|
|
98
|
+
print(f"✅ calculate_lse_metrics: SUCCESS")
|
|
99
|
+
print(f" LSE-C: {lse_c:.3f}")
|
|
100
|
+
print(f" LSE-D: {lse_d:.3f}")
|
|
101
|
+
print(f" Quality: {quality}")
|
|
102
|
+
|
|
103
|
+
except Exception as e:
|
|
104
|
+
print(f"❌ Safe wrapper test FAILED: {str(e)}")
|
|
105
|
+
|
|
106
|
+
def test_error_cases():
|
|
107
|
+
"""Test various error cases."""
|
|
108
|
+
print("\n=== Testing Error Cases ===")
|
|
109
|
+
|
|
110
|
+
pipeline = SyncNetPipeline(
|
|
111
|
+
{'s3fd_weights': '../weights/sfd_face.pth', 'syncnet_weights': '../weights/syncnet_v2.model'},
|
|
112
|
+
device='cuda'
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
# Test with non-existent file
|
|
116
|
+
try:
|
|
117
|
+
results = safe_syncnet_inference(
|
|
118
|
+
pipeline,
|
|
119
|
+
"non_existent_video.mp4",
|
|
120
|
+
audio_path=None
|
|
121
|
+
)
|
|
122
|
+
print("❌ Should have failed for non-existent file")
|
|
123
|
+
except FileNotFoundError:
|
|
124
|
+
print("✅ Correctly caught FileNotFoundError for non-existent video")
|
|
125
|
+
except Exception as e:
|
|
126
|
+
print(f"⚠️ Unexpected error: {str(e)}")
|
|
127
|
+
|
|
128
|
+
def main():
|
|
129
|
+
"""Run all tests."""
|
|
130
|
+
print("SyncNet v0.2.2 Error Handling Test Suite")
|
|
131
|
+
print("=" * 50)
|
|
132
|
+
|
|
133
|
+
test_audio_none_handling()
|
|
134
|
+
test_safe_wrapper_functions()
|
|
135
|
+
test_error_cases()
|
|
136
|
+
|
|
137
|
+
print("\n" + "=" * 50)
|
|
138
|
+
print("Test suite completed!")
|
|
139
|
+
|
|
140
|
+
if __name__ == "__main__":
|
|
141
|
+
main()
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import sys
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from syncnet_pipeline import SyncNetPipeline
|
|
5
|
+
|
|
6
|
+
logging.basicConfig(
|
|
7
|
+
level=logging.INFO,
|
|
8
|
+
format="%(asctime)s [%(levelname)s] %(message)s"
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
def calculate_lse_metrics(video_path):
|
|
12
|
+
"""
|
|
13
|
+
Calculate LSE-C and LSE-D metrics for a video
|
|
14
|
+
|
|
15
|
+
LSE-C (Lip Sync Error - Confidence):
|
|
16
|
+
- Higher confidence values indicate better lip sync
|
|
17
|
+
- Threshold typically around 3.5-4.0 for good sync
|
|
18
|
+
|
|
19
|
+
LSE-D (Lip Sync Error - Distance):
|
|
20
|
+
- Lower distance values indicate better lip sync
|
|
21
|
+
- Threshold typically around 6.5-7.0 for good sync
|
|
22
|
+
"""
|
|
23
|
+
# Initialize pipeline
|
|
24
|
+
pipe = SyncNetPipeline(
|
|
25
|
+
{
|
|
26
|
+
"s3fd_weights": "../weights/sfd_face.pth",
|
|
27
|
+
"syncnet_weights": "../weights/syncnet_v2.model",
|
|
28
|
+
},
|
|
29
|
+
device="cuda", # or "cpu"
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
# Run inference
|
|
33
|
+
print(f"\n=== Processing {video_path} ===")
|
|
34
|
+
|
|
35
|
+
# For testing, we'll extract audio from the video itself
|
|
36
|
+
results = pipe.inference(
|
|
37
|
+
video_path=video_path,
|
|
38
|
+
audio_path=video_path, # Extract audio from same video
|
|
39
|
+
cache_dir="../example/cache_lse",
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
offsets, confs, dists, max_conf, min_dist, s3fd_json, has_face = results
|
|
43
|
+
|
|
44
|
+
if not has_face:
|
|
45
|
+
print(f"No face detected in {video_path}")
|
|
46
|
+
return None, None
|
|
47
|
+
|
|
48
|
+
# LSE-C is the maximum confidence across all tracks
|
|
49
|
+
lse_c = max_conf
|
|
50
|
+
|
|
51
|
+
# LSE-D is the minimum distance across all tracks
|
|
52
|
+
lse_d = min_dist
|
|
53
|
+
|
|
54
|
+
print(f"Number of face tracks: {len(offsets)}")
|
|
55
|
+
print(f"Track offsets: {offsets}")
|
|
56
|
+
print(f"Track confidences: {[f'{c:.3f}' for c in confs]}")
|
|
57
|
+
print(f"Track distances: {[f'{d:.3f}' for d in dists]}")
|
|
58
|
+
print(f"LSE-C (max confidence): {lse_c:.3f}")
|
|
59
|
+
print(f"LSE-D (min distance): {lse_d:.3f}")
|
|
60
|
+
|
|
61
|
+
# Interpretation
|
|
62
|
+
sync_quality = "GOOD" if lse_c > 3.5 and lse_d < 7.0 else "POOR"
|
|
63
|
+
print(f"Sync Quality: {sync_quality}")
|
|
64
|
+
|
|
65
|
+
return lse_c, lse_d
|
|
66
|
+
|
|
67
|
+
def main():
|
|
68
|
+
# Test files
|
|
69
|
+
test_files = [
|
|
70
|
+
"../example/pair_0000_lipsynced.mp4",
|
|
71
|
+
"../example/pair_0001_lipsynced.mp4"
|
|
72
|
+
]
|
|
73
|
+
|
|
74
|
+
results = {}
|
|
75
|
+
|
|
76
|
+
for video_path in test_files:
|
|
77
|
+
if Path(video_path).exists():
|
|
78
|
+
lse_c, lse_d = calculate_lse_metrics(video_path)
|
|
79
|
+
results[video_path] = (lse_c, lse_d)
|
|
80
|
+
else:
|
|
81
|
+
print(f"File not found: {video_path}")
|
|
82
|
+
|
|
83
|
+
# Summary
|
|
84
|
+
print("\n=== SUMMARY ===")
|
|
85
|
+
print(f"{'Video':<40} {'LSE-C':<10} {'LSE-D':<10} {'Quality':<10}")
|
|
86
|
+
print("-" * 70)
|
|
87
|
+
|
|
88
|
+
for video_path, (lse_c, lse_d) in results.items():
|
|
89
|
+
if lse_c is not None:
|
|
90
|
+
video_name = Path(video_path).name
|
|
91
|
+
quality = "GOOD" if lse_c > 3.5 and lse_d < 7.0 else "POOR"
|
|
92
|
+
print(f"{video_name:<40} {lse_c:<10.3f} {lse_d:<10.3f} {quality:<10}")
|
|
93
|
+
|
|
94
|
+
if __name__ == "__main__":
|
|
95
|
+
main()
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: syncnet-python
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: SyncNet: Audio-visual synchronization detection using deep learning. Updated version of https://github.com/joonson/syncnet_python for modern Python versions.
|
|
5
5
|
Author: SyncNet Python Contributors
|
|
6
6
|
Maintainer: SyncNet Python Contributors
|
|
@@ -47,6 +47,10 @@ Dynamic: license-file
|
|
|
47
47
|
|
|
48
48
|
# SyncNet Python
|
|
49
49
|
|
|
50
|
+
[](https://badge.fury.io/py/syncnet-python)
|
|
51
|
+
[](https://pypi.org/project/syncnet-python/)
|
|
52
|
+
[](https://opensource.org/licenses/MIT)
|
|
53
|
+
|
|
50
54
|
Audio-visual synchronization detection using deep learning with modern Python architecture.
|
|
51
55
|
|
|
52
56
|
This is a **refactored and enhanced version** of the original [SyncNet implementation](https://github.com/joonson/syncnet_python) by Joon Son Chung, updated for Python 3.9+ with clean architecture, comprehensive error handling, and performance optimizations.
|
|
@@ -1,18 +1,21 @@
|
|
|
1
1
|
syncnet_python/SyncNetInstance.py,sha256=V-RtWL4nN2CW5n56DkE0d6hcnet4hAs_OVMaMS1ih20,6227
|
|
2
2
|
syncnet_python/SyncNetModel.py,sha256=6qk27paoyV39MTsVyu6K2sbbM-1SLU_2N1zbqm1eeYs,3575
|
|
3
|
-
syncnet_python/__init__.py,sha256=
|
|
3
|
+
syncnet_python/__init__.py,sha256=qpGU322MOioCGVW_CqFSN7WsGsRO__GEtNjHCzD7ce4,1037
|
|
4
4
|
syncnet_python/cli.py,sha256=YSQVVDChLkQ96gFRFqD7laQXqbrLC2b8SKaF-vc8FuA,3256
|
|
5
5
|
syncnet_python/run_syncnet_pipeline_on_1example.py,sha256=7I8_jEgIFw-RUqCHU29LrIWIRP77lTUiH6MePOxsuWQ,1008
|
|
6
6
|
syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py,sha256=HJHGLgKpf21bG27_x1QD3MAhqtHey26Ujxb8s2aK-tA,5928
|
|
7
7
|
syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py,sha256=m-kz_DHhka_DywpBRWdxk-S0hrwqvCR16RiVQ7HUqDk,6003
|
|
8
|
-
syncnet_python/
|
|
8
|
+
syncnet_python/safe_syncnet_utils.py,sha256=viucLggg13AW5xAjns_s-jlqv4rzvuRED0Ydo99_RbY,5381
|
|
9
|
+
syncnet_python/syncnet_pipeline.py,sha256=KiLcxsp2-TNKXUxHoFASOwpOizXPwkMmvS42wFHSmes,15770
|
|
10
|
+
syncnet_python/test_error_handling.py,sha256=i5ib1qEjbMUQg913r9plAdXYciQx2jmUjzuoXPvc5f4,4658
|
|
11
|
+
syncnet_python/test_lse_metrics.py,sha256=H_4U2LN-Wussx_KABX5y5EO3QGnaVhKNMuTjLH0zNrE,2884
|
|
9
12
|
syncnet_python/detectors/__init__.py,sha256=WLone-DTbvQUFleY93pmq7xomPDVjda9ATpZYcS6_sA,23
|
|
10
13
|
syncnet_python/detectors/s3fd/__init__.py,sha256=MIJfIEsKFTGh8brmD3fBQg8JZAp57NtaTkY6_I5QZP0,2322
|
|
11
14
|
syncnet_python/detectors/s3fd/box_utils.py,sha256=CWn46LMJKO1cdenI_IWxBbDriO2ueWTCl14Xp28fr-U,7235
|
|
12
15
|
syncnet_python/detectors/s3fd/nets.py,sha256=BYPJq9UJ5bq5Q5RgP1j-NsFzO1gP3N0PYE1X19-4PZw,5891
|
|
13
|
-
syncnet_python-0.2.
|
|
14
|
-
syncnet_python-0.2.
|
|
15
|
-
syncnet_python-0.2.
|
|
16
|
-
syncnet_python-0.2.
|
|
17
|
-
syncnet_python-0.2.
|
|
18
|
-
syncnet_python-0.2.
|
|
16
|
+
syncnet_python-0.2.2.dist-info/licenses/LICENSE,sha256=qwJOQjZqnGgzcgwg3VUkT2pGKwLyZr5nUOm0Gv7Zfog,1088
|
|
17
|
+
syncnet_python-0.2.2.dist-info/METADATA,sha256=y58FslZFThl3p0xv_415IKKW-qjbTHWzBfc6aN1kbrY,7650
|
|
18
|
+
syncnet_python-0.2.2.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
19
|
+
syncnet_python-0.2.2.dist-info/entry_points.txt,sha256=zPJrIl_ekAa31dHZNCoTHiII2k_m5GyMD5BEkMR6Ytk,59
|
|
20
|
+
syncnet_python-0.2.2.dist-info/top_level.txt,sha256=OkvKpxwq9NzQSjjdD571umGrI3meW9kVleOV2N5Ua04,15
|
|
21
|
+
syncnet_python-0.2.2.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|