syncnet-python 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/INSTALL.md +1 -1
  2. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/PKG-INFO +6 -1
  3. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/README.md +5 -0
  4. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/pyproject.toml +1 -1
  5. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/syncnet_pipeline.py +49 -12
  6. syncnet_python-0.2.1/syncnet_python/test_lse_metrics.py +95 -0
  7. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python.egg-info/PKG-INFO +6 -1
  8. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python.egg-info/SOURCES.txt +1 -0
  9. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/CLAUDE.md +0 -0
  10. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/LICENSE +0 -0
  11. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/MANIFEST.in +0 -0
  12. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/example/speech.wav +0 -0
  13. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/example/video.avi +0 -0
  14. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/requirements.txt +0 -0
  15. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/scripts/run_batch.py +0 -0
  16. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/scripts/run_example.py +0 -0
  17. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/setup.cfg +0 -0
  18. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/setup.py +0 -0
  19. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/__init__.py +0 -0
  20. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/cli.py +0 -0
  21. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/__init__.py +0 -0
  22. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/audio.py +0 -0
  23. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/base.py +0 -0
  24. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/compat.py +0 -0
  25. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/config.py +0 -0
  26. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/exceptions.py +0 -0
  27. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/inference.py +0 -0
  28. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/logging.py +0 -0
  29. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/models.py +0 -0
  30. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/sync_analyzer.py +0 -0
  31. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/types.py +0 -0
  32. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/utils.py +0 -0
  33. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/core/video.py +0 -0
  34. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/detectors/__init__.py +0 -0
  35. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/detectors/s3fd/__init__.py +0 -0
  36. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/detectors/s3fd/detector.py +0 -0
  37. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/detectors/s3fd/utils.py +0 -0
  38. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/pipeline/__init__.py +0 -0
  39. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/pipeline/config.py +0 -0
  40. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/pipeline/pipeline.py +0 -0
  41. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/utils/__init__.py +0 -0
  42. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/utils/exceptions.py +0 -0
  43. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/utils/face_detection.py +0 -0
  44. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet/utils/video.py +0 -0
  45. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/SyncNetInstance.py +0 -0
  46. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/SyncNetModel.py +0 -0
  47. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/__init__.py +0 -0
  48. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/cli.py +0 -0
  49. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/detectors/__init__.py +0 -0
  50. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/detectors/s3fd/__init__.py +0 -0
  51. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/detectors/s3fd/box_utils.py +0 -0
  52. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/detectors/s3fd/nets.py +0 -0
  53. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/run_syncnet_pipeline_on_1example.py +0 -0
  54. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py +0 -0
  55. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py +0 -0
  56. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python.egg-info/dependency_links.txt +0 -0
  57. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python.egg-info/entry_points.txt +0 -0
  58. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python.egg-info/requires.txt +0 -0
  59. {syncnet_python-0.2.0 → syncnet_python-0.2.1}/syncnet_python.egg-info/top_level.txt +0 -0
@@ -2,7 +2,7 @@
2
2
 
3
3
  ## Prerequisites
4
4
 
5
- - Python 3.13 or higher
5
+ - Python 3.9 or higher
6
6
  - CUDA-capable GPU (optional, but recommended)
7
7
  - FFmpeg
8
8
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: syncnet-python
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: SyncNet: Audio-visual synchronization detection using deep learning. Updated version of https://github.com/joonson/syncnet_python for modern Python versions.
5
5
  Author: SyncNet Python Contributors
6
6
  Maintainer: SyncNet Python Contributors
@@ -47,6 +47,11 @@ Dynamic: license-file
47
47
 
48
48
  # SyncNet Python
49
49
 
50
+ [![PyPI version](https://badge.fury.io/py/syncnet-python.svg)](https://badge.fury.io/py/syncnet-python)
51
+ [![Python](https://img.shields.io/pypi/pyversions/syncnet-python.svg)](https://pypi.org/project/syncnet-python/)
52
+ [![Downloads](https://pepy.tech/badge/syncnet-python)](https://pepy.tech/project/syncnet-python)
53
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
54
+
50
55
  Audio-visual synchronization detection using deep learning with modern Python architecture.
51
56
 
52
57
  This is a **refactored and enhanced version** of the original [SyncNet implementation](https://github.com/joonson/syncnet_python) by Joon Son Chung, updated for Python 3.9+ with clean architecture, comprehensive error handling, and performance optimizations.
@@ -1,5 +1,10 @@
1
1
  # SyncNet Python
2
2
 
3
+ [![PyPI version](https://badge.fury.io/py/syncnet-python.svg)](https://badge.fury.io/py/syncnet-python)
4
+ [![Python](https://img.shields.io/pypi/pyversions/syncnet-python.svg)](https://pypi.org/project/syncnet-python/)
5
+ [![Downloads](https://pepy.tech/badge/syncnet-python)](https://pepy.tech/project/syncnet-python)
6
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
7
+
3
8
  Audio-visual synchronization detection using deep learning with modern Python architecture.
4
9
 
5
10
  This is a **refactored and enhanced version** of the original [SyncNet implementation](https://github.com/joonson/syncnet_python) by Joon Son Chung, updated for Python 3.9+ with clean architecture, comprehensive error handling, and performance optimizations.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "syncnet-python"
7
- version = "0.2.0"
7
+ version = "0.2.1"
8
8
  description = "SyncNet: Audio-visual synchronization detection using deep learning. Updated version of https://github.com/joonson/syncnet_python for modern Python versions."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
@@ -18,10 +18,17 @@ from scipy.interpolate import interp1d
18
18
  from scenedetect import ContentDetector, SceneManager, StatsManager
19
19
  from scenedetect.video_manager import VideoManager
20
20
 
21
- from .detectors.s3fd import S3FD
22
- from .detectors.s3fd.nets import S3FDNet
23
- from .SyncNetInstance import SyncNetInstance
24
- from .SyncNetModel import S
21
+ try:
22
+ from .detectors.s3fd import S3FD
23
+ from .detectors.s3fd.nets import S3FDNet
24
+ from .SyncNetInstance import SyncNetInstance
25
+ from .SyncNetModel import S
26
+ except ImportError:
27
+ # Fallback for direct script execution
28
+ from detectors.s3fd import S3FD
29
+ from detectors.s3fd.nets import S3FDNet
30
+ from SyncNetInstance import SyncNetInstance
31
+ from SyncNetModel import S
25
32
 
26
33
  # ---------------------------------------------------------------------- #
27
34
  # Configuration #
@@ -47,6 +54,11 @@ class PipelineConfig:
47
54
  # Tools
48
55
  ffmpeg_bin: str = "ffmpeg" # assumes ffmpeg in $PATH
49
56
  audio_sample_rate: int = 16000 # resample rate for speech
57
+
58
+ def __post_init__(self):
59
+ """Validate configuration after initialization."""
60
+ if self.ffmpeg_bin is None:
61
+ self.ffmpeg_bin = "ffmpeg"
50
62
 
51
63
  @classmethod
52
64
  def from_dict(cls, d: Dict[str, Any]):
@@ -165,16 +177,41 @@ class SyncNetPipeline:
165
177
  slice_wav = f"{base}.wav"
166
178
  ss = track["frame"][0] / cfg.frame_rate
167
179
  to = (track["frame"][-1] + 1) / cfg.frame_rate
168
- subprocess.call(
169
- f'{cfg.ffmpeg_bin} -y -i "{audio_wav}" -ss {ss:.3f} -to {to:.3f} "{slice_wav}"',
170
- shell=True,
171
- )
180
+
181
+ # Ensure ffmpeg_bin is not None
182
+ ffmpeg_bin = cfg.ffmpeg_bin if cfg.ffmpeg_bin is not None else "ffmpeg"
183
+
184
+ cmd = [
185
+ ffmpeg_bin, "-y", "-i", str(audio_wav),
186
+ "-ss", f"{ss:.3f}", "-to", f"{to:.3f}",
187
+ str(slice_wav)
188
+ ]
189
+
190
+ try:
191
+ result = subprocess.run(cmd, capture_output=True, text=True, check=True)
192
+ except subprocess.CalledProcessError as e:
193
+ logging.error(f"FFmpeg audio slicing failed: {e.stderr}")
194
+ raise RuntimeError(f"FFmpeg audio slicing failed: {e.stderr}")
195
+ except FileNotFoundError:
196
+ logging.error(f"FFmpeg not found at: {ffmpeg_bin}")
197
+ raise RuntimeError(f"FFmpeg not found. Please ensure ffmpeg is installed and in PATH.")
172
198
 
173
199
  final_avi = f"{base}.avi"
174
- subprocess.call(
175
- f'{cfg.ffmpeg_bin} -y -i "{tmp_avi}" -i "{slice_wav}" -c:v copy -c:a copy "{final_avi}"',
176
- shell=True,
177
- )
200
+
201
+ cmd = [
202
+ ffmpeg_bin, "-y", "-i", str(tmp_avi), "-i", str(slice_wav),
203
+ "-c:v", "copy", "-c:a", "copy", str(final_avi)
204
+ ]
205
+
206
+ try:
207
+ result = subprocess.run(cmd, capture_output=True, text=True, check=True)
208
+ except subprocess.CalledProcessError as e:
209
+ logging.error(f"FFmpeg video/audio merge failed: {e.stderr}")
210
+ raise RuntimeError(f"FFmpeg video/audio merge failed: {e.stderr}")
211
+ except FileNotFoundError:
212
+ logging.error(f"FFmpeg not found at: {ffmpeg_bin}")
213
+ raise RuntimeError(f"FFmpeg not found. Please ensure ffmpeg is installed and in PATH.")
214
+
178
215
  os.remove(tmp_avi)
179
216
  return final_avi
180
217
 
@@ -0,0 +1,95 @@
1
+ import logging
2
+ import sys
3
+ from pathlib import Path
4
+ from syncnet_pipeline import SyncNetPipeline
5
+
6
+ logging.basicConfig(
7
+ level=logging.INFO,
8
+ format="%(asctime)s [%(levelname)s] %(message)s"
9
+ )
10
+
11
+ def calculate_lse_metrics(video_path):
12
+ """
13
+ Calculate LSE-C and LSE-D metrics for a video
14
+
15
+ LSE-C (Lip Sync Error - Confidence):
16
+ - Higher confidence values indicate better lip sync
17
+ - Threshold typically around 3.5-4.0 for good sync
18
+
19
+ LSE-D (Lip Sync Error - Distance):
20
+ - Lower distance values indicate better lip sync
21
+ - Threshold typically around 6.5-7.0 for good sync
22
+ """
23
+ # Initialize pipeline
24
+ pipe = SyncNetPipeline(
25
+ {
26
+ "s3fd_weights": "../weights/sfd_face.pth",
27
+ "syncnet_weights": "../weights/syncnet_v2.model",
28
+ },
29
+ device="cuda", # or "cpu"
30
+ )
31
+
32
+ # Run inference
33
+ print(f"\n=== Processing {video_path} ===")
34
+
35
+ # For testing, we'll extract audio from the video itself
36
+ results = pipe.inference(
37
+ video_path=video_path,
38
+ audio_path=video_path, # Extract audio from same video
39
+ cache_dir="../example/cache_lse",
40
+ )
41
+
42
+ offsets, confs, dists, max_conf, min_dist, s3fd_json, has_face = results
43
+
44
+ if not has_face:
45
+ print(f"No face detected in {video_path}")
46
+ return None, None
47
+
48
+ # LSE-C is the maximum confidence across all tracks
49
+ lse_c = max_conf
50
+
51
+ # LSE-D is the minimum distance across all tracks
52
+ lse_d = min_dist
53
+
54
+ print(f"Number of face tracks: {len(offsets)}")
55
+ print(f"Track offsets: {offsets}")
56
+ print(f"Track confidences: {[f'{c:.3f}' for c in confs]}")
57
+ print(f"Track distances: {[f'{d:.3f}' for d in dists]}")
58
+ print(f"LSE-C (max confidence): {lse_c:.3f}")
59
+ print(f"LSE-D (min distance): {lse_d:.3f}")
60
+
61
+ # Interpretation
62
+ sync_quality = "GOOD" if lse_c > 3.5 and lse_d < 7.0 else "POOR"
63
+ print(f"Sync Quality: {sync_quality}")
64
+
65
+ return lse_c, lse_d
66
+
67
+ def main():
68
+ # Test files
69
+ test_files = [
70
+ "../example/pair_0000_lipsynced.mp4",
71
+ "../example/pair_0001_lipsynced.mp4"
72
+ ]
73
+
74
+ results = {}
75
+
76
+ for video_path in test_files:
77
+ if Path(video_path).exists():
78
+ lse_c, lse_d = calculate_lse_metrics(video_path)
79
+ results[video_path] = (lse_c, lse_d)
80
+ else:
81
+ print(f"File not found: {video_path}")
82
+
83
+ # Summary
84
+ print("\n=== SUMMARY ===")
85
+ print(f"{'Video':<40} {'LSE-C':<10} {'LSE-D':<10} {'Quality':<10}")
86
+ print("-" * 70)
87
+
88
+ for video_path, (lse_c, lse_d) in results.items():
89
+ if lse_c is not None:
90
+ video_name = Path(video_path).name
91
+ quality = "GOOD" if lse_c > 3.5 and lse_d < 7.0 else "POOR"
92
+ print(f"{video_name:<40} {lse_c:<10.3f} {lse_d:<10.3f} {quality:<10}")
93
+
94
+ if __name__ == "__main__":
95
+ main()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: syncnet-python
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: SyncNet: Audio-visual synchronization detection using deep learning. Updated version of https://github.com/joonson/syncnet_python for modern Python versions.
5
5
  Author: SyncNet Python Contributors
6
6
  Maintainer: SyncNet Python Contributors
@@ -47,6 +47,11 @@ Dynamic: license-file
47
47
 
48
48
  # SyncNet Python
49
49
 
50
+ [![PyPI version](https://badge.fury.io/py/syncnet-python.svg)](https://badge.fury.io/py/syncnet-python)
51
+ [![Python](https://img.shields.io/pypi/pyversions/syncnet-python.svg)](https://pypi.org/project/syncnet-python/)
52
+ [![Downloads](https://pepy.tech/badge/syncnet-python)](https://pepy.tech/project/syncnet-python)
53
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
54
+
50
55
  Audio-visual synchronization detection using deep learning with modern Python architecture.
51
56
 
52
57
  This is a **refactored and enhanced version** of the original [SyncNet implementation](https://github.com/joonson/syncnet_python) by Joon Son Chung, updated for Python 3.9+ with clean architecture, comprehensive error handling, and performance optimizations.
@@ -44,6 +44,7 @@ syncnet_python/run_syncnet_pipeline_on_1example.py
44
44
  syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py
45
45
  syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py
46
46
  syncnet_python/syncnet_pipeline.py
47
+ syncnet_python/test_lse_metrics.py
47
48
  syncnet_python.egg-info/PKG-INFO
48
49
  syncnet_python.egg-info/SOURCES.txt
49
50
  syncnet_python.egg-info/dependency_links.txt
File without changes
File without changes
File without changes
File without changes