syncnet-python 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- syncnet_python-0.1.0/CLAUDE.md +79 -0
- syncnet_python-0.1.0/INSTALL.md +73 -0
- syncnet_python-0.1.0/LICENSE +21 -0
- syncnet_python-0.1.0/MANIFEST.in +33 -0
- syncnet_python-0.1.0/PKG-INFO +150 -0
- syncnet_python-0.1.0/README.md +103 -0
- syncnet_python-0.1.0/example/speech.wav +0 -0
- syncnet_python-0.1.0/example/video.avi +0 -0
- syncnet_python-0.1.0/pyproject.toml +102 -0
- syncnet_python-0.1.0/requirements.txt +0 -0
- syncnet_python-0.1.0/script/SyncNetInstance.py +210 -0
- syncnet_python-0.1.0/script/SyncNetModel.py +99 -0
- syncnet_python-0.1.0/script/detectors/__init__.py +1 -0
- syncnet_python-0.1.0/script/detectors/s3fd/__init__.py +66 -0
- syncnet_python-0.1.0/script/detectors/s3fd/box_utils.py +233 -0
- syncnet_python-0.1.0/script/detectors/s3fd/nets.py +177 -0
- syncnet_python-0.1.0/script/run_syncnet_pipeline_on_1example.py +28 -0
- syncnet_python-0.1.0/script/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py +157 -0
- syncnet_python-0.1.0/script/run_syncnet_pipeline_on_your_own_model_results.py +158 -0
- syncnet_python-0.1.0/script/syncnet_pipeline.py +332 -0
- syncnet_python-0.1.0/scripts/run_batch.py +75 -0
- syncnet_python-0.1.0/scripts/run_example.py +44 -0
- syncnet_python-0.1.0/setup.cfg +4 -0
- syncnet_python-0.1.0/setup.py +6 -0
- syncnet_python-0.1.0/syncnet/__init__.py +11 -0
- syncnet_python-0.1.0/syncnet/cli.py +280 -0
- syncnet_python-0.1.0/syncnet/core/__init__.py +12 -0
- syncnet_python-0.1.0/syncnet/core/compat.py +220 -0
- syncnet_python-0.1.0/syncnet/core/inference.py +300 -0
- syncnet_python-0.1.0/syncnet/core/models.py +195 -0
- syncnet_python-0.1.0/syncnet/core/types.py +48 -0
- syncnet_python-0.1.0/syncnet/detectors/__init__.py +1 -0
- syncnet_python-0.1.0/syncnet/detectors/s3fd/__init__.py +6 -0
- syncnet_python-0.1.0/syncnet/detectors/s3fd/detector.py +231 -0
- syncnet_python-0.1.0/syncnet/detectors/s3fd/utils.py +248 -0
- syncnet_python-0.1.0/syncnet/pipeline/__init__.py +11 -0
- syncnet_python-0.1.0/syncnet/pipeline/config.py +61 -0
- syncnet_python-0.1.0/syncnet/pipeline/pipeline.py +333 -0
- syncnet_python-0.1.0/syncnet/utils/__init__.py +17 -0
- syncnet_python-0.1.0/syncnet/utils/exceptions.py +88 -0
- syncnet_python-0.1.0/syncnet/utils/face_detection.py +253 -0
- syncnet_python-0.1.0/syncnet/utils/video.py +314 -0
- syncnet_python-0.1.0/syncnet_python/SyncNetInstance.py +210 -0
- syncnet_python-0.1.0/syncnet_python/SyncNetModel.py +99 -0
- syncnet_python-0.1.0/syncnet_python/__init__.py +29 -0
- syncnet_python-0.1.0/syncnet_python/cli.py +127 -0
- syncnet_python-0.1.0/syncnet_python/detectors/__init__.py +1 -0
- syncnet_python-0.1.0/syncnet_python/detectors/s3fd/__init__.py +66 -0
- syncnet_python-0.1.0/syncnet_python/detectors/s3fd/box_utils.py +233 -0
- syncnet_python-0.1.0/syncnet_python/detectors/s3fd/nets.py +177 -0
- syncnet_python-0.1.0/syncnet_python/run_syncnet_pipeline_on_1example.py +28 -0
- syncnet_python-0.1.0/syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py +157 -0
- syncnet_python-0.1.0/syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py +158 -0
- syncnet_python-0.1.0/syncnet_python/syncnet_pipeline.py +332 -0
- syncnet_python-0.1.0/syncnet_python.egg-info/PKG-INFO +150 -0
- syncnet_python-0.1.0/syncnet_python.egg-info/SOURCES.txt +58 -0
- syncnet_python-0.1.0/syncnet_python.egg-info/dependency_links.txt +1 -0
- syncnet_python-0.1.0/syncnet_python.egg-info/entry_points.txt +2 -0
- syncnet_python-0.1.0/syncnet_python.egg-info/requires.txt +18 -0
- syncnet_python-0.1.0/syncnet_python.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# CLAUDE.md
|
|
2
|
+
|
|
3
|
+
This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
|
|
4
|
+
|
|
5
|
+
## Project Overview
|
|
6
|
+
|
|
7
|
+
SyncNet_py313 is a Python 3.13 implementation of SyncNet, a neural network model for audio-visual synchronization detection. It evaluates lip-sync quality in videos by computing synchronization confidence scores between audio and visual features.
|
|
8
|
+
|
|
9
|
+
## Common Commands
|
|
10
|
+
|
|
11
|
+
### Setup and Dependencies
|
|
12
|
+
```bash
|
|
13
|
+
# Install dependencies
|
|
14
|
+
pip install -r requirements.txt
|
|
15
|
+
|
|
16
|
+
# Ensure ffmpeg is installed on the system
|
|
17
|
+
# Ubuntu/Debian: sudo apt-get install ffmpeg
|
|
18
|
+
# macOS: brew install ffmpeg
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
### Running SyncNet
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
# Run on a single example
|
|
25
|
+
cd script
|
|
26
|
+
python run_syncnet_pipeline_on_1example.py
|
|
27
|
+
|
|
28
|
+
# Run on MoChaBench evaluation set
|
|
29
|
+
python run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py
|
|
30
|
+
|
|
31
|
+
# Run on custom model results
|
|
32
|
+
python run_syncnet_pipeline_on_your_own_model_results.py
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## Architecture Overview
|
|
36
|
+
|
|
37
|
+
### Core Components
|
|
38
|
+
|
|
39
|
+
1. **SyncNetModel** (`script/SyncNetModel.py`): The neural network architecture
|
|
40
|
+
- Audio encoder: 1D convolutions + FC layers for audio feature extraction
|
|
41
|
+
- Visual encoder: 3D convolutions + FC layers for video feature extraction
|
|
42
|
+
- Outputs 1024-dimensional embeddings for both modalities
|
|
43
|
+
|
|
44
|
+
2. **SyncNetInstance** (`script/SyncNetInstance.py`): Inference wrapper
|
|
45
|
+
- Loads pre-trained weights from `weights/syncnet_v2.model`
|
|
46
|
+
- Calculates synchronization scores using sliding window approach
|
|
47
|
+
- Returns offset values and confidence scores
|
|
48
|
+
|
|
49
|
+
3. **Pipeline** (`script/syncnet_pipeline.py`): End-to-end processing
|
|
50
|
+
- Face detection using S3FD detector
|
|
51
|
+
- Audio extraction and MFCC feature computation
|
|
52
|
+
- Video frame extraction and face cropping
|
|
53
|
+
- Synchronization score calculation
|
|
54
|
+
|
|
55
|
+
4. **Face Detector** (`script/detectors/s3fd/`): S3FD implementation
|
|
56
|
+
- Uses pre-trained weights from `weights/sfd_face.pth`
|
|
57
|
+
- Handles multi-scale face detection
|
|
58
|
+
|
|
59
|
+
### Data Flow
|
|
60
|
+
|
|
61
|
+
1. Input: Video file (e.g., `.avi`, `.mp4`)
|
|
62
|
+
2. Face detection → Face tracking → Face cropping
|
|
63
|
+
3. Audio extraction → MFCC features (13 coefficients)
|
|
64
|
+
4. Parallel processing of audio and visual streams
|
|
65
|
+
5. Output: Synchronization scores and offset values
|
|
66
|
+
|
|
67
|
+
### Key Implementation Details
|
|
68
|
+
|
|
69
|
+
- Audio: 16kHz sampling, 25ms windows, 10ms stride
|
|
70
|
+
- Video: 25 FPS, 224x224 face crops, 5-frame sequences
|
|
71
|
+
- Embeddings: 1024-dimensional for both modalities
|
|
72
|
+
- Scoring: Cosine similarity between audio/visual embeddings
|
|
73
|
+
|
|
74
|
+
## Important Notes
|
|
75
|
+
|
|
76
|
+
- The model requires pre-trained weights in `weights/` directory
|
|
77
|
+
- GPU is strongly recommended for inference (falls back to CPU if unavailable)
|
|
78
|
+
- Face detection is critical - videos without detectable faces will fail
|
|
79
|
+
- The pipeline creates temporary directories for intermediate processing
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
# Installation Guide
|
|
2
|
+
|
|
3
|
+
## Prerequisites
|
|
4
|
+
|
|
5
|
+
- Python 3.13 or higher
|
|
6
|
+
- CUDA-capable GPU (optional, but recommended)
|
|
7
|
+
- FFmpeg
|
|
8
|
+
|
|
9
|
+
## Installation Steps
|
|
10
|
+
|
|
11
|
+
### 1. Clone the repository
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
git clone https://github.com/yourusername/SyncNet_py313.git
|
|
15
|
+
cd SyncNet_py313
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
### 2. Create virtual environment
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
python -m venv venv
|
|
22
|
+
source venv/bin/activate # On Windows: venv\Scripts\activate
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
### 3. Install dependencies
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install -r requirements.txt
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Or install in development mode:
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install -e .
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
### 4. Download model weights
|
|
38
|
+
|
|
39
|
+
Place the following files in the `weights/` directory:
|
|
40
|
+
- `sfd_face.pth` - S3FD face detector weights
|
|
41
|
+
- `syncnet_v2.model` - SyncNet model weights
|
|
42
|
+
|
|
43
|
+
### 5. Verify installation
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
python scripts/run_example.py
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
## Troubleshooting
|
|
50
|
+
|
|
51
|
+
### CUDA not available
|
|
52
|
+
|
|
53
|
+
If you see CUDA-related errors, ensure:
|
|
54
|
+
1. NVIDIA drivers are installed
|
|
55
|
+
2. CUDA toolkit is installed
|
|
56
|
+
3. PyTorch is installed with CUDA support:
|
|
57
|
+
```bash
|
|
58
|
+
pip install torch torchvision --index-url https://download.pytorch.org/whl/cu121
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
### FFmpeg not found
|
|
62
|
+
|
|
63
|
+
Install FFmpeg:
|
|
64
|
+
- Ubuntu/Debian: `sudo apt-get install ffmpeg`
|
|
65
|
+
- macOS: `brew install ffmpeg`
|
|
66
|
+
- Windows: Download from https://ffmpeg.org/download.html
|
|
67
|
+
|
|
68
|
+
### Missing dependencies
|
|
69
|
+
|
|
70
|
+
If you encounter import errors:
|
|
71
|
+
```bash
|
|
72
|
+
pip install --upgrade -r requirements.txt
|
|
73
|
+
```
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 SyncNet Python 3.13 Contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
include README.md
|
|
2
|
+
include LICENSE
|
|
3
|
+
include INSTALL.md
|
|
4
|
+
include CLAUDE.md
|
|
5
|
+
include requirements.txt
|
|
6
|
+
|
|
7
|
+
# Include example files
|
|
8
|
+
include example/*.wav
|
|
9
|
+
include example/*.avi
|
|
10
|
+
|
|
11
|
+
# Include scripts
|
|
12
|
+
recursive-include scripts *.py
|
|
13
|
+
|
|
14
|
+
# Include package data
|
|
15
|
+
recursive-include syncnet *.py
|
|
16
|
+
recursive-include script *.py
|
|
17
|
+
|
|
18
|
+
# Exclude unnecessary files
|
|
19
|
+
global-exclude __pycache__
|
|
20
|
+
global-exclude *.py[co]
|
|
21
|
+
global-exclude .DS_Store
|
|
22
|
+
global-exclude *.swp
|
|
23
|
+
global-exclude *.swo
|
|
24
|
+
|
|
25
|
+
# Exclude weights (too large for PyPI)
|
|
26
|
+
exclude weights/*.pth
|
|
27
|
+
exclude weights/*.model
|
|
28
|
+
|
|
29
|
+
# Exclude test and cache directories
|
|
30
|
+
prune temp_test
|
|
31
|
+
prune example/cache
|
|
32
|
+
prune .git
|
|
33
|
+
prune .claude
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: syncnet-python
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: SyncNet: Audio-visual synchronization detection using deep learning
|
|
5
|
+
Author: SyncNet Python Contributors
|
|
6
|
+
Maintainer: SyncNet Python Contributors
|
|
7
|
+
License: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/nawta/SyncNet_py313
|
|
9
|
+
Project-URL: Bug Reports, https://github.com/nawta/SyncNet_py313/issues
|
|
10
|
+
Project-URL: Source, https://github.com/nawta/SyncNet_py313
|
|
11
|
+
Keywords: audio-visual,synchronization,deep-learning,pytorch,lip-sync,video-processing,computer-vision
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Topic :: Multimedia :: Video
|
|
25
|
+
Classifier: Topic :: Scientific/Engineering :: Image Processing
|
|
26
|
+
Requires-Python: >=3.9
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
License-File: LICENSE
|
|
29
|
+
Requires-Dist: torch>=2.0.0
|
|
30
|
+
Requires-Dist: torchvision>=0.15.0
|
|
31
|
+
Requires-Dist: numpy>=1.24.0
|
|
32
|
+
Requires-Dist: scipy>=1.10.0
|
|
33
|
+
Requires-Dist: pandas>=2.0.0
|
|
34
|
+
Requires-Dist: scenedetect[opencv]>=0.6.0
|
|
35
|
+
Requires-Dist: opencv-contrib-python>=4.8.0
|
|
36
|
+
Requires-Dist: python-speech-features>=0.6
|
|
37
|
+
Requires-Dist: ffmpeg-python>=0.2.0
|
|
38
|
+
Provides-Extra: dev
|
|
39
|
+
Requires-Dist: pytest>=7.4.0; extra == "dev"
|
|
40
|
+
Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
|
|
41
|
+
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
42
|
+
Requires-Dist: ruff>=0.1.0; extra == "dev"
|
|
43
|
+
Requires-Dist: mypy>=1.7.0; extra == "dev"
|
|
44
|
+
Requires-Dist: types-opencv-python; extra == "dev"
|
|
45
|
+
Requires-Dist: types-scipy; extra == "dev"
|
|
46
|
+
Dynamic: license-file
|
|
47
|
+
|
|
48
|
+
# SyncNet Python
|
|
49
|
+
|
|
50
|
+
Audio-visual synchronization detection using deep learning.
|
|
51
|
+
|
|
52
|
+
## Overview
|
|
53
|
+
|
|
54
|
+
SyncNet Python is a PyTorch implementation of the SyncNet model, which detects audio-visual synchronization in videos. It can identify lip-sync errors by analyzing the correspondence between mouth movements and spoken audio.
|
|
55
|
+
|
|
56
|
+
## Features
|
|
57
|
+
|
|
58
|
+
- 🎥 **Audio-Visual Sync Detection**: Accurately detect synchronization between audio and video
|
|
59
|
+
- 🔍 **Face Detection**: Automatic face detection and tracking using S3FD
|
|
60
|
+
- 🚀 **Batch Processing**: Process multiple videos efficiently
|
|
61
|
+
- 🐍 **Python API**: Easy-to-use Python interface
|
|
62
|
+
- 📊 **Confidence Scores**: Get confidence metrics for sync quality
|
|
63
|
+
|
|
64
|
+
## Installation
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
pip install syncnet-python
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
### Additional Requirements
|
|
71
|
+
|
|
72
|
+
1. **FFmpeg**: Required for video processing
|
|
73
|
+
```bash
|
|
74
|
+
# Ubuntu/Debian
|
|
75
|
+
sudo apt-get install ffmpeg
|
|
76
|
+
|
|
77
|
+
# macOS
|
|
78
|
+
brew install ffmpeg
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
2. **Model Weights**: Download pre-trained weights
|
|
82
|
+
- Download `sfd_face.pth` and `syncnet_v2.model`
|
|
83
|
+
- Place them in a `weights/` directory
|
|
84
|
+
|
|
85
|
+
## Quick Start
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
from syncnet_python import SyncNetPipeline
|
|
89
|
+
|
|
90
|
+
# Initialize pipeline
|
|
91
|
+
pipeline = SyncNetPipeline(
|
|
92
|
+
s3fd_weights="weights/sfd_face.pth",
|
|
93
|
+
syncnet_weights="weights/syncnet_v2.model",
|
|
94
|
+
device="cuda" # or "cpu"
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
# Process video
|
|
98
|
+
results = pipeline.inference(
|
|
99
|
+
video_path="video.mp4",
|
|
100
|
+
audio_path=None # Extract from video
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
# Get results
|
|
104
|
+
offset, confidence = results['offset'], results['confidence']
|
|
105
|
+
print(f"AV Offset: {offset} frames")
|
|
106
|
+
print(f"Confidence: {confidence:.3f}")
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Command Line Usage
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
# Process single video
|
|
113
|
+
syncnet-python video.mp4
|
|
114
|
+
|
|
115
|
+
# Process multiple videos
|
|
116
|
+
syncnet-python video1.mp4 video2.mp4 --output results.json
|
|
117
|
+
|
|
118
|
+
# Use CPU instead of GPU
|
|
119
|
+
syncnet-python video.mp4 --device cpu
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## Requirements
|
|
123
|
+
|
|
124
|
+
- Python 3.9+
|
|
125
|
+
- PyTorch 2.0+
|
|
126
|
+
- CUDA (optional but recommended)
|
|
127
|
+
- FFmpeg
|
|
128
|
+
|
|
129
|
+
## Citation
|
|
130
|
+
|
|
131
|
+
If you use this code in your research, please cite:
|
|
132
|
+
|
|
133
|
+
```bibtex
|
|
134
|
+
@inproceedings{chung2016out,
|
|
135
|
+
title={Out of time: automated lip sync in the wild},
|
|
136
|
+
author={Chung, Joon Son and Zisserman, Andrew},
|
|
137
|
+
booktitle={Asian Conference on Computer Vision},
|
|
138
|
+
year={2016}
|
|
139
|
+
}
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
## License
|
|
143
|
+
|
|
144
|
+
MIT License - see LICENSE file for details.
|
|
145
|
+
|
|
146
|
+
## Links
|
|
147
|
+
|
|
148
|
+
- GitHub: https://github.com/yourusername/syncnet-python
|
|
149
|
+
- Documentation: https://syncnet-python.readthedocs.io
|
|
150
|
+
- Issues: https://github.com/yourusername/syncnet-python/issues
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# SyncNet Python
|
|
2
|
+
|
|
3
|
+
Audio-visual synchronization detection using deep learning.
|
|
4
|
+
|
|
5
|
+
## Overview
|
|
6
|
+
|
|
7
|
+
SyncNet Python is a PyTorch implementation of the SyncNet model, which detects audio-visual synchronization in videos. It can identify lip-sync errors by analyzing the correspondence between mouth movements and spoken audio.
|
|
8
|
+
|
|
9
|
+
## Features
|
|
10
|
+
|
|
11
|
+
- 🎥 **Audio-Visual Sync Detection**: Accurately detect synchronization between audio and video
|
|
12
|
+
- 🔍 **Face Detection**: Automatic face detection and tracking using S3FD
|
|
13
|
+
- 🚀 **Batch Processing**: Process multiple videos efficiently
|
|
14
|
+
- 🐍 **Python API**: Easy-to-use Python interface
|
|
15
|
+
- 📊 **Confidence Scores**: Get confidence metrics for sync quality
|
|
16
|
+
|
|
17
|
+
## Installation
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install syncnet-python
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
### Additional Requirements
|
|
24
|
+
|
|
25
|
+
1. **FFmpeg**: Required for video processing
|
|
26
|
+
```bash
|
|
27
|
+
# Ubuntu/Debian
|
|
28
|
+
sudo apt-get install ffmpeg
|
|
29
|
+
|
|
30
|
+
# macOS
|
|
31
|
+
brew install ffmpeg
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
2. **Model Weights**: Download pre-trained weights
|
|
35
|
+
- Download `sfd_face.pth` and `syncnet_v2.model`
|
|
36
|
+
- Place them in a `weights/` directory
|
|
37
|
+
|
|
38
|
+
## Quick Start
|
|
39
|
+
|
|
40
|
+
```python
|
|
41
|
+
from syncnet_python import SyncNetPipeline
|
|
42
|
+
|
|
43
|
+
# Initialize pipeline
|
|
44
|
+
pipeline = SyncNetPipeline(
|
|
45
|
+
s3fd_weights="weights/sfd_face.pth",
|
|
46
|
+
syncnet_weights="weights/syncnet_v2.model",
|
|
47
|
+
device="cuda" # or "cpu"
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
# Process video
|
|
51
|
+
results = pipeline.inference(
|
|
52
|
+
video_path="video.mp4",
|
|
53
|
+
audio_path=None # Extract from video
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
# Get results
|
|
57
|
+
offset, confidence = results['offset'], results['confidence']
|
|
58
|
+
print(f"AV Offset: {offset} frames")
|
|
59
|
+
print(f"Confidence: {confidence:.3f}")
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## Command Line Usage
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
# Process single video
|
|
66
|
+
syncnet-python video.mp4
|
|
67
|
+
|
|
68
|
+
# Process multiple videos
|
|
69
|
+
syncnet-python video1.mp4 video2.mp4 --output results.json
|
|
70
|
+
|
|
71
|
+
# Use CPU instead of GPU
|
|
72
|
+
syncnet-python video.mp4 --device cpu
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
## Requirements
|
|
76
|
+
|
|
77
|
+
- Python 3.9+
|
|
78
|
+
- PyTorch 2.0+
|
|
79
|
+
- CUDA (optional but recommended)
|
|
80
|
+
- FFmpeg
|
|
81
|
+
|
|
82
|
+
## Citation
|
|
83
|
+
|
|
84
|
+
If you use this code in your research, please cite:
|
|
85
|
+
|
|
86
|
+
```bibtex
|
|
87
|
+
@inproceedings{chung2016out,
|
|
88
|
+
title={Out of time: automated lip sync in the wild},
|
|
89
|
+
author={Chung, Joon Son and Zisserman, Andrew},
|
|
90
|
+
booktitle={Asian Conference on Computer Vision},
|
|
91
|
+
year={2016}
|
|
92
|
+
}
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## License
|
|
96
|
+
|
|
97
|
+
MIT License - see LICENSE file for details.
|
|
98
|
+
|
|
99
|
+
## Links
|
|
100
|
+
|
|
101
|
+
- GitHub: https://github.com/yourusername/syncnet-python
|
|
102
|
+
- Documentation: https://syncnet-python.readthedocs.io
|
|
103
|
+
- Issues: https://github.com/yourusername/syncnet-python/issues
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=65", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "syncnet-python"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "SyncNet: Audio-visual synchronization detection using deep learning"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = {text = "MIT"}
|
|
12
|
+
authors = [
|
|
13
|
+
{name = "SyncNet Python Contributors"},
|
|
14
|
+
]
|
|
15
|
+
maintainers = [
|
|
16
|
+
{name = "SyncNet Python Contributors"},
|
|
17
|
+
]
|
|
18
|
+
keywords = ["audio-visual", "synchronization", "deep-learning", "pytorch", "lip-sync", "video-processing", "computer-vision"]
|
|
19
|
+
classifiers = [
|
|
20
|
+
"Development Status :: 4 - Beta",
|
|
21
|
+
"Intended Audience :: Science/Research",
|
|
22
|
+
"Intended Audience :: Developers",
|
|
23
|
+
"License :: OSI Approved :: MIT License",
|
|
24
|
+
"Operating System :: OS Independent",
|
|
25
|
+
"Programming Language :: Python :: 3",
|
|
26
|
+
"Programming Language :: Python :: 3.9",
|
|
27
|
+
"Programming Language :: Python :: 3.10",
|
|
28
|
+
"Programming Language :: Python :: 3.11",
|
|
29
|
+
"Programming Language :: Python :: 3.12",
|
|
30
|
+
"Programming Language :: Python :: 3.13",
|
|
31
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
32
|
+
"Topic :: Multimedia :: Video",
|
|
33
|
+
"Topic :: Scientific/Engineering :: Image Processing",
|
|
34
|
+
]
|
|
35
|
+
dependencies = [
|
|
36
|
+
"torch>=2.0.0",
|
|
37
|
+
"torchvision>=0.15.0",
|
|
38
|
+
"numpy>=1.24.0",
|
|
39
|
+
"scipy>=1.10.0",
|
|
40
|
+
"pandas>=2.0.0",
|
|
41
|
+
"scenedetect[opencv]>=0.6.0",
|
|
42
|
+
"opencv-contrib-python>=4.8.0",
|
|
43
|
+
"python-speech-features>=0.6",
|
|
44
|
+
"ffmpeg-python>=0.2.0",
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
[project.urls]
|
|
48
|
+
"Homepage" = "https://github.com/nawta/SyncNet_py313"
|
|
49
|
+
"Bug Reports" = "https://github.com/nawta/SyncNet_py313/issues"
|
|
50
|
+
"Source" = "https://github.com/nawta/SyncNet_py313"
|
|
51
|
+
|
|
52
|
+
[project.optional-dependencies]
|
|
53
|
+
dev = [
|
|
54
|
+
"pytest>=7.4.0",
|
|
55
|
+
"pytest-asyncio>=0.21.0",
|
|
56
|
+
"black>=23.0.0",
|
|
57
|
+
"ruff>=0.1.0",
|
|
58
|
+
"mypy>=1.7.0",
|
|
59
|
+
"types-opencv-python",
|
|
60
|
+
"types-scipy",
|
|
61
|
+
]
|
|
62
|
+
|
|
63
|
+
[project.scripts]
|
|
64
|
+
syncnet-python = "syncnet_python.cli:main"
|
|
65
|
+
|
|
66
|
+
[tool.setuptools.packages.find]
|
|
67
|
+
where = ["."]
|
|
68
|
+
include = ["syncnet_python*"]
|
|
69
|
+
exclude = ["tests*", "docs*", "examples*"]
|
|
70
|
+
|
|
71
|
+
[tool.setuptools.package-data]
|
|
72
|
+
syncnet_python = ["*.py", "detectors/*.py", "detectors/s3fd/*.py"]
|
|
73
|
+
|
|
74
|
+
[tool.black]
|
|
75
|
+
line-length = 88
|
|
76
|
+
target-version = ['py313']
|
|
77
|
+
|
|
78
|
+
[tool.ruff]
|
|
79
|
+
target-version = "py313"
|
|
80
|
+
line-length = 88
|
|
81
|
+
select = [
|
|
82
|
+
"E", # pycodestyle errors
|
|
83
|
+
"W", # pycodestyle warnings
|
|
84
|
+
"F", # pyflakes
|
|
85
|
+
"I", # isort
|
|
86
|
+
"B", # flake8-bugbear
|
|
87
|
+
"C4", # flake8-comprehensions
|
|
88
|
+
"UP", # pyupgrade
|
|
89
|
+
]
|
|
90
|
+
|
|
91
|
+
[tool.mypy]
|
|
92
|
+
python_version = "3.13"
|
|
93
|
+
warn_return_any = true
|
|
94
|
+
warn_unused_configs = true
|
|
95
|
+
disallow_untyped_defs = true
|
|
96
|
+
disallow_incomplete_defs = true
|
|
97
|
+
check_untyped_defs = true
|
|
98
|
+
no_implicit_optional = true
|
|
99
|
+
warn_redundant_casts = true
|
|
100
|
+
warn_unused_ignores = true
|
|
101
|
+
warn_no_return = true
|
|
102
|
+
strict_equality = true
|
|
Binary file
|