syncnet-python 0.2.2__py3-none-any.whl → 0.2.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- syncnet_python/__init__.py +10 -10
- syncnet_python/syncnet_pipeline.py +11 -0
- syncnet_python-0.2.3.dist-info/METADATA +204 -0
- {syncnet_python-0.2.2.dist-info → syncnet_python-0.2.3.dist-info}/RECORD +10 -8
- {syncnet_python-0.2.2.dist-info → syncnet_python-0.2.3.dist-info}/WHEEL +1 -1
- {syncnet_python-0.2.2.dist-info → syncnet_python-0.2.3.dist-info}/licenses/LICENSE +2 -1
- syncnet_python-0.2.3.dist-info/licenses/LICENSE-APACHE +202 -0
- syncnet_python-0.2.3.dist-info/licenses/NOTICE +43 -0
- syncnet_python-0.2.2.dist-info/METADATA +0 -216
- {syncnet_python-0.2.2.dist-info → syncnet_python-0.2.3.dist-info}/entry_points.txt +0 -0
- {syncnet_python-0.2.2.dist-info → syncnet_python-0.2.3.dist-info}/top_level.txt +0 -0
syncnet_python/__init__.py
CHANGED
|
@@ -4,9 +4,11 @@ This package provides a PyTorch implementation of SyncNet for detecting
|
|
|
4
4
|
synchronization between audio and video in multimedia content.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
__version__ = "0.2.
|
|
7
|
+
__version__ = "0.2.3"
|
|
8
8
|
|
|
9
|
-
# Import main components
|
|
9
|
+
# Import main components. An ImportError here (for example a missing
|
|
10
|
+
# dependency or an incompatible scenedetect version) is raised with its
|
|
11
|
+
# original message instead of being replaced by None.
|
|
10
12
|
try:
|
|
11
13
|
from .syncnet_pipeline import SyncNetPipeline
|
|
12
14
|
from .SyncNetModel import S as SyncNetModel
|
|
@@ -16,14 +18,12 @@ try:
|
|
|
16
18
|
extract_audio_from_video,
|
|
17
19
|
calculate_lse_metrics
|
|
18
20
|
)
|
|
19
|
-
except ImportError:
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
extract_audio_from_video = None
|
|
26
|
-
calculate_lse_metrics = None
|
|
21
|
+
except ImportError as e:
|
|
22
|
+
raise ImportError(
|
|
23
|
+
f"syncnet_python could not import its components: {e}. "
|
|
24
|
+
"Check that the dependencies are installed with the versions listed "
|
|
25
|
+
"in pyproject.toml (scenedetect must be >=0.6,<0.7)."
|
|
26
|
+
) from e
|
|
27
27
|
|
|
28
28
|
__all__ = [
|
|
29
29
|
"SyncNetPipeline",
|
|
@@ -1,3 +1,14 @@
|
|
|
1
|
+
# Modified from MoChaBench eval-lipsync/script/syncnet_pipeline.py
|
|
2
|
+
# (https://github.com/congwei1230/MoChaBench, Apache-2.0) by nawta:
|
|
3
|
+
# package-relative imports with a fallback for direct script execution;
|
|
4
|
+
# ffmpeg_bin defaults to "ffmpeg" when None; the two shell
|
|
5
|
+
# subprocess.call ffmpeg commands in _crop became subprocess.run argument
|
|
6
|
+
# lists that raise RuntimeError on failure; added _extract_audio_from_video
|
|
7
|
+
# so inference() accepts audio_path=None; the first three ffmpeg-python
|
|
8
|
+
# calls in inference() raise RuntimeError on ffmpeg.Error; inference()
|
|
9
|
+
# raises RuntimeError when no frames are extracted.
|
|
10
|
+
# See NOTICE for licensing.
|
|
11
|
+
|
|
1
12
|
import json
|
|
2
13
|
import logging
|
|
3
14
|
import os
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: syncnet-python
|
|
3
|
+
Version: 0.2.3
|
|
4
|
+
Summary: SyncNet: Audio-visual synchronization detection using deep learning. Updated version of https://github.com/joonson/syncnet_python for modern Python versions.
|
|
5
|
+
Author: SyncNet Python Contributors
|
|
6
|
+
Maintainer: SyncNet Python Contributors
|
|
7
|
+
License-Expression: MIT AND Apache-2.0
|
|
8
|
+
Project-URL: Homepage, https://github.com/nawta/SyncNet_py309_313
|
|
9
|
+
Project-URL: Repository, https://github.com/nawta/SyncNet_py309_313
|
|
10
|
+
Project-URL: Issues, https://github.com/nawta/SyncNet_py309_313/issues
|
|
11
|
+
Keywords: audio-visual,synchronization,deep-learning,pytorch,lip-sync,video-processing,computer-vision
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
|
+
Classifier: Topic :: Multimedia :: Video
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Image Processing
|
|
25
|
+
Requires-Python: >=3.9
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
License-File: LICENSE-APACHE
|
|
29
|
+
License-File: NOTICE
|
|
30
|
+
Requires-Dist: torch>=2.0.0
|
|
31
|
+
Requires-Dist: torchvision>=0.15.0
|
|
32
|
+
Requires-Dist: numpy>=1.24.0
|
|
33
|
+
Requires-Dist: scipy>=1.10.0
|
|
34
|
+
Requires-Dist: pandas>=2.0.0
|
|
35
|
+
Requires-Dist: scenedetect<0.7,>=0.6.0
|
|
36
|
+
Requires-Dist: opencv-contrib-python>=4.8.0
|
|
37
|
+
Requires-Dist: python-speech-features>=0.6
|
|
38
|
+
Requires-Dist: ffmpeg-python>=0.2.0
|
|
39
|
+
Provides-Extra: dev
|
|
40
|
+
Requires-Dist: pytest>=7.4.0; extra == "dev"
|
|
41
|
+
Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
|
|
42
|
+
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
43
|
+
Requires-Dist: ruff>=0.1.0; extra == "dev"
|
|
44
|
+
Requires-Dist: mypy>=1.7.0; extra == "dev"
|
|
45
|
+
Requires-Dist: types-opencv-python; extra == "dev"
|
|
46
|
+
Requires-Dist: types-scipy; extra == "dev"
|
|
47
|
+
Dynamic: license-file
|
|
48
|
+
|
|
49
|
+
# syncnet-python
|
|
50
|
+
|
|
51
|
+
[](https://badge.fury.io/py/syncnet-python)
|
|
52
|
+
[](https://pypi.org/project/syncnet-python/)
|
|
53
|
+
[](https://github.com/nawta/SyncNet_py309_313/blob/main/NOTICE)
|
|
54
|
+
|
|
55
|
+
A pip-installable SyncNet for Python 3.9 to 3.13 and PyTorch 2. It computes the confidence and minimum distance scores that talking-head and lip-sync papers report as LSE-C and LSE-D.
|
|
56
|
+
|
|
57
|
+
SyncNet (Chung and Zisserman, 2016) is a network that measures how well mouth movements in a video match the speech audio. This package wraps the original SyncNet model and its S3FD face detector in one Python class and one command. Given a video, it finds and tracks faces, crops each face track, and returns three numbers per track: the audio-video offset in frames, the SyncNet confidence (LSE-C, higher means better sync), and the minimum audio-video feature distance (LSE-D, lower means better sync).
|
|
58
|
+
|
|
59
|
+
## Quickstart
|
|
60
|
+
|
|
61
|
+
### 1. Install
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
pip install syncnet-python
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
You also need `ffmpeg` on your `PATH` (`brew install ffmpeg` or `sudo apt-get install ffmpeg`).
|
|
68
|
+
|
|
69
|
+
Use version 0.2.3 or later. Versions 0.2.2 and earlier accept scenedetect 0.7, which removed the `scenedetect.video_manager` module the pipeline imports, and then `from syncnet_python import SyncNetPipeline` silently gives `None`. Version 0.2.3 requires `scenedetect>=0.6,<0.7` and raises an `ImportError` with the underlying message if an import fails. If you must stay on 0.2.2, install it with `pip install syncnet-python==0.2.2 "scenedetect<0.7"`.
|
|
70
|
+
|
|
71
|
+
The package uses `opencv-contrib-python` for `cv2`. Since 0.2.3 it is the only OpenCV package the install pulls in. Versions up to 0.2.2 also installed `opencv-python` through `scenedetect[opencv]`; if you upgrade an old environment and `cv2` fails to import, run `pip uninstall opencv-python opencv-contrib-python` and then `pip install opencv-contrib-python`.
|
|
72
|
+
|
|
73
|
+
### 2. Download the weights
|
|
74
|
+
|
|
75
|
+
The package does not include the model weights. Both files come from the Oxford VGG SyncNet page, the same URLs that the original repository's `download_model.sh` uses:
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
mkdir -p weights
|
|
79
|
+
wget https://www.robots.ox.ac.uk/~vgg/software/lipsync/data/syncnet_v2.model -O weights/syncnet_v2.model
|
|
80
|
+
wget https://www.robots.ox.ac.uk/~vgg/software/lipsync/data/sfd_face.pth -O weights/sfd_face.pth
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
The same two files are also in the [`weights/`](https://github.com/nawta/SyncNet_py309_313/tree/main/weights) folder of this repository. They are byte-identical to the Oxford files (SHA-256 `961e8696…5442` for `syncnet_v2.model` and `d54a87c2…c491` for `sfd_face.pth`).
|
|
84
|
+
|
|
85
|
+
### 3. Python API
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
from syncnet_python import SyncNetPipeline
|
|
89
|
+
|
|
90
|
+
pipe = SyncNetPipeline(
|
|
91
|
+
{"s3fd_weights": "weights/sfd_face.pth", "syncnet_weights": "weights/syncnet_v2.model"},
|
|
92
|
+
device="cpu", # or "cuda"
|
|
93
|
+
)
|
|
94
|
+
offsets, confs, dists, best_conf, min_dist, s3fd_json, has_face = pipe.inference(
|
|
95
|
+
video_path="clip.mp4",
|
|
96
|
+
audio_path=None, # None uses the video's own audio track; or pass a .wav path
|
|
97
|
+
)
|
|
98
|
+
print("AV offset (frames):", offsets[0])
|
|
99
|
+
print("LSE-C (confidence):", round(best_conf, 3))
|
|
100
|
+
print("LSE-D (min distance):", round(min_dist, 3))
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
`offsets`, `confs` and `dists` are lists with one entry per face track. `best_conf` is the largest confidence over all tracks and `min_dist` is the smallest distance over all tracks. For a video with one face these equal `confs[0]` and `dists[0]`. With several faces the two values can come from different tracks, so use the per-track lists if you need scores for one speaker.
|
|
104
|
+
|
|
105
|
+
`syncnet_python.calculate_lse_metrics(pipe, video_path)` returns `(lse_c, lse_d, quality_label)` from the same values.
|
|
106
|
+
|
|
107
|
+
### 4. Command line
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
syncnet-python clip.mp4 --device cpu -o results.json
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
The command reads the weights from `weights/` in the current directory by default; change this with `--s3fd-weights` and `--syncnet-weights`. It prints the offset and confidence for each video and writes offset, confidence and minimum distance to the JSON file. The default device is `cuda`, so pass `--device cpu` on a machine without an NVIDIA GPU.
|
|
114
|
+
|
|
115
|
+
### Tested setup
|
|
116
|
+
|
|
117
|
+
The examples above were run on 2026-10-06 on an Apple Silicon Mac (CPU), using [`example/video.avi`](https://github.com/nawta/SyncNet_py309_313/blob/main/example/video.avi) converted to MP4. Each run used a fresh environment. The OpenCV column gives the version of `opencv-contrib-python`. In the 0.2.3 runs it was the only OpenCV package installed; in the 0.2.2 run `opencv-python` was also installed at the same version.
|
|
118
|
+
|
|
119
|
+
| Package | Python | PyTorch | NumPy | OpenCV | scenedetect | Offset | LSE-C | LSE-D |
|
|
120
|
+
|---|---|---|---|---|---|---|---|---|
|
|
121
|
+
| 0.2.3 | 3.13 (arm64) | 2.14.1 | 2.5.3 | 5.0.0 | 0.6.7.1 | 1 | 4.529 | 9.237 |
|
|
122
|
+
| 0.2.3 | 3.10 (arm64) | 2.14.1 | 2.2.6 | 5.0.0 | 0.6.7.1 | 1 | 4.529 | 9.237 |
|
|
123
|
+
| 0.2.2 | 3.9 (x86_64 under Rosetta 2) | 2.2.2 | 1.26.4 | 4.11.0 | 0.6.7.1 | 1 | 4.524 | 9.291 |
|
|
124
|
+
|
|
125
|
+
The 0.2.3 rows used the wheel built from this repository with a plain install. The Python API, `calculate_lse_metrics` and the CLI gave the same values. The Python 3.9 interpreter was an x86_64 build running under Rosetta 2. PyTorch 2.2.2 is the newest release with macOS x86_64 wheels, and it does not work with NumPy 2, so that run needed `numpy<2`, `opencv-python<4.12` and `opencv-contrib-python<4.12` installed by hand. The full run on the 5.3-second clip took about 10 seconds on CPU.
|
|
126
|
+
|
|
127
|
+
## Changes in 0.2.3
|
|
128
|
+
|
|
129
|
+
- Requires `scenedetect>=0.6,<0.7`, so a plain `pip install syncnet-python` works again.
|
|
130
|
+
- A failed import inside the package raises `ImportError` with the original message. Earlier versions set `SyncNetPipeline` and the other exports to `None`.
|
|
131
|
+
- The repository now holds the 0.2.2 code that was published on PyPI but not committed (`calculate_lse_metrics`, `audio_path=None`, `ffmpeg` error handling).
|
|
132
|
+
- Package metadata declares the license as `MIT AND Apache-2.0` and ships `LICENSE`, `LICENSE-APACHE` and `NOTICE`.
|
|
133
|
+
|
|
134
|
+
See [CHANGELOG.md](https://github.com/nawta/SyncNet_py309_313/blob/main/CHANGELOG.md) for earlier versions.
|
|
135
|
+
|
|
136
|
+
## Comparison with joonson/syncnet_python
|
|
137
|
+
|
|
138
|
+
The original repository, [joonson/syncnet_python](https://github.com/joonson/syncnet_python), was updated by its author on 2026-04-17 (PR #78). This table compares that version with this package.
|
|
139
|
+
|
|
140
|
+
| | joonson/syncnet_python (2026-04) | syncnet-python 0.2.3 |
|
|
141
|
+
|---|---|---|
|
|
142
|
+
| Install | clone, then `conda env create -f environment.yml`; no `setup.py` or `pyproject.toml` | `pip install syncnet-python` |
|
|
143
|
+
| Python | 3.10 (pinned in `environment.yml`) | 3.9 to 3.13 (3.9, 3.10 and 3.13 tested above) |
|
|
144
|
+
| PyTorch | 2.5.1 (pinned) | `torch>=2.0.0` |
|
|
145
|
+
| scenedetect | 0.6.7.1 (pinned) | `>=0.6,<0.7` |
|
|
146
|
+
| Python API | `SyncNetInstance.evaluate()` and `extract_feature()` score a pre-cropped face clip; face detection, tracking and cropping run only through `run_pipeline.py` | `SyncNetPipeline(...).inference(video_path, audio_path)` runs face detection, tracking, cropping and scoring in one call |
|
|
147
|
+
| Command line | `run_pipeline.py`, `run_syncnet.py`, `run_visualise.py` run in sequence, plus `demo_syncnet.py` for pre-cropped clips | one `syncnet-python` command that runs detection, tracking, cropping and scoring |
|
|
148
|
+
| Output | offset, minimum distance and confidence written to the log; per-frame distances saved as `activesd.pckl` under `--data_dir` | values returned to Python, or written to JSON by the CLI |
|
|
149
|
+
| Weights | downloaded by `download_model.sh` | downloaded separately (see above) |
|
|
150
|
+
| Visualisation of the result | `run_visualise.py` | not included |
|
|
151
|
+
|
|
152
|
+
## Errors this fixes
|
|
153
|
+
|
|
154
|
+
Before the April 2026 update, the original repository pinned `scenedetect==0.5.1` and used `np.int`. Users hit these two errors in `run_pipeline.py`:
|
|
155
|
+
|
|
156
|
+
```
|
|
157
|
+
TypeError: 'tuple' object does not support item assignment
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
This comes from scenedetect 0.5.1's `ContentDetector` running with newer OpenCV (upstream issues [#55](https://github.com/joonson/syncnet_python/issues/55) and [#69](https://github.com/joonson/syncnet_python/issues/69)). This package uses scenedetect 0.6.
|
|
161
|
+
|
|
162
|
+
```
|
|
163
|
+
AttributeError: module 'numpy' has no attribute 'int'.
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
This comes from `.astype(np.int)` in `detectors/s3fd/box_utils.py`. NumPy 1.24 removed `np.int`. This package uses `.astype(int)`.
|
|
167
|
+
|
|
168
|
+
## Repository layout
|
|
169
|
+
|
|
170
|
+
The PyPI package contains only the `syncnet_python/` folder. Most of its files (`syncnet_pipeline.py`, `SyncNetInstance.py`, `SyncNetModel.py`, `detectors/` and the scripts named `run_syncnet_pipeline_on_*.py`) come from the SyncNet evaluation code in [MoChaBench](https://github.com/congwei1230/MoChaBench) (`eval-lipsync/script/`), which builds on the original repository. This package adds error handling around the `ffmpeg` calls in `syncnet_pipeline.py`, plus `cli.py`.
|
|
171
|
+
|
|
172
|
+
The `syncnet/` folder holds a separate refactor with configuration files, logging and batch helpers. It is in this repository only and pip does not install it. `scripts/` has example scripts, and `example/` has a short test video with its audio.
|
|
173
|
+
|
|
174
|
+
## Credits
|
|
175
|
+
|
|
176
|
+
The SyncNet model, its pretrained weights and the original code are by Joon Son Chung and Andrew Zisserman ([joonson/syncnet_python](https://github.com/joonson/syncnet_python), [project page](https://www.robots.ox.ac.uk/~vgg/software/lipsync/)). The S3FD face detector weights (`sfd_face.pth`) are downloaded from the same page. The model, detector and pipeline files in `syncnet_python/` come from [MoChaBench](https://github.com/congwei1230/MoChaBench); [NOTICE](https://github.com/nawta/SyncNet_py309_313/blob/main/NOTICE) lists them.
|
|
177
|
+
|
|
178
|
+
## Citation
|
|
179
|
+
|
|
180
|
+
If you use this code in your research, please cite the original paper:
|
|
181
|
+
|
|
182
|
+
```bibtex
|
|
183
|
+
@InProceedings{Chung16a,
|
|
184
|
+
author = "Chung, J.~S. and Zisserman, A.",
|
|
185
|
+
title = "Out of time: automated lip sync in the wild",
|
|
186
|
+
booktitle = "Workshop on Multi-view Lip-reading, ACCV",
|
|
187
|
+
year = "2016",
|
|
188
|
+
}
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
## License
|
|
192
|
+
|
|
193
|
+
This repository uses two licenses. [NOTICE](https://github.com/nawta/SyncNet_py309_313/blob/main/NOTICE) lists which file falls under which.
|
|
194
|
+
|
|
195
|
+
- The original SyncNet code by Joon Son Chung and the code written for this repository are under the MIT License ([LICENSE](https://github.com/nawta/SyncNet_py309_313/blob/main/LICENSE)).
|
|
196
|
+
- The files taken from MoChaBench are under the Apache License 2.0 ([LICENSE-APACHE](https://github.com/nawta/SyncNet_py309_313/blob/main/LICENSE-APACHE)).
|
|
197
|
+
|
|
198
|
+
The model weights have their own terms. The Oxford VGG SyncNet page says: "The model can be used for research purposes under Creative Commons Attribution License." The MIT and Apache licenses above do not cover the weights.
|
|
199
|
+
|
|
200
|
+
## Links
|
|
201
|
+
|
|
202
|
+
- Source and issues: https://github.com/nawta/SyncNet_py309_313
|
|
203
|
+
- PyPI: https://pypi.org/project/syncnet-python/
|
|
204
|
+
- Original SyncNet: https://github.com/joonson/syncnet_python
|
|
@@ -1,21 +1,23 @@
|
|
|
1
1
|
syncnet_python/SyncNetInstance.py,sha256=V-RtWL4nN2CW5n56DkE0d6hcnet4hAs_OVMaMS1ih20,6227
|
|
2
2
|
syncnet_python/SyncNetModel.py,sha256=6qk27paoyV39MTsVyu6K2sbbM-1SLU_2N1zbqm1eeYs,3575
|
|
3
|
-
syncnet_python/__init__.py,sha256=
|
|
3
|
+
syncnet_python/__init__.py,sha256=4OhqOByDwnM2ZhqaE2y4L02kPr5VRhhCEVSLck9QxSs,1241
|
|
4
4
|
syncnet_python/cli.py,sha256=YSQVVDChLkQ96gFRFqD7laQXqbrLC2b8SKaF-vc8FuA,3256
|
|
5
5
|
syncnet_python/run_syncnet_pipeline_on_1example.py,sha256=7I8_jEgIFw-RUqCHU29LrIWIRP77lTUiH6MePOxsuWQ,1008
|
|
6
6
|
syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py,sha256=HJHGLgKpf21bG27_x1QD3MAhqtHey26Ujxb8s2aK-tA,5928
|
|
7
7
|
syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py,sha256=m-kz_DHhka_DywpBRWdxk-S0hrwqvCR16RiVQ7HUqDk,6003
|
|
8
8
|
syncnet_python/safe_syncnet_utils.py,sha256=viucLggg13AW5xAjns_s-jlqv4rzvuRED0Ydo99_RbY,5381
|
|
9
|
-
syncnet_python/syncnet_pipeline.py,sha256=
|
|
9
|
+
syncnet_python/syncnet_pipeline.py,sha256=G9msGO9JDzfCkLD4sBQ8SOlk9QfCEPqjb9mbVAl93SA,16410
|
|
10
10
|
syncnet_python/test_error_handling.py,sha256=i5ib1qEjbMUQg913r9plAdXYciQx2jmUjzuoXPvc5f4,4658
|
|
11
11
|
syncnet_python/test_lse_metrics.py,sha256=H_4U2LN-Wussx_KABX5y5EO3QGnaVhKNMuTjLH0zNrE,2884
|
|
12
12
|
syncnet_python/detectors/__init__.py,sha256=WLone-DTbvQUFleY93pmq7xomPDVjda9ATpZYcS6_sA,23
|
|
13
13
|
syncnet_python/detectors/s3fd/__init__.py,sha256=MIJfIEsKFTGh8brmD3fBQg8JZAp57NtaTkY6_I5QZP0,2322
|
|
14
14
|
syncnet_python/detectors/s3fd/box_utils.py,sha256=CWn46LMJKO1cdenI_IWxBbDriO2ueWTCl14Xp28fr-U,7235
|
|
15
15
|
syncnet_python/detectors/s3fd/nets.py,sha256=BYPJq9UJ5bq5Q5RgP1j-NsFzO1gP3N0PYE1X19-4PZw,5891
|
|
16
|
-
syncnet_python-0.2.
|
|
17
|
-
syncnet_python-0.2.
|
|
18
|
-
syncnet_python-0.2.
|
|
19
|
-
syncnet_python-0.2.
|
|
20
|
-
syncnet_python-0.2.
|
|
21
|
-
syncnet_python-0.2.
|
|
16
|
+
syncnet_python-0.2.3.dist-info/licenses/LICENSE,sha256=JMCE5BdW3pk3NVn8ksaNiLxEABgPjlLkW-3ERoLPhug,1170
|
|
17
|
+
syncnet_python-0.2.3.dist-info/licenses/LICENSE-APACHE,sha256=z8d0m5b2O9McPEK1xHG_dWgUBT6EfBDz6wA0F7xSPTA,11358
|
|
18
|
+
syncnet_python-0.2.3.dist-info/licenses/NOTICE,sha256=rtNvKHi6F6xy8MOG15aetzhkOt8DqdPcswsqCb-msLs,2033
|
|
19
|
+
syncnet_python-0.2.3.dist-info/METADATA,sha256=wM7GvQfaJxBJ4JUtl_m8vyX5s2z2tEysI7UCCQq5QMk,13109
|
|
20
|
+
syncnet_python-0.2.3.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
21
|
+
syncnet_python-0.2.3.dist-info/entry_points.txt,sha256=zPJrIl_ekAa31dHZNCoTHiII2k_m5GyMD5BEkMR6Ytk,59
|
|
22
|
+
syncnet_python-0.2.3.dist-info/top_level.txt,sha256=OkvKpxwq9NzQSjjdD571umGrI3meW9kVleOV2N5Ua04,15
|
|
23
|
+
syncnet_python-0.2.3.dist-info/RECORD,,
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
MIT License
|
|
2
2
|
|
|
3
|
-
Copyright (c)
|
|
3
|
+
Copyright (c) 2016-present Joon Son Chung (original SyncNet code)
|
|
4
|
+
Copyright (c) 2024 SyncNet Python 3.13 Contributors (modifications)
|
|
4
5
|
|
|
5
6
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
7
|
of this software and associated documentation files (the "Software"), to deal
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
|
|
2
|
+
Apache License
|
|
3
|
+
Version 2.0, January 2004
|
|
4
|
+
http://www.apache.org/licenses/
|
|
5
|
+
|
|
6
|
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
|
7
|
+
|
|
8
|
+
1. Definitions.
|
|
9
|
+
|
|
10
|
+
"License" shall mean the terms and conditions for use, reproduction,
|
|
11
|
+
and distribution as defined by Sections 1 through 9 of this document.
|
|
12
|
+
|
|
13
|
+
"Licensor" shall mean the copyright owner or entity authorized by
|
|
14
|
+
the copyright owner that is granting the License.
|
|
15
|
+
|
|
16
|
+
"Legal Entity" shall mean the union of the acting entity and all
|
|
17
|
+
other entities that control, are controlled by, or are under common
|
|
18
|
+
control with that entity. For the purposes of this definition,
|
|
19
|
+
"control" means (i) the power, direct or indirect, to cause the
|
|
20
|
+
direction or management of such entity, whether by contract or
|
|
21
|
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
|
22
|
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
|
23
|
+
|
|
24
|
+
"You" (or "Your") shall mean an individual or Legal Entity
|
|
25
|
+
exercising permissions granted by this License.
|
|
26
|
+
|
|
27
|
+
"Source" form shall mean the preferred form for making modifications,
|
|
28
|
+
including but not limited to software source code, documentation
|
|
29
|
+
source, and configuration files.
|
|
30
|
+
|
|
31
|
+
"Object" form shall mean any form resulting from mechanical
|
|
32
|
+
transformation or translation of a Source form, including but
|
|
33
|
+
not limited to compiled object code, generated documentation,
|
|
34
|
+
and conversions to other media types.
|
|
35
|
+
|
|
36
|
+
"Work" shall mean the work of authorship, whether in Source or
|
|
37
|
+
Object form, made available under the License, as indicated by a
|
|
38
|
+
copyright notice that is included in or attached to the work
|
|
39
|
+
(an example is provided in the Appendix below).
|
|
40
|
+
|
|
41
|
+
"Derivative Works" shall mean any work, whether in Source or Object
|
|
42
|
+
form, that is based on (or derived from) the Work and for which the
|
|
43
|
+
editorial revisions, annotations, elaborations, or other modifications
|
|
44
|
+
represent, as a whole, an original work of authorship. For the purposes
|
|
45
|
+
of this License, Derivative Works shall not include works that remain
|
|
46
|
+
separable from, or merely link (or bind by name) to the interfaces of,
|
|
47
|
+
the Work and Derivative Works thereof.
|
|
48
|
+
|
|
49
|
+
"Contribution" shall mean any work of authorship, including
|
|
50
|
+
the original version of the Work and any modifications or additions
|
|
51
|
+
to that Work or Derivative Works thereof, that is intentionally
|
|
52
|
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
|
53
|
+
or by an individual or Legal Entity authorized to submit on behalf of
|
|
54
|
+
the copyright owner. For the purposes of this definition, "submitted"
|
|
55
|
+
means any form of electronic, verbal, or written communication sent
|
|
56
|
+
to the Licensor or its representatives, including but not limited to
|
|
57
|
+
communication on electronic mailing lists, source code control systems,
|
|
58
|
+
and issue tracking systems that are managed by, or on behalf of, the
|
|
59
|
+
Licensor for the purpose of discussing and improving the Work, but
|
|
60
|
+
excluding communication that is conspicuously marked or otherwise
|
|
61
|
+
designated in writing by the copyright owner as "Not a Contribution."
|
|
62
|
+
|
|
63
|
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
|
64
|
+
on behalf of whom a Contribution has been received by Licensor and
|
|
65
|
+
subsequently incorporated within the Work.
|
|
66
|
+
|
|
67
|
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
|
68
|
+
this License, each Contributor hereby grants to You a perpetual,
|
|
69
|
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
|
70
|
+
copyright license to reproduce, prepare Derivative Works of,
|
|
71
|
+
publicly display, publicly perform, sublicense, and distribute the
|
|
72
|
+
Work and such Derivative Works in Source or Object form.
|
|
73
|
+
|
|
74
|
+
3. Grant of Patent License. Subject to the terms and conditions of
|
|
75
|
+
this License, each Contributor hereby grants to You a perpetual,
|
|
76
|
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
|
77
|
+
(except as stated in this section) patent license to make, have made,
|
|
78
|
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
|
79
|
+
where such license applies only to those patent claims licensable
|
|
80
|
+
by such Contributor that are necessarily infringed by their
|
|
81
|
+
Contribution(s) alone or by combination of their Contribution(s)
|
|
82
|
+
with the Work to which such Contribution(s) was submitted. If You
|
|
83
|
+
institute patent litigation against any entity (including a
|
|
84
|
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
|
85
|
+
or a Contribution incorporated within the Work constitutes direct
|
|
86
|
+
or contributory patent infringement, then any patent licenses
|
|
87
|
+
granted to You under this License for that Work shall terminate
|
|
88
|
+
as of the date such litigation is filed.
|
|
89
|
+
|
|
90
|
+
4. Redistribution. You may reproduce and distribute copies of the
|
|
91
|
+
Work or Derivative Works thereof in any medium, with or without
|
|
92
|
+
modifications, and in Source or Object form, provided that You
|
|
93
|
+
meet the following conditions:
|
|
94
|
+
|
|
95
|
+
(a) You must give any other recipients of the Work or
|
|
96
|
+
Derivative Works a copy of this License; and
|
|
97
|
+
|
|
98
|
+
(b) You must cause any modified files to carry prominent notices
|
|
99
|
+
stating that You changed the files; and
|
|
100
|
+
|
|
101
|
+
(c) You must retain, in the Source form of any Derivative Works
|
|
102
|
+
that You distribute, all copyright, patent, trademark, and
|
|
103
|
+
attribution notices from the Source form of the Work,
|
|
104
|
+
excluding those notices that do not pertain to any part of
|
|
105
|
+
the Derivative Works; and
|
|
106
|
+
|
|
107
|
+
(d) If the Work includes a "NOTICE" text file as part of its
|
|
108
|
+
distribution, then any Derivative Works that You distribute must
|
|
109
|
+
include a readable copy of the attribution notices contained
|
|
110
|
+
within such NOTICE file, excluding those notices that do not
|
|
111
|
+
pertain to any part of the Derivative Works, in at least one
|
|
112
|
+
of the following places: within a NOTICE text file distributed
|
|
113
|
+
as part of the Derivative Works; within the Source form or
|
|
114
|
+
documentation, if provided along with the Derivative Works; or,
|
|
115
|
+
within a display generated by the Derivative Works, if and
|
|
116
|
+
wherever such third-party notices normally appear. The contents
|
|
117
|
+
of the NOTICE file are for informational purposes only and
|
|
118
|
+
do not modify the License. You may add Your own attribution
|
|
119
|
+
notices within Derivative Works that You distribute, alongside
|
|
120
|
+
or as an addendum to the NOTICE text from the Work, provided
|
|
121
|
+
that such additional attribution notices cannot be construed
|
|
122
|
+
as modifying the License.
|
|
123
|
+
|
|
124
|
+
You may add Your own copyright statement to Your modifications and
|
|
125
|
+
may provide additional or different license terms and conditions
|
|
126
|
+
for use, reproduction, or distribution of Your modifications, or
|
|
127
|
+
for any such Derivative Works as a whole, provided Your use,
|
|
128
|
+
reproduction, and distribution of the Work otherwise complies with
|
|
129
|
+
the conditions stated in this License.
|
|
130
|
+
|
|
131
|
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
|
132
|
+
any Contribution intentionally submitted for inclusion in the Work
|
|
133
|
+
by You to the Licensor shall be under the terms and conditions of
|
|
134
|
+
this License, without any additional terms or conditions.
|
|
135
|
+
Notwithstanding the above, nothing herein shall supersede or modify
|
|
136
|
+
the terms of any separate license agreement you may have executed
|
|
137
|
+
with Licensor regarding such Contributions.
|
|
138
|
+
|
|
139
|
+
6. Trademarks. This License does not grant permission to use the trade
|
|
140
|
+
names, trademarks, service marks, or product names of the Licensor,
|
|
141
|
+
except as required for reasonable and customary use in describing the
|
|
142
|
+
origin of the Work and reproducing the content of the NOTICE file.
|
|
143
|
+
|
|
144
|
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
|
145
|
+
agreed to in writing, Licensor provides the Work (and each
|
|
146
|
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
|
147
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
|
148
|
+
implied, including, without limitation, any warranties or conditions
|
|
149
|
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
|
150
|
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
|
151
|
+
appropriateness of using or redistributing the Work and assume any
|
|
152
|
+
risks associated with Your exercise of permissions under this License.
|
|
153
|
+
|
|
154
|
+
8. Limitation of Liability. In no event and under no legal theory,
|
|
155
|
+
whether in tort (including negligence), contract, or otherwise,
|
|
156
|
+
unless required by applicable law (such as deliberate and grossly
|
|
157
|
+
negligent acts) or agreed to in writing, shall any Contributor be
|
|
158
|
+
liable to You for damages, including any direct, indirect, special,
|
|
159
|
+
incidental, or consequential damages of any character arising as a
|
|
160
|
+
result of this License or out of the use or inability to use the
|
|
161
|
+
Work (including but not limited to damages for loss of goodwill,
|
|
162
|
+
work stoppage, computer failure or malfunction, or any and all
|
|
163
|
+
other commercial damages or losses), even if such Contributor
|
|
164
|
+
has been advised of the possibility of such damages.
|
|
165
|
+
|
|
166
|
+
9. Accepting Warranty or Additional Liability. While redistributing
|
|
167
|
+
the Work or Derivative Works thereof, You may choose to offer,
|
|
168
|
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
|
169
|
+
or other liability obligations and/or rights consistent with this
|
|
170
|
+
License. However, in accepting such obligations, You may act only
|
|
171
|
+
on Your own behalf and on Your sole responsibility, not on behalf
|
|
172
|
+
of any other Contributor, and only if You agree to indemnify,
|
|
173
|
+
defend, and hold each Contributor harmless for any liability
|
|
174
|
+
incurred by, or claims asserted against, such Contributor by reason
|
|
175
|
+
of your accepting any such warranty or additional liability.
|
|
176
|
+
|
|
177
|
+
END OF TERMS AND CONDITIONS
|
|
178
|
+
|
|
179
|
+
APPENDIX: How to apply the Apache License to your work.
|
|
180
|
+
|
|
181
|
+
To apply the Apache License to your work, attach the following
|
|
182
|
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
|
183
|
+
replaced with your own identifying information. (Don't include
|
|
184
|
+
the brackets!) The text should be enclosed in the appropriate
|
|
185
|
+
comment syntax for the file format. We also recommend that a
|
|
186
|
+
file or class name and description of purpose be included on the
|
|
187
|
+
same "printed page" as the copyright notice for easier
|
|
188
|
+
identification within third-party archives.
|
|
189
|
+
|
|
190
|
+
Copyright [yyyy] [name of copyright owner]
|
|
191
|
+
|
|
192
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
193
|
+
you may not use this file except in compliance with the License.
|
|
194
|
+
You may obtain a copy of the License at
|
|
195
|
+
|
|
196
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
197
|
+
|
|
198
|
+
Unless required by applicable law or agreed to in writing, software
|
|
199
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
200
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
201
|
+
See the License for the specific language governing permissions and
|
|
202
|
+
limitations under the License.
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
syncnet-python
|
|
2
|
+
https://github.com/nawta/SyncNet_py309_313
|
|
3
|
+
|
|
4
|
+
This repository contains code under two licenses.
|
|
5
|
+
|
|
6
|
+
1. MIT License (see LICENSE)
|
|
7
|
+
The original SyncNet code by Joon Son Chung
|
|
8
|
+
(https://github.com/joonson/syncnet_python) and the code written for
|
|
9
|
+
this repository: syncnet_python/__init__.py, syncnet_python/cli.py,
|
|
10
|
+
syncnet_python/safe_syncnet_utils.py,
|
|
11
|
+
syncnet_python/test_error_handling.py,
|
|
12
|
+
syncnet_python/test_lse_metrics.py, and the syncnet/, scripts/ and
|
|
13
|
+
config/ folders.
|
|
14
|
+
|
|
15
|
+
2. Apache License 2.0 (see LICENSE-APACHE)
|
|
16
|
+
The following files come from MoChaBench
|
|
17
|
+
(https://github.com/congwei1230/MoChaBench, folder eval-lipsync/script/),
|
|
18
|
+
which is distributed under the Apache License 2.0. MoChaBench's LICENSE
|
|
19
|
+
file does not name a copyright holder; the files are credited here to
|
|
20
|
+
the MoChaBench authors. MoChaBench's versions are in turn based on the
|
|
21
|
+
original SyncNet code above. All files in this list except
|
|
22
|
+
syncnet_pipeline.py are unchanged copies of MoChaBench's files.
|
|
23
|
+
|
|
24
|
+
syncnet_python/syncnet_pipeline.py (modified by nawta; the file
|
|
25
|
+
header lists the changes)
|
|
26
|
+
syncnet_python/SyncNetInstance.py
|
|
27
|
+
syncnet_python/SyncNetModel.py
|
|
28
|
+
syncnet_python/detectors/__init__.py
|
|
29
|
+
syncnet_python/detectors/s3fd/__init__.py
|
|
30
|
+
syncnet_python/detectors/s3fd/box_utils.py
|
|
31
|
+
syncnet_python/detectors/s3fd/nets.py
|
|
32
|
+
syncnet_python/run_syncnet_pipeline_on_1example.py
|
|
33
|
+
syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py
|
|
34
|
+
syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py
|
|
35
|
+
|
|
36
|
+
Model weights
|
|
37
|
+
|
|
38
|
+
weights/syncnet_v2.model and weights/sfd_face.pth are copies of the files
|
|
39
|
+
published by the Visual Geometry Group, University of Oxford, at
|
|
40
|
+
https://www.robots.ox.ac.uk/~vgg/software/lipsync/. That page states:
|
|
41
|
+
"The model can be used for research purposes under Creative Commons
|
|
42
|
+
Attribution License." Neither the MIT License nor the Apache License 2.0
|
|
43
|
+
above applies to these files.
|
|
@@ -1,216 +0,0 @@
|
|
|
1
|
-
Metadata-Version: 2.4
|
|
2
|
-
Name: syncnet-python
|
|
3
|
-
Version: 0.2.2
|
|
4
|
-
Summary: SyncNet: Audio-visual synchronization detection using deep learning. Updated version of https://github.com/joonson/syncnet_python for modern Python versions.
|
|
5
|
-
Author: SyncNet Python Contributors
|
|
6
|
-
Maintainer: SyncNet Python Contributors
|
|
7
|
-
License: MIT
|
|
8
|
-
Project-URL: Homepage, https://github.com/nawta/SyncNet_py313
|
|
9
|
-
Project-URL: Bug Reports, https://github.com/nawta/SyncNet_py313/issues
|
|
10
|
-
Project-URL: Source, https://github.com/nawta/SyncNet_py313
|
|
11
|
-
Keywords: audio-visual,synchronization,deep-learning,pytorch,lip-sync,video-processing,computer-vision
|
|
12
|
-
Classifier: Development Status :: 4 - Beta
|
|
13
|
-
Classifier: Intended Audience :: Science/Research
|
|
14
|
-
Classifier: Intended Audience :: Developers
|
|
15
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
-
Classifier: Operating System :: OS Independent
|
|
17
|
-
Classifier: Programming Language :: Python :: 3
|
|
18
|
-
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
-
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
-
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
-
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
-
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
-
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
-
Classifier: Topic :: Multimedia :: Video
|
|
25
|
-
Classifier: Topic :: Scientific/Engineering :: Image Processing
|
|
26
|
-
Requires-Python: >=3.9
|
|
27
|
-
Description-Content-Type: text/markdown
|
|
28
|
-
License-File: LICENSE
|
|
29
|
-
Requires-Dist: torch>=2.0.0
|
|
30
|
-
Requires-Dist: torchvision>=0.15.0
|
|
31
|
-
Requires-Dist: numpy>=1.24.0
|
|
32
|
-
Requires-Dist: scipy>=1.10.0
|
|
33
|
-
Requires-Dist: pandas>=2.0.0
|
|
34
|
-
Requires-Dist: scenedetect[opencv]>=0.6.0
|
|
35
|
-
Requires-Dist: opencv-contrib-python>=4.8.0
|
|
36
|
-
Requires-Dist: python-speech-features>=0.6
|
|
37
|
-
Requires-Dist: ffmpeg-python>=0.2.0
|
|
38
|
-
Provides-Extra: dev
|
|
39
|
-
Requires-Dist: pytest>=7.4.0; extra == "dev"
|
|
40
|
-
Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
|
|
41
|
-
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
42
|
-
Requires-Dist: ruff>=0.1.0; extra == "dev"
|
|
43
|
-
Requires-Dist: mypy>=1.7.0; extra == "dev"
|
|
44
|
-
Requires-Dist: types-opencv-python; extra == "dev"
|
|
45
|
-
Requires-Dist: types-scipy; extra == "dev"
|
|
46
|
-
Dynamic: license-file
|
|
47
|
-
|
|
48
|
-
# SyncNet Python
|
|
49
|
-
|
|
50
|
-
[](https://badge.fury.io/py/syncnet-python)
|
|
51
|
-
[](https://pypi.org/project/syncnet-python/)
|
|
52
|
-
[](https://opensource.org/licenses/MIT)
|
|
53
|
-
|
|
54
|
-
Audio-visual synchronization detection using deep learning with modern Python architecture.
|
|
55
|
-
|
|
56
|
-
This is a **refactored and enhanced version** of the original [SyncNet implementation](https://github.com/joonson/syncnet_python) by Joon Son Chung, updated for Python 3.9+ with clean architecture, comprehensive error handling, and performance optimizations.
|
|
57
|
-
|
|
58
|
-
## Overview
|
|
59
|
-
|
|
60
|
-
SyncNet Python is a PyTorch implementation of the SyncNet model, which detects audio-visual synchronization in videos. It can identify lip-sync errors by analyzing the correspondence between mouth movements and spoken audio.
|
|
61
|
-
|
|
62
|
-
## Features
|
|
63
|
-
|
|
64
|
-
### Core Functionality
|
|
65
|
-
- 🎥 **Audio-Visual Sync Detection**: Accurately detect synchronization between audio and video
|
|
66
|
-
- 🔍 **Face Detection**: Automatic face detection and tracking using S3FD
|
|
67
|
-
- 📊 **Detailed Analysis**: Per-crop offsets, confidence scores, and minimum distances
|
|
68
|
-
- 🚀 **Batch Processing**: Process multiple videos efficiently
|
|
69
|
-
- 🐍 **Python API**: Easy-to-use Python interface with proper error handling
|
|
70
|
-
|
|
71
|
-
### Architecture Improvements
|
|
72
|
-
- 🏗️ **Clean Architecture**: Abstract base classes and factory patterns
|
|
73
|
-
- ⚡ **Performance Optimized**: Parallel processing and memory management
|
|
74
|
-
- 🛡️ **Robust Error Handling**: Comprehensive exception hierarchy
|
|
75
|
-
- ⚙️ **Configuration Management**: YAML/JSON configuration support
|
|
76
|
-
- 📝 **Advanced Logging**: Structured logging with progress tracking
|
|
77
|
-
- 🔄 **Backward Compatibility**: Maintains compatibility with original API
|
|
78
|
-
|
|
79
|
-
## Installation
|
|
80
|
-
|
|
81
|
-
```bash
|
|
82
|
-
pip install syncnet-python
|
|
83
|
-
```
|
|
84
|
-
|
|
85
|
-
### Additional Requirements
|
|
86
|
-
|
|
87
|
-
1. **FFmpeg**: Required for video processing
|
|
88
|
-
```bash
|
|
89
|
-
# Ubuntu/Debian
|
|
90
|
-
sudo apt-get install ffmpeg
|
|
91
|
-
|
|
92
|
-
# macOS
|
|
93
|
-
brew install ffmpeg
|
|
94
|
-
```
|
|
95
|
-
|
|
96
|
-
2. **Model Weights**: Download pre-trained weights
|
|
97
|
-
- Download `sfd_face.pth` and `syncnet_v2.model`
|
|
98
|
-
- Place them in a `weights/` directory
|
|
99
|
-
|
|
100
|
-
## Quick Start
|
|
101
|
-
|
|
102
|
-
```python
|
|
103
|
-
from syncnet_python import SyncNetPipeline
|
|
104
|
-
|
|
105
|
-
# Initialize pipeline
|
|
106
|
-
pipeline = SyncNetPipeline(
|
|
107
|
-
s3fd_weights="weights/sfd_face.pth",
|
|
108
|
-
syncnet_weights="weights/syncnet_v2.model",
|
|
109
|
-
device="cuda" # or "cpu"
|
|
110
|
-
)
|
|
111
|
-
|
|
112
|
-
# Process video
|
|
113
|
-
results = pipeline.inference(
|
|
114
|
-
video_path="video.mp4",
|
|
115
|
-
audio_path=None # Extract from video
|
|
116
|
-
)
|
|
117
|
-
|
|
118
|
-
# Extract results (returns tuple)
|
|
119
|
-
offset_list, confidence_list, min_dist_list, best_confidence, best_min_dist, detections_json, success = results
|
|
120
|
-
|
|
121
|
-
# Get best results
|
|
122
|
-
offset = offset_list[0] # AV offset in frames
|
|
123
|
-
confidence = confidence_list[0] # Confidence score
|
|
124
|
-
min_distance = min_dist_list[0] # Minimum distance
|
|
125
|
-
|
|
126
|
-
print(f"AV Offset: {offset} frames")
|
|
127
|
-
print(f"Confidence: {confidence:.3f}")
|
|
128
|
-
print(f"Min Distance: {min_distance:.3f}")
|
|
129
|
-
```
|
|
130
|
-
|
|
131
|
-
### Detailed Analysis
|
|
132
|
-
|
|
133
|
-
```python
|
|
134
|
-
# For detailed per-crop analysis
|
|
135
|
-
for i, (offset, conf, dist) in enumerate(zip(offset_list, confidence_list, min_dist_list)):
|
|
136
|
-
print(f"Crop {i+1}: offset={offset}, confidence={conf:.3f}, min_dist={dist:.3f}")
|
|
137
|
-
|
|
138
|
-
# Parse face detections
|
|
139
|
-
import json
|
|
140
|
-
detections = json.loads(detections_json)
|
|
141
|
-
print(f"Total frames with face detection: {len(detections)}")
|
|
142
|
-
```
|
|
143
|
-
|
|
144
|
-
## Command Line Usage
|
|
145
|
-
|
|
146
|
-
```bash
|
|
147
|
-
# Process single video
|
|
148
|
-
syncnet-python video.mp4
|
|
149
|
-
|
|
150
|
-
# Process multiple videos
|
|
151
|
-
syncnet-python video1.mp4 video2.mp4 --output results.json
|
|
152
|
-
|
|
153
|
-
# Use CPU instead of GPU
|
|
154
|
-
syncnet-python video.mp4 --device cpu
|
|
155
|
-
```
|
|
156
|
-
|
|
157
|
-
## Performance
|
|
158
|
-
|
|
159
|
-
Tested with example files:
|
|
160
|
-
- **Processing Speed**: 191.4 fps
|
|
161
|
-
- **Face Detection**: 100% success rate
|
|
162
|
-
- **Accuracy**: Detects 1-frame offsets with high confidence (4.5+)
|
|
163
|
-
- **Compute Time**: ~0.65 seconds for 134 frames
|
|
164
|
-
|
|
165
|
-
## Architecture
|
|
166
|
-
|
|
167
|
-
### Refactored Core Modules
|
|
168
|
-
- `syncnet/core/` - Modern refactored implementation
|
|
169
|
-
- `base.py` - Abstract base classes and interfaces
|
|
170
|
-
- `models.py` - Enhanced SyncNet model with factory pattern
|
|
171
|
-
- `audio.py` - MFCC audio processing with streaming support
|
|
172
|
-
- `video.py` - Parallel video processing with OpenCV
|
|
173
|
-
- `sync_analyzer.py` - Optimized sync analysis with caching
|
|
174
|
-
- `config.py` - Configuration management system
|
|
175
|
-
- `exceptions.py` - Comprehensive error handling
|
|
176
|
-
- `logging.py` - Advanced logging with progress tracking
|
|
177
|
-
- `utils.py` - Memory management and utility functions
|
|
178
|
-
|
|
179
|
-
### Legacy Compatibility
|
|
180
|
-
- `syncnet_python/` - Maintains original API compatibility
|
|
181
|
-
- Full backward compatibility with existing code
|
|
182
|
-
|
|
183
|
-
## Requirements
|
|
184
|
-
|
|
185
|
-
- Python 3.9+ (tested on 3.13)
|
|
186
|
-
- PyTorch 2.0+
|
|
187
|
-
- CUDA (optional but recommended)
|
|
188
|
-
- FFmpeg
|
|
189
|
-
- Additional dependencies: OpenCV, SciPy, NumPy, pandas
|
|
190
|
-
|
|
191
|
-
## Credits
|
|
192
|
-
|
|
193
|
-
This package is based on the original [SyncNet implementation](https://github.com/joonson/syncnet_python) by Joon Son Chung, enhanced with modern Python architecture and performance optimizations.
|
|
194
|
-
|
|
195
|
-
## Citation
|
|
196
|
-
|
|
197
|
-
If you use this code in your research, please cite the original paper:
|
|
198
|
-
|
|
199
|
-
```bibtex
|
|
200
|
-
@inproceedings{chung2016out,
|
|
201
|
-
title={Out of time: automated lip sync in the wild},
|
|
202
|
-
author={Chung, Joon Son and Zisserman, Andrew},
|
|
203
|
-
booktitle={Asian Conference on Computer Vision},
|
|
204
|
-
year={2016}
|
|
205
|
-
}
|
|
206
|
-
```
|
|
207
|
-
|
|
208
|
-
## License
|
|
209
|
-
|
|
210
|
-
MIT License - see LICENSE file for details.
|
|
211
|
-
|
|
212
|
-
## Links
|
|
213
|
-
|
|
214
|
-
- GitHub: https://github.com/yourusername/syncnet-python
|
|
215
|
-
- Documentation: https://syncnet-python.readthedocs.io
|
|
216
|
-
- Issues: https://github.com/yourusername/syncnet-python/issues
|
|
File without changes
|
|
File without changes
|