syncnet-python 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- syncnet_python/SyncNetInstance.py +210 -0
- syncnet_python/SyncNetModel.py +99 -0
- syncnet_python/__init__.py +29 -0
- syncnet_python/cli.py +127 -0
- syncnet_python/detectors/__init__.py +1 -0
- syncnet_python/detectors/s3fd/__init__.py +66 -0
- syncnet_python/detectors/s3fd/box_utils.py +233 -0
- syncnet_python/detectors/s3fd/nets.py +177 -0
- syncnet_python/run_syncnet_pipeline_on_1example.py +28 -0
- syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py +157 -0
- syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py +158 -0
- syncnet_python/syncnet_pipeline.py +332 -0
- syncnet_python-0.1.0.dist-info/METADATA +150 -0
- syncnet_python-0.1.0.dist-info/RECORD +18 -0
- syncnet_python-0.1.0.dist-info/WHEEL +5 -0
- syncnet_python-0.1.0.dist-info/entry_points.txt +2 -0
- syncnet_python-0.1.0.dist-info/licenses/LICENSE +21 -0
- syncnet_python-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
#!/usr/bin/env python
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""
|
|
4
|
+
Batch SyncNet evaluation for MoChaBench.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import csv
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
import time
|
|
11
|
+
from collections import defaultdict
|
|
12
|
+
|
|
13
|
+
import pandas as pd
|
|
14
|
+
from syncnet_pipeline import SyncNetPipeline
|
|
15
|
+
|
|
16
|
+
# ------------------------------------------------------------------ #
|
|
17
|
+
# 1) paths & config #
|
|
18
|
+
# ------------------------------------------------------------------ #
|
|
19
|
+
|
|
20
|
+
# folder that contains *this* script
|
|
21
|
+
ROOT = Path(r"/full/path/to/MoChaBench") # <<< fill with your own repo path/MoChaBench
|
|
22
|
+
|
|
23
|
+
# ---- adapt these if you move folders around ---------------------- #
|
|
24
|
+
BASE_VIDEO = ROOT / "your model output videos" # <<< fill with your own path e.g. …/MoChaBench/xxx-model-out
|
|
25
|
+
BASE_BENCHMARK = ROOT / "benchmark" # parent folder that contains 'speeches/'
|
|
26
|
+
CSV_FILE = BASE_BENCHMARK / "benchmark.csv"
|
|
27
|
+
# ------------------------------------------------------------------ #
|
|
28
|
+
|
|
29
|
+
OUT_CSV = ROOT / "eval-lipsync" / "your own model-eval-results" / "sync_scores.csv" # <<< fill with your own path
|
|
30
|
+
OUT_JSON = ROOT / "eval-lipsync"/ "your own model-eval-results" / "sync_scores.json" # <<< fill with your own path
|
|
31
|
+
|
|
32
|
+
pipe = SyncNetPipeline(
|
|
33
|
+
{
|
|
34
|
+
"s3fd_weights": ROOT / "eval-lipsync" / "weights" / "sfd_face.pth",
|
|
35
|
+
"syncnet_weights": ROOT / "eval-lipsync" /"weights" / "syncnet_v2.model",
|
|
36
|
+
},
|
|
37
|
+
device="cuda",
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
# category buckets for your extra means
|
|
41
|
+
ENGLISH_1P_CATEGORIES = {
|
|
42
|
+
"1p_closeup_facingcamera",
|
|
43
|
+
"1p_camera_movement",
|
|
44
|
+
"1p_emotion",
|
|
45
|
+
"1p_mediumshot_actioncontrol",
|
|
46
|
+
"2p_1clip_1talk",
|
|
47
|
+
"1p_protrait",
|
|
48
|
+
}
|
|
49
|
+
TURNTALK_CATEGORY = {"2p_2clip_2talk"}
|
|
50
|
+
|
|
51
|
+
# ------------------------------------------------------------------ #
|
|
52
|
+
# 2) helper: run one sample #
|
|
53
|
+
# ------------------------------------------------------------------ #
|
|
54
|
+
def run_sample(row):
|
|
55
|
+
idx = int(row["idx_in_category"])
|
|
56
|
+
cat = row["category"].strip()
|
|
57
|
+
base_name = row["context_id"].strip() # e.g. 1_man_bag_of_gold
|
|
58
|
+
id = f"{cat}_{base_name}"
|
|
59
|
+
|
|
60
|
+
video_fp = BASE_VIDEO / cat / f"{base_name}.mp4"
|
|
61
|
+
audio_fp = BASE_BENCHMARK / "speeches" / cat / f"{base_name}_speech.wav"
|
|
62
|
+
|
|
63
|
+
if not video_fp.exists():
|
|
64
|
+
raise FileNotFoundError(f"Video not found: {video_fp}")
|
|
65
|
+
if not audio_fp.exists():
|
|
66
|
+
raise FileNotFoundError(f"Audio not found: {audio_fp}")
|
|
67
|
+
|
|
68
|
+
t0 = time.time()
|
|
69
|
+
off, confs, dists, best_conf, min_dist, _, has_face = pipe.inference(
|
|
70
|
+
video_path=str(video_fp),
|
|
71
|
+
audio_path=str(audio_fp),
|
|
72
|
+
cache_dir= ROOT / "eval-lipsync"/ "mocha-eval-results" / "cache" / id
|
|
73
|
+
)
|
|
74
|
+
return {
|
|
75
|
+
"idx": idx,
|
|
76
|
+
"category": cat,
|
|
77
|
+
"video": str(Path(cat) / f"{base_name}.mp4"),
|
|
78
|
+
"audio": str(Path(cat) / f"{base_name}_speech.wav"),
|
|
79
|
+
"offsets": [int(o) for o in off],
|
|
80
|
+
"best_conf": float(best_conf),
|
|
81
|
+
"min_dist": float(min_dist),
|
|
82
|
+
"has_face": has_face,
|
|
83
|
+
"runtime_s": round(time.time() - t0, 2),
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# ------------------------------------------------------------------ #
|
|
88
|
+
# 3) main loop #
|
|
89
|
+
# ------------------------------------------------------------------ #
|
|
90
|
+
def main():
|
|
91
|
+
df = pd.read_csv(CSV_FILE)
|
|
92
|
+
results = []
|
|
93
|
+
for _, row in df.iterrows():
|
|
94
|
+
try:
|
|
95
|
+
res = run_sample(row)
|
|
96
|
+
results.append(res)
|
|
97
|
+
print(
|
|
98
|
+
f"[{res['idx']}] {res['category']} "
|
|
99
|
+
f"Δ={res['offsets']} conf={res['best_conf']:.3f} "
|
|
100
|
+
f"dist={res['min_dist']:.3f}"
|
|
101
|
+
)
|
|
102
|
+
except Exception as e:
|
|
103
|
+
print(f"[{row['idx']}] ERROR – {e}")
|
|
104
|
+
|
|
105
|
+
# ------------------------------------------------------------------ #
|
|
106
|
+
# 4) save full table #
|
|
107
|
+
# ------------------------------------------------------------------ #
|
|
108
|
+
pd.DataFrame(results).to_csv(OUT_CSV, index=False)
|
|
109
|
+
with open(OUT_JSON, "w") as f:
|
|
110
|
+
json.dump(results, f, indent=2)
|
|
111
|
+
|
|
112
|
+
# ------------------------------------------------------------------ #
|
|
113
|
+
# 5) category aggregates #
|
|
114
|
+
# ------------------------------------------------------------------ #
|
|
115
|
+
# Separate stats for each category (only when face is detected)
|
|
116
|
+
cat_dists = defaultdict(list)
|
|
117
|
+
cat_confs = defaultdict(list)
|
|
118
|
+
|
|
119
|
+
for r in results:
|
|
120
|
+
# !!!! sometimes SyncNetPipeline fails to detect faces, we should not inlcude those
|
|
121
|
+
if r["has_face"]:
|
|
122
|
+
cat_dists[r["category"]].append(r["min_dist"])
|
|
123
|
+
cat_confs[r["category"]].append(r["best_conf"])
|
|
124
|
+
|
|
125
|
+
def mean(vals):
|
|
126
|
+
return sum(vals) / len(vals) if vals else float("nan")
|
|
127
|
+
|
|
128
|
+
print("\n=== per-category averages (only if face detected) ===")
|
|
129
|
+
for cat in sorted(set(cat_dists.keys()) | set(cat_confs.keys())):
|
|
130
|
+
avg_dist = mean(cat_dists[cat])
|
|
131
|
+
avg_conf = mean(cat_confs[cat])
|
|
132
|
+
print(f"{cat:30s}: dist={avg_dist:.3f} conf={avg_conf:.3f} (n={len(cat_dists[cat])})")
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
# super-sets
|
|
136
|
+
english_dists = [
|
|
137
|
+
d for c, v in cat_dists.items() if c in ENGLISH_1P_CATEGORIES for d in v
|
|
138
|
+
]
|
|
139
|
+
english_confs = [
|
|
140
|
+
c for cat, v in cat_confs.items() if cat in ENGLISH_1P_CATEGORIES for c in v
|
|
141
|
+
]
|
|
142
|
+
|
|
143
|
+
dialog_dists = [
|
|
144
|
+
d for c, v in cat_dists.items() if c in TURNTALK_CATEGORY for d in v
|
|
145
|
+
]
|
|
146
|
+
dialog_confs = [
|
|
147
|
+
c for cat, v in cat_confs.items() if cat in TURNTALK_CATEGORY for c in v
|
|
148
|
+
]
|
|
149
|
+
print("\n--- aggregate groups (only face-detected entries) ---")
|
|
150
|
+
print(f"single-character English (1p*): dist={mean(english_dists):.3f} conf={mean(english_confs):.3f}")
|
|
151
|
+
print(f"turn-based dialogue English (2p_2clip): dist={mean(dialog_dists):.3f} conf={mean(dialog_confs):.3f}")
|
|
152
|
+
|
|
153
|
+
print(f"\nSaved detailed table → {OUT_CSV}")
|
|
154
|
+
print(f"Saved JSON dump → {OUT_JSON}")
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
if __name__ == "__main__":
|
|
158
|
+
main()
|
|
@@ -0,0 +1,332 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import logging
|
|
3
|
+
import os
|
|
4
|
+
import shutil
|
|
5
|
+
import subprocess
|
|
6
|
+
import tempfile
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from glob import glob
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any, Dict, List, Optional, Tuple, Union
|
|
11
|
+
|
|
12
|
+
import cv2
|
|
13
|
+
import ffmpeg
|
|
14
|
+
import numpy as np
|
|
15
|
+
import torch
|
|
16
|
+
from scipy import signal
|
|
17
|
+
from scipy.interpolate import interp1d
|
|
18
|
+
from scenedetect import ContentDetector, SceneManager, StatsManager
|
|
19
|
+
from scenedetect.video_manager import VideoManager
|
|
20
|
+
|
|
21
|
+
from detectors.s3fd import S3FD
|
|
22
|
+
from detectors.s3fd.nets import S3FDNet
|
|
23
|
+
from SyncNetInstance import SyncNetInstance
|
|
24
|
+
from SyncNetModel import S
|
|
25
|
+
|
|
26
|
+
# ---------------------------------------------------------------------- #
|
|
27
|
+
# Configuration #
|
|
28
|
+
# ---------------------------------------------------------------------- #
|
|
29
|
+
@dataclass
|
|
30
|
+
class PipelineConfig:
|
|
31
|
+
# Face-detection / tracking
|
|
32
|
+
facedet_scale: float = 0.25
|
|
33
|
+
crop_scale: float = 0.40
|
|
34
|
+
min_track: int = 50
|
|
35
|
+
frame_rate: int = 25
|
|
36
|
+
num_failed_det: int = 25
|
|
37
|
+
min_face_size: int = 100
|
|
38
|
+
|
|
39
|
+
# SyncNet
|
|
40
|
+
batch_size: int = 20
|
|
41
|
+
vshift: int = 15
|
|
42
|
+
|
|
43
|
+
# Local weight paths
|
|
44
|
+
s3fd_weights: str = "sfd_face.pth"
|
|
45
|
+
syncnet_weights: str = "syncnet_v2.model"
|
|
46
|
+
|
|
47
|
+
# Tools
|
|
48
|
+
ffmpeg_bin: str = "ffmpeg" # assumes ffmpeg in $PATH
|
|
49
|
+
audio_sample_rate: int = 16000 # resample rate for speech
|
|
50
|
+
|
|
51
|
+
@classmethod
|
|
52
|
+
def from_dict(cls, d: Dict[str, Any]):
|
|
53
|
+
return cls(**{k: v for k, v in d.items() if k in cls.__annotations__})
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# ---------------------------------------------------------------------- #
|
|
57
|
+
# Pipeline #
|
|
58
|
+
# ---------------------------------------------------------------------- #
|
|
59
|
+
class SyncNetPipeline:
|
|
60
|
+
def __init__(
|
|
61
|
+
self,
|
|
62
|
+
cfg: Union[PipelineConfig, Dict[str, Any], None] = None,
|
|
63
|
+
*,
|
|
64
|
+
device: str = "cuda",
|
|
65
|
+
**override,
|
|
66
|
+
):
|
|
67
|
+
base = cfg if isinstance(cfg, PipelineConfig) else PipelineConfig.from_dict(cfg or {})
|
|
68
|
+
for k, v in override.items():
|
|
69
|
+
if hasattr(base, k):
|
|
70
|
+
setattr(base, k, v)
|
|
71
|
+
self.cfg = base
|
|
72
|
+
self.device = device
|
|
73
|
+
|
|
74
|
+
self.s3fd = self._load_s3fd(self.cfg.s3fd_weights)
|
|
75
|
+
self.syncnet = self._load_syncnet(self.cfg.syncnet_weights)
|
|
76
|
+
|
|
77
|
+
# ---------------------------- model loading ---------------------------- #
|
|
78
|
+
def _load_s3fd(self, path: str) -> S3FD:
|
|
79
|
+
logging.info(f"Loading S3FD from {path}")
|
|
80
|
+
net = S3FDNet(device=self.device)
|
|
81
|
+
net.load_state_dict(torch.load(path, map_location=self.device))
|
|
82
|
+
net.eval()
|
|
83
|
+
return S3FD(net=net, device=self.device)
|
|
84
|
+
|
|
85
|
+
def _load_syncnet(self, path: str) -> SyncNetInstance:
|
|
86
|
+
logging.info(f"Loading SyncNet from {path}")
|
|
87
|
+
model = S()
|
|
88
|
+
model.load_state_dict(torch.load(path, map_location=self.device))
|
|
89
|
+
model.eval()
|
|
90
|
+
return SyncNetInstance(net=model, device=self.device)
|
|
91
|
+
|
|
92
|
+
# ---------------------------- helpers ---------------------------------- #
|
|
93
|
+
@staticmethod
|
|
94
|
+
def _iou(a, b):
|
|
95
|
+
xA, yA = max(a[0], b[0]), max(a[1], b[1])
|
|
96
|
+
xB, yB = min(a[2], b[2]), min(a[3], b[3])
|
|
97
|
+
inter = max(0, xB - xA) * max(0, yB - yA)
|
|
98
|
+
areaA = (a[2] - a[0]) * (a[3] - a[1])
|
|
99
|
+
areaB = (b[2] - b[0]) * (b[3] - b[1])
|
|
100
|
+
return inter / (areaA + areaB - inter + 1e-8)
|
|
101
|
+
|
|
102
|
+
def _track(self, dets):
|
|
103
|
+
cfg = self.cfg
|
|
104
|
+
tracks = []
|
|
105
|
+
while True:
|
|
106
|
+
t = []
|
|
107
|
+
for faces in dets:
|
|
108
|
+
for f in faces:
|
|
109
|
+
if not t:
|
|
110
|
+
t.append(f)
|
|
111
|
+
faces.remove(f)
|
|
112
|
+
elif (
|
|
113
|
+
f["frame"] - t[-1]["frame"] <= cfg.num_failed_det
|
|
114
|
+
and self._iou(f["bbox"], t[-1]["bbox"]) > 0.5
|
|
115
|
+
):
|
|
116
|
+
t.append(f)
|
|
117
|
+
faces.remove(f)
|
|
118
|
+
continue
|
|
119
|
+
else:
|
|
120
|
+
break
|
|
121
|
+
if not t:
|
|
122
|
+
break
|
|
123
|
+
if len(t) > cfg.min_track:
|
|
124
|
+
fr = np.array([d["frame"] for d in t])
|
|
125
|
+
bb = np.array([d["bbox"] for d in t])
|
|
126
|
+
full_f = np.arange(fr[0], fr[-1] + 1)
|
|
127
|
+
bb_i = np.stack([interp1d(fr, bb[:, i])(full_f) for i in range(4)], 1)
|
|
128
|
+
if max(
|
|
129
|
+
np.mean(bb_i[:, 2] - bb_i[:, 0]),
|
|
130
|
+
np.mean(bb_i[:, 3] - bb_i[:, 1]),
|
|
131
|
+
) > cfg.min_face_size:
|
|
132
|
+
tracks.append({"frame": full_f, "bbox": bb_i})
|
|
133
|
+
return tracks
|
|
134
|
+
|
|
135
|
+
def _crop(self, track, frames, audio_wav, base):
|
|
136
|
+
cfg = self.cfg
|
|
137
|
+
base.parent.mkdir(parents=True, exist_ok=True)
|
|
138
|
+
tmp_avi = f"{base}t.avi"
|
|
139
|
+
vw = cv2.VideoWriter(tmp_avi, cv2.VideoWriter_fourcc(*"XVID"), cfg.frame_rate, (224, 224))
|
|
140
|
+
|
|
141
|
+
s, x, y = [], [], []
|
|
142
|
+
for b in track["bbox"]:
|
|
143
|
+
s.append(max(b[3] - b[1], b[2] - b[0]) / 2)
|
|
144
|
+
x.append((b[0] + b[2]) / 2)
|
|
145
|
+
y.append((b[1] + b[3]) / 2)
|
|
146
|
+
s, x, y = map(lambda v: signal.medfilt(v, 13), (s, x, y))
|
|
147
|
+
|
|
148
|
+
for i, fidx in enumerate(track["frame"]):
|
|
149
|
+
img = cv2.imread(frames[fidx])
|
|
150
|
+
if img is None:
|
|
151
|
+
continue
|
|
152
|
+
bs = s[i]
|
|
153
|
+
cs = cfg.crop_scale
|
|
154
|
+
pad = int(bs * (1 + 2 * cs))
|
|
155
|
+
img_p = cv2.copyMakeBorder(
|
|
156
|
+
img, pad, pad, pad, pad, cv2.BORDER_CONSTANT, value=(110, 110, 110)
|
|
157
|
+
)
|
|
158
|
+
my, mx = y[i] + pad, x[i] + pad
|
|
159
|
+
y1, y2 = int(my - bs), int(my + bs * (1 + 2 * cs))
|
|
160
|
+
x1, x2 = int(mx - bs * (1 + cs)), int(mx + bs * (1 + cs))
|
|
161
|
+
crop = cv2.resize(img_p[y1:y2, x1:x2], (224, 224))
|
|
162
|
+
vw.write(crop)
|
|
163
|
+
vw.release()
|
|
164
|
+
|
|
165
|
+
slice_wav = f"{base}.wav"
|
|
166
|
+
ss = track["frame"][0] / cfg.frame_rate
|
|
167
|
+
to = (track["frame"][-1] + 1) / cfg.frame_rate
|
|
168
|
+
subprocess.call(
|
|
169
|
+
f'{cfg.ffmpeg_bin} -y -i "{audio_wav}" -ss {ss:.3f} -to {to:.3f} "{slice_wav}"',
|
|
170
|
+
shell=True,
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
final_avi = f"{base}.avi"
|
|
174
|
+
subprocess.call(
|
|
175
|
+
f'{cfg.ffmpeg_bin} -y -i "{tmp_avi}" -i "{slice_wav}" -c:v copy -c:a copy "{final_avi}"',
|
|
176
|
+
shell=True,
|
|
177
|
+
)
|
|
178
|
+
os.remove(tmp_avi)
|
|
179
|
+
return final_avi
|
|
180
|
+
|
|
181
|
+
# ---------------------------- inference -------------------------------- #
|
|
182
|
+
def inference(
|
|
183
|
+
self,
|
|
184
|
+
video_path: str, # We do not extract audio from video_path!
|
|
185
|
+
audio_path: str,
|
|
186
|
+
*,
|
|
187
|
+
cache_dir: Optional[str] = None,
|
|
188
|
+
) -> Tuple[List[int], List[float], List[float], float, float, str, bool]:
|
|
189
|
+
cfg = self.cfg
|
|
190
|
+
work = Path(cache_dir) if cache_dir else Path(tempfile.mkdtemp())
|
|
191
|
+
if cache_dir:
|
|
192
|
+
work.mkdir(parents=True, exist_ok=True)
|
|
193
|
+
|
|
194
|
+
try:
|
|
195
|
+
# 1) Convert video to constant-fps AVI
|
|
196
|
+
avi = work / "video.avi"
|
|
197
|
+
(
|
|
198
|
+
ffmpeg.input(video_path)
|
|
199
|
+
.output(str(avi), **{"q:v": 2}, r=cfg.frame_rate, **{"async": 1})
|
|
200
|
+
.overwrite_output()
|
|
201
|
+
.run()
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
# 2) Extract frames
|
|
205
|
+
frames_dir = work / "frames"
|
|
206
|
+
frames_dir.mkdir(exist_ok=True)
|
|
207
|
+
(
|
|
208
|
+
ffmpeg.input(str(avi))
|
|
209
|
+
.output(str(frames_dir / "%06d.jpg"), **{"q:v": 2}, f="image2", threads=1)
|
|
210
|
+
.overwrite_output()
|
|
211
|
+
.run()
|
|
212
|
+
)
|
|
213
|
+
frames = sorted(glob(str(frames_dir / "*.jpg")))
|
|
214
|
+
|
|
215
|
+
# 3) Resample speech
|
|
216
|
+
audio_wav = work / "speech.wav"
|
|
217
|
+
(
|
|
218
|
+
ffmpeg.input(audio_path)
|
|
219
|
+
.output(str(audio_wav), ac=1, ar=cfg.audio_sample_rate, format="wav")
|
|
220
|
+
.overwrite_output()
|
|
221
|
+
.run()
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
# 4) Face detection
|
|
225
|
+
detections = []
|
|
226
|
+
for i, fp in enumerate(frames):
|
|
227
|
+
img = cv2.imread(fp)
|
|
228
|
+
boxes = (
|
|
229
|
+
self.s3fd.detect_faces(
|
|
230
|
+
cv2.cvtColor(img, cv2.COLOR_BGR2RGB),
|
|
231
|
+
conf_th=0.9,
|
|
232
|
+
scales=[cfg.facedet_scale],
|
|
233
|
+
)
|
|
234
|
+
if img is not None
|
|
235
|
+
else []
|
|
236
|
+
)
|
|
237
|
+
detections.append(
|
|
238
|
+
[
|
|
239
|
+
{"frame": i, "bbox": b[:-1].tolist(), "conf": float(b[-1])}
|
|
240
|
+
for b in boxes
|
|
241
|
+
]
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
flat = [f for fs in detections for f in fs]
|
|
245
|
+
s3fd_json = json.dumps(flat) if flat else ""
|
|
246
|
+
has_face = bool(flat)
|
|
247
|
+
|
|
248
|
+
# 5) Scene detection
|
|
249
|
+
vm = VideoManager([str(avi)])
|
|
250
|
+
sm = SceneManager(StatsManager())
|
|
251
|
+
sm.add_detector(ContentDetector())
|
|
252
|
+
vm.start()
|
|
253
|
+
sm.detect_scenes(frame_source=vm)
|
|
254
|
+
scenes = sm.get_scene_list(vm.get_base_timecode()) or [
|
|
255
|
+
(vm.get_base_timecode(), vm.get_current_timecode())
|
|
256
|
+
]
|
|
257
|
+
|
|
258
|
+
# 6) Track faces
|
|
259
|
+
tracks = []
|
|
260
|
+
for sc in scenes:
|
|
261
|
+
s, e = sc[0].frame_num, sc[1].frame_num
|
|
262
|
+
if e - s >= cfg.min_track:
|
|
263
|
+
tracks.extend(self._track([lst.copy() for lst in detections[s:e]]))
|
|
264
|
+
|
|
265
|
+
# 7) Crop tracks
|
|
266
|
+
crops = [
|
|
267
|
+
self._crop(t, frames, str(audio_wav), Path(work) / "cropped" / f"{i:05d}") for i, t in enumerate(tracks)
|
|
268
|
+
]
|
|
269
|
+
# AV offset: 5
|
|
270
|
+
# Min dist: 5.370
|
|
271
|
+
# Confidence: 9.892
|
|
272
|
+
|
|
273
|
+
# crops = [work / ".." / ".."/ "data" / "example.avi"]
|
|
274
|
+
# AV offset: 3
|
|
275
|
+
# Min dist: 5.348
|
|
276
|
+
# Confidence: 10.081
|
|
277
|
+
|
|
278
|
+
# crops = [work / "video.avi"]
|
|
279
|
+
# AV offset: 3
|
|
280
|
+
# Min dist: 6.668
|
|
281
|
+
# Confidence: 8.337
|
|
282
|
+
|
|
283
|
+
# 8) SyncNet evaluation
|
|
284
|
+
offsets, confs, dists = [], [], []
|
|
285
|
+
class Opt: ...
|
|
286
|
+
for i, cp in enumerate(crops):
|
|
287
|
+
crop_dir = work / "cropped" / f"crop_{i:05d}"
|
|
288
|
+
frames_dir = crop_dir
|
|
289
|
+
frames_dir.mkdir(parents=True, exist_ok=True)
|
|
290
|
+
audio_path = crop_dir / "audio.wav"
|
|
291
|
+
|
|
292
|
+
# Extract frames
|
|
293
|
+
(
|
|
294
|
+
ffmpeg.input(cp)
|
|
295
|
+
.output(str(frames_dir / "%06d.jpg"), f="image2", threads=1)
|
|
296
|
+
.overwrite_output()
|
|
297
|
+
.run()
|
|
298
|
+
)
|
|
299
|
+
|
|
300
|
+
# Extract audio
|
|
301
|
+
(
|
|
302
|
+
ffmpeg.input(cp)
|
|
303
|
+
.output(
|
|
304
|
+
str(audio_path),
|
|
305
|
+
ac=1,
|
|
306
|
+
vn=None,
|
|
307
|
+
acodec="pcm_s16le",
|
|
308
|
+
ar=16000,
|
|
309
|
+
af="aresample=async=1",
|
|
310
|
+
)
|
|
311
|
+
.overwrite_output()
|
|
312
|
+
.run()
|
|
313
|
+
)
|
|
314
|
+
|
|
315
|
+
opt = Opt()
|
|
316
|
+
opt.tmp_dir = str(crop_dir)
|
|
317
|
+
opt.batch_size = cfg.batch_size
|
|
318
|
+
opt.vshift = cfg.vshift
|
|
319
|
+
|
|
320
|
+
off, conf, dist = self.syncnet.evaluate(opt=opt)
|
|
321
|
+
offsets.append(off)
|
|
322
|
+
confs.append(conf)
|
|
323
|
+
dists.append(dist)
|
|
324
|
+
|
|
325
|
+
if not offsets:
|
|
326
|
+
return ([], [], [], 0.0, 0.0, "", False)
|
|
327
|
+
|
|
328
|
+
return offsets, confs, dists, max(confs), min(dists), s3fd_json, has_face
|
|
329
|
+
|
|
330
|
+
finally:
|
|
331
|
+
if not cache_dir:
|
|
332
|
+
shutil.rmtree(work, ignore_errors=True)
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: syncnet-python
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: SyncNet: Audio-visual synchronization detection using deep learning
|
|
5
|
+
Author: SyncNet Python Contributors
|
|
6
|
+
Maintainer: SyncNet Python Contributors
|
|
7
|
+
License: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/nawta/SyncNet_py313
|
|
9
|
+
Project-URL: Bug Reports, https://github.com/nawta/SyncNet_py313/issues
|
|
10
|
+
Project-URL: Source, https://github.com/nawta/SyncNet_py313
|
|
11
|
+
Keywords: audio-visual,synchronization,deep-learning,pytorch,lip-sync,video-processing,computer-vision
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Topic :: Multimedia :: Video
|
|
25
|
+
Classifier: Topic :: Scientific/Engineering :: Image Processing
|
|
26
|
+
Requires-Python: >=3.9
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
License-File: LICENSE
|
|
29
|
+
Requires-Dist: torch>=2.0.0
|
|
30
|
+
Requires-Dist: torchvision>=0.15.0
|
|
31
|
+
Requires-Dist: numpy>=1.24.0
|
|
32
|
+
Requires-Dist: scipy>=1.10.0
|
|
33
|
+
Requires-Dist: pandas>=2.0.0
|
|
34
|
+
Requires-Dist: scenedetect[opencv]>=0.6.0
|
|
35
|
+
Requires-Dist: opencv-contrib-python>=4.8.0
|
|
36
|
+
Requires-Dist: python-speech-features>=0.6
|
|
37
|
+
Requires-Dist: ffmpeg-python>=0.2.0
|
|
38
|
+
Provides-Extra: dev
|
|
39
|
+
Requires-Dist: pytest>=7.4.0; extra == "dev"
|
|
40
|
+
Requires-Dist: pytest-asyncio>=0.21.0; extra == "dev"
|
|
41
|
+
Requires-Dist: black>=23.0.0; extra == "dev"
|
|
42
|
+
Requires-Dist: ruff>=0.1.0; extra == "dev"
|
|
43
|
+
Requires-Dist: mypy>=1.7.0; extra == "dev"
|
|
44
|
+
Requires-Dist: types-opencv-python; extra == "dev"
|
|
45
|
+
Requires-Dist: types-scipy; extra == "dev"
|
|
46
|
+
Dynamic: license-file
|
|
47
|
+
|
|
48
|
+
# SyncNet Python
|
|
49
|
+
|
|
50
|
+
Audio-visual synchronization detection using deep learning.
|
|
51
|
+
|
|
52
|
+
## Overview
|
|
53
|
+
|
|
54
|
+
SyncNet Python is a PyTorch implementation of the SyncNet model, which detects audio-visual synchronization in videos. It can identify lip-sync errors by analyzing the correspondence between mouth movements and spoken audio.
|
|
55
|
+
|
|
56
|
+
## Features
|
|
57
|
+
|
|
58
|
+
- 🎥 **Audio-Visual Sync Detection**: Accurately detect synchronization between audio and video
|
|
59
|
+
- 🔍 **Face Detection**: Automatic face detection and tracking using S3FD
|
|
60
|
+
- 🚀 **Batch Processing**: Process multiple videos efficiently
|
|
61
|
+
- 🐍 **Python API**: Easy-to-use Python interface
|
|
62
|
+
- 📊 **Confidence Scores**: Get confidence metrics for sync quality
|
|
63
|
+
|
|
64
|
+
## Installation
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
pip install syncnet-python
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
### Additional Requirements
|
|
71
|
+
|
|
72
|
+
1. **FFmpeg**: Required for video processing
|
|
73
|
+
```bash
|
|
74
|
+
# Ubuntu/Debian
|
|
75
|
+
sudo apt-get install ffmpeg
|
|
76
|
+
|
|
77
|
+
# macOS
|
|
78
|
+
brew install ffmpeg
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
2. **Model Weights**: Download pre-trained weights
|
|
82
|
+
- Download `sfd_face.pth` and `syncnet_v2.model`
|
|
83
|
+
- Place them in a `weights/` directory
|
|
84
|
+
|
|
85
|
+
## Quick Start
|
|
86
|
+
|
|
87
|
+
```python
|
|
88
|
+
from syncnet_python import SyncNetPipeline
|
|
89
|
+
|
|
90
|
+
# Initialize pipeline
|
|
91
|
+
pipeline = SyncNetPipeline(
|
|
92
|
+
s3fd_weights="weights/sfd_face.pth",
|
|
93
|
+
syncnet_weights="weights/syncnet_v2.model",
|
|
94
|
+
device="cuda" # or "cpu"
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
# Process video
|
|
98
|
+
results = pipeline.inference(
|
|
99
|
+
video_path="video.mp4",
|
|
100
|
+
audio_path=None # Extract from video
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
# Get results
|
|
104
|
+
offset, confidence = results['offset'], results['confidence']
|
|
105
|
+
print(f"AV Offset: {offset} frames")
|
|
106
|
+
print(f"Confidence: {confidence:.3f}")
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Command Line Usage
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
# Process single video
|
|
113
|
+
syncnet-python video.mp4
|
|
114
|
+
|
|
115
|
+
# Process multiple videos
|
|
116
|
+
syncnet-python video1.mp4 video2.mp4 --output results.json
|
|
117
|
+
|
|
118
|
+
# Use CPU instead of GPU
|
|
119
|
+
syncnet-python video.mp4 --device cpu
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## Requirements
|
|
123
|
+
|
|
124
|
+
- Python 3.9+
|
|
125
|
+
- PyTorch 2.0+
|
|
126
|
+
- CUDA (optional but recommended)
|
|
127
|
+
- FFmpeg
|
|
128
|
+
|
|
129
|
+
## Citation
|
|
130
|
+
|
|
131
|
+
If you use this code in your research, please cite:
|
|
132
|
+
|
|
133
|
+
```bibtex
|
|
134
|
+
@inproceedings{chung2016out,
|
|
135
|
+
title={Out of time: automated lip sync in the wild},
|
|
136
|
+
author={Chung, Joon Son and Zisserman, Andrew},
|
|
137
|
+
booktitle={Asian Conference on Computer Vision},
|
|
138
|
+
year={2016}
|
|
139
|
+
}
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
## License
|
|
143
|
+
|
|
144
|
+
MIT License - see LICENSE file for details.
|
|
145
|
+
|
|
146
|
+
## Links
|
|
147
|
+
|
|
148
|
+
- GitHub: https://github.com/yourusername/syncnet-python
|
|
149
|
+
- Documentation: https://syncnet-python.readthedocs.io
|
|
150
|
+
- Issues: https://github.com/yourusername/syncnet-python/issues
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
syncnet_python/SyncNetInstance.py,sha256=V-RtWL4nN2CW5n56DkE0d6hcnet4hAs_OVMaMS1ih20,6227
|
|
2
|
+
syncnet_python/SyncNetModel.py,sha256=6qk27paoyV39MTsVyu6K2sbbM-1SLU_2N1zbqm1eeYs,3575
|
|
3
|
+
syncnet_python/__init__.py,sha256=RUVJaCRWFMqFPDA8DOihkkCxJj_yoCBanVsHnOmgrS4,702
|
|
4
|
+
syncnet_python/cli.py,sha256=YSQVVDChLkQ96gFRFqD7laQXqbrLC2b8SKaF-vc8FuA,3256
|
|
5
|
+
syncnet_python/run_syncnet_pipeline_on_1example.py,sha256=7I8_jEgIFw-RUqCHU29LrIWIRP77lTUiH6MePOxsuWQ,1008
|
|
6
|
+
syncnet_python/run_syncnet_pipeline_on_mocha_generation_on_mocha_bench.py,sha256=HJHGLgKpf21bG27_x1QD3MAhqtHey26Ujxb8s2aK-tA,5928
|
|
7
|
+
syncnet_python/run_syncnet_pipeline_on_your_own_model_results.py,sha256=m-kz_DHhka_DywpBRWdxk-S0hrwqvCR16RiVQ7HUqDk,6003
|
|
8
|
+
syncnet_python/syncnet_pipeline.py,sha256=9mSuNHnpgwxCNRUGmgat8peEtDNI6MFc_0VAykaSubM,11679
|
|
9
|
+
syncnet_python/detectors/__init__.py,sha256=WLone-DTbvQUFleY93pmq7xomPDVjda9ATpZYcS6_sA,23
|
|
10
|
+
syncnet_python/detectors/s3fd/__init__.py,sha256=MIJfIEsKFTGh8brmD3fBQg8JZAp57NtaTkY6_I5QZP0,2322
|
|
11
|
+
syncnet_python/detectors/s3fd/box_utils.py,sha256=CWn46LMJKO1cdenI_IWxBbDriO2ueWTCl14Xp28fr-U,7235
|
|
12
|
+
syncnet_python/detectors/s3fd/nets.py,sha256=BYPJq9UJ5bq5Q5RgP1j-NsFzO1gP3N0PYE1X19-4PZw,5891
|
|
13
|
+
syncnet_python-0.1.0.dist-info/licenses/LICENSE,sha256=qwJOQjZqnGgzcgwg3VUkT2pGKwLyZr5nUOm0Gv7Zfog,1088
|
|
14
|
+
syncnet_python-0.1.0.dist-info/METADATA,sha256=-ij02QUR1EUpy61vt7LoTnojQ06CYAeuYwIBUuspf3c,4445
|
|
15
|
+
syncnet_python-0.1.0.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
16
|
+
syncnet_python-0.1.0.dist-info/entry_points.txt,sha256=zPJrIl_ekAa31dHZNCoTHiII2k_m5GyMD5BEkMR6Ytk,59
|
|
17
|
+
syncnet_python-0.1.0.dist-info/top_level.txt,sha256=OkvKpxwq9NzQSjjdD571umGrI3meW9kVleOV2N5Ua04,15
|
|
18
|
+
syncnet_python-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 SyncNet Python 3.13 Contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
syncnet_python
|