cutan 0.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cutan/__init__.py +80 -0
- cutan/audio/__init__.py +50 -0
- cutan/audio/injectable_lipsync.py +123 -0
- cutan/audio/offline_lipsync.py +126 -0
- cutan/audio/rhubarb_lipsync.py +152 -0
- cutan/audio/whisper_lipsync.py +143 -0
- cutan/bench.py +244 -0
- cutan/characters/__init__.py +125 -0
- cutan/characters/brows.py +504 -0
- cutan/characters/checks.py +610 -0
- cutan/characters/cli.py +629 -0
- cutan/characters/colour_roles.py +176 -0
- cutan/characters/dicebear.py +238 -0
- cutan/characters/drawn.py +94 -0
- cutan/characters/factory.py +2237 -0
- cutan/characters/idle.py +220 -0
- cutan/characters/licenses.py +385 -0
- cutan/characters/methods.py +708 -0
- cutan/characters/mouth_set.py +235 -0
- cutan/characters/play.py +1244 -0
- cutan/characters/promote.py +189 -0
- cutan/characters/record.py +167 -0
- cutan/characters/registration.py +239 -0
- cutan/characters/schema.py +921 -0
- cutan/characters/silhouette.py +155 -0
- cutan/characters/svg_utils.py +346 -0
- cutan/characters/validate.py +946 -0
- cutan/characters/vocabulary.py +194 -0
- cutan/compile/__init__.py +1 -0
- cutan/compile/coarticulate.py +445 -0
- cutan/compile/gaze.py +147 -0
- cutan/compile/lowering.py +109 -0
- cutan/compile/passes.py +2581 -0
- cutan/conftest.py +8 -0
- cutan/data/dicebear_9x_styles.json +38 -0
- cutan/expression/__init__.py +80 -0
- cutan/expression/axes.py +122 -0
- cutan/expression/binding.py +326 -0
- cutan/expression/blendshapes.py +125 -0
- cutan/expression/presets.py +203 -0
- cutan/expression/provider.py +210 -0
- cutan/expression/registration.py +187 -0
- cutan/genre.py +220 -0
- cutan/impacts/__init__.py +89 -0
- cutan/impacts/cli.py +166 -0
- cutan/impacts/clip.py +603 -0
- cutan/impacts/objects.py +255 -0
- cutan/impacts/performance.py +257 -0
- cutan/impacts/stroke.py +252 -0
- cutan/impacts/truth.py +368 -0
- cutan/library.py +318 -0
- cutan/nw.py +58 -0
- cutan/runtime/__init__.py +1 -0
- cutan/runtime/visuals.js +129 -0
- cutan/verify/__init__.py +1 -0
- cutan/verify/style.py +1194 -0
- cutan-0.0.2.dist-info/METADATA +64 -0
- cutan-0.0.2.dist-info/RECORD +61 -0
- cutan-0.0.2.dist-info/WHEEL +4 -0
- cutan-0.0.2.dist-info/entry_points.txt +2 -0
- cutan-0.0.2.dist-info/licenses/LICENSE +21 -0
cutan/__init__.py
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""cutan: cut-out animation, as a genre package on the ``an`` core.
|
|
2
|
+
|
|
3
|
+
``an`` is the core of structured animation (scene documents, the timing kernel,
|
|
4
|
+
renderers, audio, storage, verification). A *genre* adds what one kind of
|
|
5
|
+
animation needs and registers it with the core; ``cutan`` is the first genre to
|
|
6
|
+
leave ``an``: rigged characters, faces and expressions, lip-sync visemes, swap
|
|
7
|
+
sets and views, cut-out styles and impacts (ADR 0001 in ``an``'s
|
|
8
|
+
``misc/docs/adr/``, tracked by an#225 under the epic an#231).
|
|
9
|
+
|
|
10
|
+
Rigged characters (``cutan.characters``), faces (``cutan.expression``), impacts
|
|
11
|
+
(``cutan.impacts``), the cut-out compile passes (``cutan.compile``), lip-sync
|
|
12
|
+
providers (``cutan.audio``), the style lint (``cutan.verify``) and the genre object
|
|
13
|
+
(``cutan.genre``) lived inside ``an`` until the P8 move (an#225); ``an`` keeps
|
|
14
|
+
warning aliases at the old import paths. Install it with ``pip install "an[cutout]"``;
|
|
15
|
+
``an`` finds it through the ``an.genres`` entry point.
|
|
16
|
+
|
|
17
|
+
The identifiers below are the genre's **persisted** names (ADR 0001 decision 9):
|
|
18
|
+
they are written into documents, stores and entry-point metadata, and none of
|
|
19
|
+
them changes when code moves between distributions.
|
|
20
|
+
|
|
21
|
+
>>> GENRE_NAME
|
|
22
|
+
'cutout_animation'
|
|
23
|
+
>>> ENTRY_POINT_GROUP, ENTRY_POINT_NAME
|
|
24
|
+
('an.genres', 'cutout_animation')
|
|
25
|
+
>>> RENDERER_NAME, LIBRARY_NAME
|
|
26
|
+
('cutout', 'cutan')
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
#: The genre's slug: the ``an.genres`` entry-point name and the ``nw`` genre id.
|
|
30
|
+
GENRE_NAME: str = "cutout_animation"
|
|
31
|
+
|
|
32
|
+
#: The entry-point group ``an.genres.load()`` reads.
|
|
33
|
+
ENTRY_POINT_GROUP: str = "an.genres"
|
|
34
|
+
|
|
35
|
+
#: The entry-point name this distribution declares (``pyproject.toml``): the name
|
|
36
|
+
#: ``an`` used to declare for the in-distribution genre, so the handover was by name.
|
|
37
|
+
ENTRY_POINT_NAME: str = GENRE_NAME
|
|
38
|
+
|
|
39
|
+
#: Where the entry point points.
|
|
40
|
+
ENTRY_POINT_VALUE: str = "cutan.genre:CUTOUT"
|
|
41
|
+
|
|
42
|
+
#: The persisted renderer name of cut-out shots (``an.stage`` claims it).
|
|
43
|
+
RENDERER_NAME: str = "cutout"
|
|
44
|
+
|
|
45
|
+
#: The package whose data root holds the genre's asset library and projects
|
|
46
|
+
#: (``~/.local/share/cutan`` by default; ``CUTAN_HOME`` overrides it).
|
|
47
|
+
LIBRARY_NAME: str = "cutan"
|
|
48
|
+
|
|
49
|
+
#: The lowest ``an.genres.API_LEVEL`` this ``cutan`` runs against ("the lowest ``an``
|
|
50
|
+
#: it supports", ADR 0001 decision 8, said without a version pin: ``an``'s version is
|
|
51
|
+
#: assigned by CI at merge). Level 2 is the move itself: ``Genre.services``,
|
|
52
|
+
#: ``ActionKind.lowering``, ``EntityKind.swap_declaration`` and ``an.stage.rig``.
|
|
53
|
+
REQUIRED_AN_API_LEVEL: int = 2
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def require_an() -> None:
|
|
57
|
+
"""Refuse, with an upgrade hint, to load against an ``an`` older than this ``cutan`` needs.
|
|
58
|
+
|
|
59
|
+
>>> require_an()
|
|
60
|
+
"""
|
|
61
|
+
import an.genres
|
|
62
|
+
|
|
63
|
+
level = getattr(an.genres, "API_LEVEL", 0)
|
|
64
|
+
if level < REQUIRED_AN_API_LEVEL:
|
|
65
|
+
raise ImportError(
|
|
66
|
+
f"cutan needs an.genres API level {REQUIRED_AN_API_LEVEL} or higher; this an "
|
|
67
|
+
f"provides {level}. Upgrade an: pip install -U an"
|
|
68
|
+
)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
__all__ = [
|
|
72
|
+
"ENTRY_POINT_GROUP",
|
|
73
|
+
"ENTRY_POINT_NAME",
|
|
74
|
+
"ENTRY_POINT_VALUE",
|
|
75
|
+
"GENRE_NAME",
|
|
76
|
+
"LIBRARY_NAME",
|
|
77
|
+
"REQUIRED_AN_API_LEVEL",
|
|
78
|
+
"RENDERER_NAME",
|
|
79
|
+
"require_an",
|
|
80
|
+
]
|
cutan/audio/__init__.py
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""The cut-out genre's lip-sync providers: letters, Rhubarb and word timings to mouth shapes.
|
|
2
|
+
|
|
3
|
+
Moved from ``an.audio`` (an#225). ``an.audio`` keeps the protocols
|
|
4
|
+
(:class:`~an.audio.lipsync.LipSyncProvider`, :class:`~an.audio.lipsync.VisemeTrack`),
|
|
5
|
+
text-to-speech and the pipeline; a viseme only means something to a genre that
|
|
6
|
+
draws mouths. The genre registers these by name as ``lipsync.<name>`` services
|
|
7
|
+
(:mod:`cutan.genre`), which is how ``render(lipsync="offline")`` and
|
|
8
|
+
``an render --lipsync rhubarb`` keep working.
|
|
9
|
+
|
|
10
|
+
>>> offline_factory().name
|
|
11
|
+
'offline'
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from cutan.audio.injectable_lipsync import StaticWordTimings, WordTimingsLipSync
|
|
17
|
+
from cutan.audio.offline_lipsync import OfflineLipSync
|
|
18
|
+
from cutan.audio.rhubarb_lipsync import RhubarbLipSync
|
|
19
|
+
from cutan.audio.whisper_lipsync import WhisperLipSync
|
|
20
|
+
|
|
21
|
+
#: The language a provider aligns for when the caller says nothing (Rhubarb's
|
|
22
|
+
#: recognizer follows it, an#96): the same default ``an.audio.providers`` has.
|
|
23
|
+
DFLT_LANGUAGE: str = "en"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def offline_factory(**_: object) -> OfflineLipSync:
|
|
27
|
+
"""The deterministic char-to-viseme provider (the default)."""
|
|
28
|
+
return OfflineLipSync()
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def rhubarb_factory(*, language: str = DFLT_LANGUAGE, **_: object) -> RhubarbLipSync:
|
|
32
|
+
"""Rhubarb Lip Sync (needs the ``rhubarb`` binary); ``language`` picks its recognizer."""
|
|
33
|
+
return RhubarbLipSync(language=language)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def whisper_factory(**_: object) -> WhisperLipSync:
|
|
37
|
+
"""Word timings from Whisper, distributed over the mouth shapes."""
|
|
38
|
+
return WhisperLipSync()
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
__all__ = [
|
|
42
|
+
"OfflineLipSync",
|
|
43
|
+
"RhubarbLipSync",
|
|
44
|
+
"StaticWordTimings",
|
|
45
|
+
"WhisperLipSync",
|
|
46
|
+
"WordTimingsLipSync",
|
|
47
|
+
"offline_factory",
|
|
48
|
+
"rhubarb_factory",
|
|
49
|
+
"whisper_factory",
|
|
50
|
+
]
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""Lip-sync provider that consumes pre-computed word timings.
|
|
2
|
+
|
|
3
|
+
Useful when an upstream system already has authoritative word-level
|
|
4
|
+
timings and would otherwise force ``an`` to re-transcribe the same
|
|
5
|
+
audio. The canonical case is ``muvid``, where the lyric → audio
|
|
6
|
+
alignment store (``lacing``) is the SSOT and re-running
|
|
7
|
+
``WhisperLipSync`` on the audio produces a redundant (and possibly
|
|
8
|
+
divergent) word-timestamp set.
|
|
9
|
+
|
|
10
|
+
Two pieces:
|
|
11
|
+
|
|
12
|
+
- :class:`StaticWordTimings` — a :class:`WordTimingProvider` over a
|
|
13
|
+
fixed list of ``(word, start, end)`` tuples.
|
|
14
|
+
- :class:`WordTimingsLipSync` — a :class:`LipSyncProvider` that reads
|
|
15
|
+
from any :class:`WordTimingProvider` and runs the same
|
|
16
|
+
word→viseme conversion as :class:`WhisperLipSync` (via
|
|
17
|
+
:func:`word_timings_to_visemes`), so output is shape-compatible with
|
|
18
|
+
the rest of the cutout pipeline.
|
|
19
|
+
|
|
20
|
+
Drop-in usage::
|
|
21
|
+
|
|
22
|
+
from cutan.audio.injectable_lipsync import (
|
|
23
|
+
StaticWordTimings, WordTimingsLipSync,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
timings = [("hello", 0.5, 1.0), ("world", 1.2, 1.8)]
|
|
27
|
+
lipsync = WordTimingsLipSync(StaticWordTimings(timings))
|
|
28
|
+
track = lipsync.align(audio_clip, "hello world")
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
from typing import Sequence
|
|
34
|
+
|
|
35
|
+
from an.audio.lipsync import (
|
|
36
|
+
LipSyncProvider,
|
|
37
|
+
VisemeTrack,
|
|
38
|
+
WordTiming,
|
|
39
|
+
WordTimingProvider,
|
|
40
|
+
word_timings_to_visemes,
|
|
41
|
+
)
|
|
42
|
+
from cutan.audio.offline_lipsync import _CHAR_TO_VISEME, _REST_VISEME
|
|
43
|
+
from an.audio.tts import AudioClip
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
class StaticWordTimings:
|
|
47
|
+
"""A :class:`WordTimingProvider` over a fixed list of timings."""
|
|
48
|
+
|
|
49
|
+
name: str = "static"
|
|
50
|
+
|
|
51
|
+
def __init__(self, words: Sequence[WordTiming], *, label: str = "static") -> None:
|
|
52
|
+
self._words = tuple(words)
|
|
53
|
+
self.name = label
|
|
54
|
+
|
|
55
|
+
def words_for(
|
|
56
|
+
self, audio: AudioClip, *, transcript: str = ""
|
|
57
|
+
) -> Sequence[WordTiming]:
|
|
58
|
+
return self._words
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class WordTimingsLipSync:
|
|
62
|
+
""":class:`LipSyncProvider` driven by a :class:`WordTimingProvider`.
|
|
63
|
+
|
|
64
|
+
Skips transcription entirely. Use this when the caller already has
|
|
65
|
+
authoritative word timings (e.g. from a separate lyric-alignment
|
|
66
|
+
pipeline).
|
|
67
|
+
|
|
68
|
+
Args:
|
|
69
|
+
provider: any :class:`WordTimingProvider`.
|
|
70
|
+
char_to_viseme: optional override of the character→viseme code
|
|
71
|
+
mapping; defaults to the one shared with
|
|
72
|
+
:class:`OfflineLipSync` / :class:`WhisperLipSync`.
|
|
73
|
+
convention: declared viseme convention string for the produced
|
|
74
|
+
track. Defaults to ``"rhubarb"`` for compatibility with the
|
|
75
|
+
existing cutout adapter.
|
|
76
|
+
rest_viseme: code emitted in silent gaps. Defaults to
|
|
77
|
+
:data:`_REST_VISEME`.
|
|
78
|
+
min_gap_for_rest: minimum inter-word silence (seconds) before
|
|
79
|
+
we insert a rest keyframe. Defaults to ``0.20``.
|
|
80
|
+
"""
|
|
81
|
+
|
|
82
|
+
convention: str = "rhubarb"
|
|
83
|
+
|
|
84
|
+
def __init__(
|
|
85
|
+
self,
|
|
86
|
+
provider: WordTimingProvider,
|
|
87
|
+
*,
|
|
88
|
+
char_to_viseme: dict[str, str] | None = None,
|
|
89
|
+
convention: str = "rhubarb",
|
|
90
|
+
rest_viseme: str = _REST_VISEME,
|
|
91
|
+
min_gap_for_rest: float = 0.20,
|
|
92
|
+
) -> None:
|
|
93
|
+
self.provider = provider
|
|
94
|
+
self.convention = convention
|
|
95
|
+
self._mapping = (
|
|
96
|
+
dict(char_to_viseme) if char_to_viseme else dict(_CHAR_TO_VISEME)
|
|
97
|
+
)
|
|
98
|
+
self._rest = rest_viseme
|
|
99
|
+
self._min_gap = min_gap_for_rest
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def name(self) -> str:
|
|
103
|
+
# Disambiguate cache keys per-provider so swapping providers
|
|
104
|
+
# invalidates the viseme cache.
|
|
105
|
+
return f"word-timings:{getattr(self.provider, 'name', 'unknown')}"
|
|
106
|
+
|
|
107
|
+
#: Built from words, so the track carries them (an#96).
|
|
108
|
+
emits_word_timings: bool = True
|
|
109
|
+
|
|
110
|
+
def align(self, audio: AudioClip, transcript: str) -> VisemeTrack:
|
|
111
|
+
words = list(self.provider.words_for(audio, transcript=transcript))
|
|
112
|
+
return VisemeTrack(
|
|
113
|
+
visemes=word_timings_to_visemes(
|
|
114
|
+
words,
|
|
115
|
+
total_duration=audio.duration,
|
|
116
|
+
char_to_viseme=self._mapping,
|
|
117
|
+
rest_viseme=self._rest,
|
|
118
|
+
min_gap_for_rest=self._min_gap,
|
|
119
|
+
),
|
|
120
|
+
convention=self.convention,
|
|
121
|
+
duration=audio.duration,
|
|
122
|
+
words=words,
|
|
123
|
+
)
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""OfflineLipSync — deterministic transcript → viseme track. No network, no binary.
|
|
2
|
+
|
|
3
|
+
The default lip-sync provider for `an`. Maps each meaningful character of the
|
|
4
|
+
transcript to a Rhubarb-convention viseme letter (A–H + X), then distributes
|
|
5
|
+
keyframes evenly across the audio's duration. Repeated visemes get collapsed
|
|
6
|
+
into a single keyframe so the mouth doesn't "stutter" on long vowel runs.
|
|
7
|
+
|
|
8
|
+
Crude but visible. Use ``RhubarbLipSync`` for real phoneme alignment once
|
|
9
|
+
you've installed the Rhubarb binary.
|
|
10
|
+
|
|
11
|
+
>>> from an.audio.tts import AudioClip
|
|
12
|
+
>>> ls = OfflineLipSync()
|
|
13
|
+
>>> track = ls.align(AudioClip(duration=1.0, transcript="hello"), "hello")
|
|
14
|
+
>>> track.convention
|
|
15
|
+
'rhubarb'
|
|
16
|
+
>>> track.duration
|
|
17
|
+
1.0
|
|
18
|
+
>>> len(track.visemes) >= 2
|
|
19
|
+
True
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from typing import Iterable
|
|
25
|
+
|
|
26
|
+
from an.audio.lipsync import Viseme, VisemeTrack
|
|
27
|
+
from an.audio.tts import AudioClip
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# Map a single ASCII character (lowercased) to a Rhubarb viseme letter.
|
|
31
|
+
# A=closed (P/B/M, rest), B=tight open (D/S/T/Z/N/L), C=eh, D=ah/wide,
|
|
32
|
+
# E=ohh, F=ooh, G=teeth-on-lip (F/V), H=th, X=idle.
|
|
33
|
+
_CHAR_TO_VISEME: dict[str, str] = {
|
|
34
|
+
# closed lips
|
|
35
|
+
"p": "A",
|
|
36
|
+
"b": "A",
|
|
37
|
+
"m": "A",
|
|
38
|
+
# teeth-on-lip
|
|
39
|
+
"f": "G",
|
|
40
|
+
"v": "G",
|
|
41
|
+
# th-ish
|
|
42
|
+
"h": "C",
|
|
43
|
+
# vowel families
|
|
44
|
+
"a": "D",
|
|
45
|
+
"e": "C",
|
|
46
|
+
"i": "B",
|
|
47
|
+
"o": "E",
|
|
48
|
+
"u": "F",
|
|
49
|
+
"y": "B",
|
|
50
|
+
# alveolar consonants
|
|
51
|
+
"d": "B",
|
|
52
|
+
"t": "B",
|
|
53
|
+
"s": "B",
|
|
54
|
+
"z": "B",
|
|
55
|
+
"n": "B",
|
|
56
|
+
"l": "B",
|
|
57
|
+
"r": "B",
|
|
58
|
+
# remaining consonants — neutral semi-open
|
|
59
|
+
"c": "C",
|
|
60
|
+
"g": "C",
|
|
61
|
+
"j": "C",
|
|
62
|
+
"k": "C",
|
|
63
|
+
"q": "C",
|
|
64
|
+
"w": "F",
|
|
65
|
+
"x": "C",
|
|
66
|
+
}
|
|
67
|
+
_REST_VISEME: str = "X"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class OfflineLipSync:
|
|
71
|
+
"""Default lip-sync provider: deterministic char-to-viseme mapping.
|
|
72
|
+
|
|
73
|
+
Implements the ``LipSyncProvider`` protocol.
|
|
74
|
+
"""
|
|
75
|
+
|
|
76
|
+
name: str = "offline"
|
|
77
|
+
convention: str = "rhubarb"
|
|
78
|
+
|
|
79
|
+
def __init__(self, *, char_to_viseme: dict[str, str] | None = None) -> None:
|
|
80
|
+
self._mapping = (
|
|
81
|
+
dict(char_to_viseme) if char_to_viseme else dict(_CHAR_TO_VISEME)
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
def align(self, audio: AudioClip, transcript: str) -> VisemeTrack:
|
|
85
|
+
codes = list(self._codes_for(transcript))
|
|
86
|
+
if not codes:
|
|
87
|
+
# Empty transcript → just rest at start, rest at end.
|
|
88
|
+
return VisemeTrack(
|
|
89
|
+
visemes=[
|
|
90
|
+
Viseme(time=0.0, code=_REST_VISEME),
|
|
91
|
+
Viseme(time=max(0.0, audio.duration), code=_REST_VISEME),
|
|
92
|
+
],
|
|
93
|
+
convention=self.convention,
|
|
94
|
+
duration=audio.duration,
|
|
95
|
+
)
|
|
96
|
+
# Distribute one keyframe per code over [0, duration].
|
|
97
|
+
n = len(codes)
|
|
98
|
+
duration = max(audio.duration, 1e-3)
|
|
99
|
+
visemes: list[Viseme] = []
|
|
100
|
+
# Always lead with rest at t=0
|
|
101
|
+
visemes.append(Viseme(time=0.0, code=_REST_VISEME))
|
|
102
|
+
for i, code in enumerate(codes):
|
|
103
|
+
t = (i + 1) / (n + 1) * duration
|
|
104
|
+
visemes.append(Viseme(time=t, code=code))
|
|
105
|
+
# Trailing rest
|
|
106
|
+
visemes.append(Viseme(time=duration, code=_REST_VISEME))
|
|
107
|
+
return VisemeTrack(
|
|
108
|
+
visemes=visemes, convention=self.convention, duration=audio.duration
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
def _codes_for(self, transcript: str) -> Iterable[str]:
|
|
112
|
+
"""Per-character viseme codes; collapses adjacent duplicates."""
|
|
113
|
+
previous: str | None = None
|
|
114
|
+
for raw in transcript:
|
|
115
|
+
ch = raw.lower()
|
|
116
|
+
if not ch.isalpha():
|
|
117
|
+
# Spaces, punctuation → mouth rest. Insert if not redundant.
|
|
118
|
+
if previous != _REST_VISEME:
|
|
119
|
+
previous = _REST_VISEME
|
|
120
|
+
yield _REST_VISEME
|
|
121
|
+
continue
|
|
122
|
+
code = self._mapping.get(ch, "C")
|
|
123
|
+
if code == previous:
|
|
124
|
+
continue
|
|
125
|
+
previous = code
|
|
126
|
+
yield code
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
"""RhubarbLipSync — calls the rhubarb-lip-sync binary for phoneme-aligned visemes.
|
|
2
|
+
|
|
3
|
+
Requires the ``rhubarb`` binary on PATH. macOS: ``brew install rhubarb-lipsync``.
|
|
4
|
+
Linux/Windows: download from the project's GitHub releases.
|
|
5
|
+
|
|
6
|
+
Falls back gracefully (raises a clear error) if the binary is missing — the
|
|
7
|
+
default ``OfflineLipSync`` keeps the pipeline functional in the meantime.
|
|
8
|
+
|
|
9
|
+
**The recognizer follows the language** (an#96, epic #9 defect 5a). Rhubarb has
|
|
10
|
+
two: ``pocketSphinx`` — its default, "use for English recordings", the only one
|
|
11
|
+
that reads ``--dialogFile`` (it builds a dialog language model and mixes it 90/10
|
|
12
|
+
with the default) — and ``phonetic``, "use for non-English recordings", which
|
|
13
|
+
``UNUSED(dialog)``s the transcript at source. This module used to pass
|
|
14
|
+
``-r phonetic`` **and** ``--dialogFile`` unconditionally: English speech from an
|
|
15
|
+
English transcript ran the language-independent recognizer and the transcript
|
|
16
|
+
it wrote to disk was never read. Now ``recognizer=None`` (the default) resolves
|
|
17
|
+
per ``language`` — ``"en"`` → ``pocketSphinx`` with the dialog file, anything
|
|
18
|
+
else → ``phonetic`` and **no transcript is written** (a file nothing reads is a
|
|
19
|
+
lie waiting for the next reader). An explicit ``recognizer`` still overrides.
|
|
20
|
+
``name`` carries the recognizer so the viseme cache key changes with it and no
|
|
21
|
+
stale ``phonetic`` track replays.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import json
|
|
27
|
+
import shutil
|
|
28
|
+
import subprocess
|
|
29
|
+
import tempfile
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
|
|
32
|
+
from an.audio.lipsync import Viseme, VisemeTrack
|
|
33
|
+
from an.audio.tts import AudioClip
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
_DEFAULT_TIMEOUT_S: float = 60.0
|
|
37
|
+
#: Rhubarb's own default and its English recognizer — the one that reads the
|
|
38
|
+
#: dialog file.
|
|
39
|
+
_ENGLISH_RECOGNIZER: str = "pocketSphinx"
|
|
40
|
+
#: Language-independent; ignores the dialog file at source.
|
|
41
|
+
_PHONETIC_RECOGNIZER: str = "phonetic"
|
|
42
|
+
#: The languages `pocketSphinx` (CMU Sphinx US English acoustic model) covers.
|
|
43
|
+
ENGLISH_LANGUAGES: frozenset[str] = frozenset({"en"})
|
|
44
|
+
RECOGNIZERS: frozenset[str] = frozenset({_ENGLISH_RECOGNIZER, _PHONETIC_RECOGNIZER})
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def recognizer_for(language: str) -> str:
|
|
48
|
+
"""The Rhubarb recognizer for a BCP-47 language tag (primary subtag only).
|
|
49
|
+
|
|
50
|
+
Accepts the POSIX locale spelling too (``en_US``); an empty tag is refused
|
|
51
|
+
rather than read as "non-English" (an#96 review).
|
|
52
|
+
|
|
53
|
+
>>> recognizer_for("en"), recognizer_for("en-GB"), recognizer_for("en_US"), recognizer_for("fr")
|
|
54
|
+
('pocketSphinx', 'pocketSphinx', 'pocketSphinx', 'phonetic')
|
|
55
|
+
"""
|
|
56
|
+
primary = language.replace("_", "-").split("-", 1)[0].strip().lower()
|
|
57
|
+
if not primary:
|
|
58
|
+
raise ValueError("language must be a BCP-47 tag such as 'en' or 'fr'; got ''")
|
|
59
|
+
return _ENGLISH_RECOGNIZER if primary in ENGLISH_LANGUAGES else _PHONETIC_RECOGNIZER
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class RhubarbLipSync:
|
|
63
|
+
"""Wrap the rhubarb CLI. Implements the ``LipSyncProvider`` protocol.
|
|
64
|
+
|
|
65
|
+
>>> RhubarbLipSync(binary_path="/bin/rhubarb").recognizer
|
|
66
|
+
'pocketSphinx'
|
|
67
|
+
>>> RhubarbLipSync(binary_path="/bin/rhubarb", language="de").recognizer
|
|
68
|
+
'phonetic'
|
|
69
|
+
>>> RhubarbLipSync(binary_path="/bin/rhubarb", language="de").name
|
|
70
|
+
'rhubarb:phonetic'
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
convention: str = "rhubarb"
|
|
74
|
+
|
|
75
|
+
def __init__(
|
|
76
|
+
self,
|
|
77
|
+
*,
|
|
78
|
+
binary_path: str | None = None,
|
|
79
|
+
language: str = "en",
|
|
80
|
+
recognizer: str | None = None,
|
|
81
|
+
timeout_s: float = _DEFAULT_TIMEOUT_S,
|
|
82
|
+
) -> None:
|
|
83
|
+
self.binary_path = binary_path or shutil.which("rhubarb")
|
|
84
|
+
self.language = language
|
|
85
|
+
chosen = recognizer if recognizer is not None else recognizer_for(language)
|
|
86
|
+
if chosen not in RECOGNIZERS:
|
|
87
|
+
raise ValueError(
|
|
88
|
+
f"unknown rhubarb recognizer {chosen!r}; known: {sorted(RECOGNIZERS)}"
|
|
89
|
+
)
|
|
90
|
+
self.recognizer = chosen
|
|
91
|
+
self.timeout_s = timeout_s
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def name(self) -> str:
|
|
95
|
+
# The recognizer is part of the identity: it changes the track, so it
|
|
96
|
+
# must change the viseme cache key (the pipeline hashes `name`).
|
|
97
|
+
return f"rhubarb:{self.recognizer}"
|
|
98
|
+
|
|
99
|
+
@property
|
|
100
|
+
def uses_dialog_file(self) -> bool:
|
|
101
|
+
"""Whether the chosen recognizer reads a transcript at all."""
|
|
102
|
+
return self.recognizer == _ENGLISH_RECOGNIZER
|
|
103
|
+
|
|
104
|
+
def align(self, audio: AudioClip, transcript: str) -> VisemeTrack:
|
|
105
|
+
if not self.binary_path:
|
|
106
|
+
raise RuntimeError(
|
|
107
|
+
"rhubarb binary not found on PATH. Install with: "
|
|
108
|
+
"brew install rhubarb-lipsync (macOS) or grab a release from "
|
|
109
|
+
"https://github.com/DanielSWolf/rhubarb-lip-sync/releases."
|
|
110
|
+
)
|
|
111
|
+
with tempfile.TemporaryDirectory() as d:
|
|
112
|
+
d = Path(d)
|
|
113
|
+
audio_path = audio.path
|
|
114
|
+
if audio_path is None:
|
|
115
|
+
if audio.bytes_ is None:
|
|
116
|
+
raise ValueError("AudioClip needs either .path or .bytes_")
|
|
117
|
+
audio_path = d / "audio.wav"
|
|
118
|
+
audio_path.write_bytes(audio.bytes_)
|
|
119
|
+
|
|
120
|
+
out_json = d / "out.json"
|
|
121
|
+
cmd = [self.binary_path, "-f", "json", "-r", self.recognizer]
|
|
122
|
+
if self.uses_dialog_file:
|
|
123
|
+
dialog_path = d / "transcript.txt"
|
|
124
|
+
dialog_path.write_text(transcript, encoding="utf-8")
|
|
125
|
+
cmd += ["--dialogFile", str(dialog_path)]
|
|
126
|
+
cmd += ["-o", str(out_json), str(audio_path)]
|
|
127
|
+
try:
|
|
128
|
+
subprocess.run(
|
|
129
|
+
cmd,
|
|
130
|
+
capture_output=True,
|
|
131
|
+
text=True,
|
|
132
|
+
timeout=self.timeout_s,
|
|
133
|
+
check=True,
|
|
134
|
+
)
|
|
135
|
+
except subprocess.CalledProcessError as e:
|
|
136
|
+
raise RuntimeError(
|
|
137
|
+
f"rhubarb failed (rc={e.returncode}): {e.stderr}"
|
|
138
|
+
) from e
|
|
139
|
+
data = json.loads(out_json.read_text(encoding="utf-8"))
|
|
140
|
+
|
|
141
|
+
cues = data.get("mouthCues", [])
|
|
142
|
+
visemes = [Viseme(time=float(c["start"]), code=str(c["value"])) for c in cues]
|
|
143
|
+
if cues:
|
|
144
|
+
# Append a final rest at the last cue's "end" so the track matches duration.
|
|
145
|
+
last_end = float(cues[-1]["end"])
|
|
146
|
+
if not visemes or visemes[-1].code != "X":
|
|
147
|
+
visemes.append(Viseme(time=last_end, code="X"))
|
|
148
|
+
return VisemeTrack(
|
|
149
|
+
visemes=visemes,
|
|
150
|
+
convention=self.convention,
|
|
151
|
+
duration=audio.duration,
|
|
152
|
+
)
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
"""WhisperLipSync — faster-whisper word timestamps → viseme keyframes.
|
|
2
|
+
|
|
3
|
+
Phase 9. Bridges the gap between deterministic ``OfflineLipSync`` (twitchy,
|
|
4
|
+
char-distributed) and the system-binary-dependent ``RhubarbLipSync``. Uses
|
|
5
|
+
``faster-whisper`` to transcribe the rendered audio into word-level
|
|
6
|
+
timestamps, then distributes visemes within each word's [start, end] span
|
|
7
|
+
based on the word's letter→viseme mapping (collapsed-duplicates).
|
|
8
|
+
|
|
9
|
+
This gives ~75% of Rhubarb-quality lip-sync without any system binaries,
|
|
10
|
+
just a ~75 MB model download (cached after first use).
|
|
11
|
+
|
|
12
|
+
Trade-offs vs. OfflineLipSync:
|
|
13
|
+
|
|
14
|
+
- **Better**: timing is locked to actual word boundaries. Mouth holds shape
|
|
15
|
+
through silent gaps between words instead of cycling through visemes.
|
|
16
|
+
- **Same**: viseme codes per phoneme are still our simple letter mapping;
|
|
17
|
+
no IPA/ARPAbet awareness yet (that's a future upgrade with cmudict).
|
|
18
|
+
- **Cost**: ~3–5 seconds CPU inference for a 10s clip on first call;
|
|
19
|
+
subsequent calls in the same process re-use the cached model.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import io
|
|
25
|
+
import tempfile
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import Iterable
|
|
28
|
+
|
|
29
|
+
from an.audio.lipsync import (
|
|
30
|
+
LipSyncProvider,
|
|
31
|
+
Viseme,
|
|
32
|
+
VisemeTrack,
|
|
33
|
+
WordTiming,
|
|
34
|
+
word_timings_to_visemes,
|
|
35
|
+
)
|
|
36
|
+
from an.audio.tts import AudioClip
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
# Reuse the offline char→viseme mapping so the two providers stay consistent.
|
|
40
|
+
from cutan.audio.offline_lipsync import _CHAR_TO_VISEME, _REST_VISEME
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
_DEFAULT_MODEL_SIZE: str = "tiny" # ~75 MB; "base" gives slightly better word-timing
|
|
44
|
+
_DEFAULT_DEVICE: str = "cpu"
|
|
45
|
+
_DEFAULT_COMPUTE_TYPE: str = "int8"
|
|
46
|
+
_MIN_WORD_GAP_FOR_REST: float = (
|
|
47
|
+
0.20 # seconds; insert rest viseme in gaps wider than this
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class WhisperLipSync:
|
|
52
|
+
"""faster-whisper word timestamps → visemes.
|
|
53
|
+
|
|
54
|
+
Implements the ``LipSyncProvider`` protocol. The model is lazy-loaded on
|
|
55
|
+
the first call (subsequent calls in the same process reuse the instance
|
|
56
|
+
via the class-level ``_model`` cache).
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
#: Whisper aligns from words, so the track carries them (an#96).
|
|
60
|
+
emits_word_timings: bool = True
|
|
61
|
+
|
|
62
|
+
name: str = "whisper"
|
|
63
|
+
convention: str = "rhubarb"
|
|
64
|
+
|
|
65
|
+
_model = None # class-level cache so repeated align() calls share the model
|
|
66
|
+
|
|
67
|
+
def __init__(
|
|
68
|
+
self,
|
|
69
|
+
*,
|
|
70
|
+
model_size: str = _DEFAULT_MODEL_SIZE,
|
|
71
|
+
device: str = _DEFAULT_DEVICE,
|
|
72
|
+
compute_type: str = _DEFAULT_COMPUTE_TYPE,
|
|
73
|
+
char_to_viseme: dict[str, str] | None = None,
|
|
74
|
+
) -> None:
|
|
75
|
+
self.model_size = model_size
|
|
76
|
+
self.device = device
|
|
77
|
+
self.compute_type = compute_type
|
|
78
|
+
self._mapping = (
|
|
79
|
+
dict(char_to_viseme) if char_to_viseme else dict(_CHAR_TO_VISEME)
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
def _get_model(self):
|
|
83
|
+
if WhisperLipSync._model is not None:
|
|
84
|
+
return WhisperLipSync._model
|
|
85
|
+
try:
|
|
86
|
+
from faster_whisper import WhisperModel # type: ignore
|
|
87
|
+
except ImportError as e:
|
|
88
|
+
raise RuntimeError(
|
|
89
|
+
"faster-whisper not installed. pip install faster-whisper "
|
|
90
|
+
"(adds ~100 MB) or use the offline lipsync provider."
|
|
91
|
+
) from e
|
|
92
|
+
WhisperLipSync._model = WhisperModel(
|
|
93
|
+
self.model_size, device=self.device, compute_type=self.compute_type
|
|
94
|
+
)
|
|
95
|
+
return WhisperLipSync._model
|
|
96
|
+
|
|
97
|
+
def align(self, audio: AudioClip, transcript: str) -> VisemeTrack:
|
|
98
|
+
# faster-whisper takes a path or file-like. Materialize bytes to disk
|
|
99
|
+
# if the caller didn't pass a path.
|
|
100
|
+
if audio.path and Path(audio.path).exists():
|
|
101
|
+
audio_path: str | Path = audio.path
|
|
102
|
+
cleanup: Path | None = None
|
|
103
|
+
elif audio.bytes_:
|
|
104
|
+
tmp = tempfile.NamedTemporaryFile(delete=False, suffix=".bin")
|
|
105
|
+
tmp.write(audio.bytes_)
|
|
106
|
+
tmp.close()
|
|
107
|
+
audio_path = tmp.name
|
|
108
|
+
cleanup = Path(tmp.name)
|
|
109
|
+
else:
|
|
110
|
+
raise ValueError("AudioClip needs either .path or .bytes_")
|
|
111
|
+
|
|
112
|
+
try:
|
|
113
|
+
model = self._get_model()
|
|
114
|
+
segments, _info = model.transcribe(
|
|
115
|
+
str(audio_path),
|
|
116
|
+
word_timestamps=True,
|
|
117
|
+
# faster-whisper auto-detects language; pass the transcript as
|
|
118
|
+
# an "initial prompt" to bias decoding toward the known text.
|
|
119
|
+
initial_prompt=transcript or None,
|
|
120
|
+
)
|
|
121
|
+
words: list[WordTiming] = []
|
|
122
|
+
for seg in segments:
|
|
123
|
+
for w in seg.words or []:
|
|
124
|
+
words.append((w.word, float(w.start), float(w.end)))
|
|
125
|
+
finally:
|
|
126
|
+
if cleanup is not None:
|
|
127
|
+
try:
|
|
128
|
+
cleanup.unlink()
|
|
129
|
+
except OSError:
|
|
130
|
+
pass
|
|
131
|
+
|
|
132
|
+
return VisemeTrack(
|
|
133
|
+
visemes=word_timings_to_visemes(
|
|
134
|
+
words,
|
|
135
|
+
total_duration=audio.duration,
|
|
136
|
+
char_to_viseme=self._mapping,
|
|
137
|
+
rest_viseme=_REST_VISEME,
|
|
138
|
+
min_gap_for_rest=_MIN_WORD_GAP_FOR_REST,
|
|
139
|
+
),
|
|
140
|
+
convention=self.convention,
|
|
141
|
+
duration=audio.duration,
|
|
142
|
+
words=list(words),
|
|
143
|
+
)
|