cutan 0.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. cutan/__init__.py +80 -0
  2. cutan/audio/__init__.py +50 -0
  3. cutan/audio/injectable_lipsync.py +123 -0
  4. cutan/audio/offline_lipsync.py +126 -0
  5. cutan/audio/rhubarb_lipsync.py +152 -0
  6. cutan/audio/whisper_lipsync.py +143 -0
  7. cutan/bench.py +244 -0
  8. cutan/characters/__init__.py +125 -0
  9. cutan/characters/brows.py +504 -0
  10. cutan/characters/checks.py +610 -0
  11. cutan/characters/cli.py +629 -0
  12. cutan/characters/colour_roles.py +176 -0
  13. cutan/characters/dicebear.py +238 -0
  14. cutan/characters/drawn.py +94 -0
  15. cutan/characters/factory.py +2237 -0
  16. cutan/characters/idle.py +220 -0
  17. cutan/characters/licenses.py +385 -0
  18. cutan/characters/methods.py +708 -0
  19. cutan/characters/mouth_set.py +235 -0
  20. cutan/characters/play.py +1244 -0
  21. cutan/characters/promote.py +189 -0
  22. cutan/characters/record.py +167 -0
  23. cutan/characters/registration.py +239 -0
  24. cutan/characters/schema.py +921 -0
  25. cutan/characters/silhouette.py +155 -0
  26. cutan/characters/svg_utils.py +346 -0
  27. cutan/characters/validate.py +946 -0
  28. cutan/characters/vocabulary.py +194 -0
  29. cutan/compile/__init__.py +1 -0
  30. cutan/compile/coarticulate.py +445 -0
  31. cutan/compile/gaze.py +147 -0
  32. cutan/compile/lowering.py +109 -0
  33. cutan/compile/passes.py +2581 -0
  34. cutan/conftest.py +8 -0
  35. cutan/data/dicebear_9x_styles.json +38 -0
  36. cutan/expression/__init__.py +80 -0
  37. cutan/expression/axes.py +122 -0
  38. cutan/expression/binding.py +326 -0
  39. cutan/expression/blendshapes.py +125 -0
  40. cutan/expression/presets.py +203 -0
  41. cutan/expression/provider.py +210 -0
  42. cutan/expression/registration.py +187 -0
  43. cutan/genre.py +220 -0
  44. cutan/impacts/__init__.py +89 -0
  45. cutan/impacts/cli.py +166 -0
  46. cutan/impacts/clip.py +603 -0
  47. cutan/impacts/objects.py +255 -0
  48. cutan/impacts/performance.py +257 -0
  49. cutan/impacts/stroke.py +252 -0
  50. cutan/impacts/truth.py +368 -0
  51. cutan/library.py +318 -0
  52. cutan/nw.py +58 -0
  53. cutan/runtime/__init__.py +1 -0
  54. cutan/runtime/visuals.js +129 -0
  55. cutan/verify/__init__.py +1 -0
  56. cutan/verify/style.py +1194 -0
  57. cutan-0.0.2.dist-info/METADATA +64 -0
  58. cutan-0.0.2.dist-info/RECORD +61 -0
  59. cutan-0.0.2.dist-info/WHEEL +4 -0
  60. cutan-0.0.2.dist-info/entry_points.txt +2 -0
  61. cutan-0.0.2.dist-info/licenses/LICENSE +21 -0
cutan/__init__.py ADDED
@@ -0,0 +1,80 @@
1
+ """cutan: cut-out animation, as a genre package on the ``an`` core.
2
+
3
+ ``an`` is the core of structured animation (scene documents, the timing kernel,
4
+ renderers, audio, storage, verification). A *genre* adds what one kind of
5
+ animation needs and registers it with the core; ``cutan`` is the first genre to
6
+ leave ``an``: rigged characters, faces and expressions, lip-sync visemes, swap
7
+ sets and views, cut-out styles and impacts (ADR 0001 in ``an``'s
8
+ ``misc/docs/adr/``, tracked by an#225 under the epic an#231).
9
+
10
+ Rigged characters (``cutan.characters``), faces (``cutan.expression``), impacts
11
+ (``cutan.impacts``), the cut-out compile passes (``cutan.compile``), lip-sync
12
+ providers (``cutan.audio``), the style lint (``cutan.verify``) and the genre object
13
+ (``cutan.genre``) lived inside ``an`` until the P8 move (an#225); ``an`` keeps
14
+ warning aliases at the old import paths. Install it with ``pip install "an[cutout]"``;
15
+ ``an`` finds it through the ``an.genres`` entry point.
16
+
17
+ The identifiers below are the genre's **persisted** names (ADR 0001 decision 9):
18
+ they are written into documents, stores and entry-point metadata, and none of
19
+ them changes when code moves between distributions.
20
+
21
+ >>> GENRE_NAME
22
+ 'cutout_animation'
23
+ >>> ENTRY_POINT_GROUP, ENTRY_POINT_NAME
24
+ ('an.genres', 'cutout_animation')
25
+ >>> RENDERER_NAME, LIBRARY_NAME
26
+ ('cutout', 'cutan')
27
+ """
28
+
29
+ #: The genre's slug: the ``an.genres`` entry-point name and the ``nw`` genre id.
30
+ GENRE_NAME: str = "cutout_animation"
31
+
32
+ #: The entry-point group ``an.genres.load()`` reads.
33
+ ENTRY_POINT_GROUP: str = "an.genres"
34
+
35
+ #: The entry-point name this distribution declares (``pyproject.toml``): the name
36
+ #: ``an`` used to declare for the in-distribution genre, so the handover was by name.
37
+ ENTRY_POINT_NAME: str = GENRE_NAME
38
+
39
+ #: Where the entry point points.
40
+ ENTRY_POINT_VALUE: str = "cutan.genre:CUTOUT"
41
+
42
+ #: The persisted renderer name of cut-out shots (``an.stage`` claims it).
43
+ RENDERER_NAME: str = "cutout"
44
+
45
+ #: The package whose data root holds the genre's asset library and projects
46
+ #: (``~/.local/share/cutan`` by default; ``CUTAN_HOME`` overrides it).
47
+ LIBRARY_NAME: str = "cutan"
48
+
49
+ #: The lowest ``an.genres.API_LEVEL`` this ``cutan`` runs against ("the lowest ``an``
50
+ #: it supports", ADR 0001 decision 8, said without a version pin: ``an``'s version is
51
+ #: assigned by CI at merge). Level 2 is the move itself: ``Genre.services``,
52
+ #: ``ActionKind.lowering``, ``EntityKind.swap_declaration`` and ``an.stage.rig``.
53
+ REQUIRED_AN_API_LEVEL: int = 2
54
+
55
+
56
+ def require_an() -> None:
57
+ """Refuse, with an upgrade hint, to load against an ``an`` older than this ``cutan`` needs.
58
+
59
+ >>> require_an()
60
+ """
61
+ import an.genres
62
+
63
+ level = getattr(an.genres, "API_LEVEL", 0)
64
+ if level < REQUIRED_AN_API_LEVEL:
65
+ raise ImportError(
66
+ f"cutan needs an.genres API level {REQUIRED_AN_API_LEVEL} or higher; this an "
67
+ f"provides {level}. Upgrade an: pip install -U an"
68
+ )
69
+
70
+
71
+ __all__ = [
72
+ "ENTRY_POINT_GROUP",
73
+ "ENTRY_POINT_NAME",
74
+ "ENTRY_POINT_VALUE",
75
+ "GENRE_NAME",
76
+ "LIBRARY_NAME",
77
+ "REQUIRED_AN_API_LEVEL",
78
+ "RENDERER_NAME",
79
+ "require_an",
80
+ ]
@@ -0,0 +1,50 @@
1
+ """The cut-out genre's lip-sync providers: letters, Rhubarb and word timings to mouth shapes.
2
+
3
+ Moved from ``an.audio`` (an#225). ``an.audio`` keeps the protocols
4
+ (:class:`~an.audio.lipsync.LipSyncProvider`, :class:`~an.audio.lipsync.VisemeTrack`),
5
+ text-to-speech and the pipeline; a viseme only means something to a genre that
6
+ draws mouths. The genre registers these by name as ``lipsync.<name>`` services
7
+ (:mod:`cutan.genre`), which is how ``render(lipsync="offline")`` and
8
+ ``an render --lipsync rhubarb`` keep working.
9
+
10
+ >>> offline_factory().name
11
+ 'offline'
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from cutan.audio.injectable_lipsync import StaticWordTimings, WordTimingsLipSync
17
+ from cutan.audio.offline_lipsync import OfflineLipSync
18
+ from cutan.audio.rhubarb_lipsync import RhubarbLipSync
19
+ from cutan.audio.whisper_lipsync import WhisperLipSync
20
+
21
+ #: The language a provider aligns for when the caller says nothing (Rhubarb's
22
+ #: recognizer follows it, an#96): the same default ``an.audio.providers`` has.
23
+ DFLT_LANGUAGE: str = "en"
24
+
25
+
26
+ def offline_factory(**_: object) -> OfflineLipSync:
27
+ """The deterministic char-to-viseme provider (the default)."""
28
+ return OfflineLipSync()
29
+
30
+
31
+ def rhubarb_factory(*, language: str = DFLT_LANGUAGE, **_: object) -> RhubarbLipSync:
32
+ """Rhubarb Lip Sync (needs the ``rhubarb`` binary); ``language`` picks its recognizer."""
33
+ return RhubarbLipSync(language=language)
34
+
35
+
36
+ def whisper_factory(**_: object) -> WhisperLipSync:
37
+ """Word timings from Whisper, distributed over the mouth shapes."""
38
+ return WhisperLipSync()
39
+
40
+
41
+ __all__ = [
42
+ "OfflineLipSync",
43
+ "RhubarbLipSync",
44
+ "StaticWordTimings",
45
+ "WhisperLipSync",
46
+ "WordTimingsLipSync",
47
+ "offline_factory",
48
+ "rhubarb_factory",
49
+ "whisper_factory",
50
+ ]
@@ -0,0 +1,123 @@
1
+ """Lip-sync provider that consumes pre-computed word timings.
2
+
3
+ Useful when an upstream system already has authoritative word-level
4
+ timings and would otherwise force ``an`` to re-transcribe the same
5
+ audio. The canonical case is ``muvid``, where the lyric → audio
6
+ alignment store (``lacing``) is the SSOT and re-running
7
+ ``WhisperLipSync`` on the audio produces a redundant (and possibly
8
+ divergent) word-timestamp set.
9
+
10
+ Two pieces:
11
+
12
+ - :class:`StaticWordTimings` — a :class:`WordTimingProvider` over a
13
+ fixed list of ``(word, start, end)`` tuples.
14
+ - :class:`WordTimingsLipSync` — a :class:`LipSyncProvider` that reads
15
+ from any :class:`WordTimingProvider` and runs the same
16
+ word→viseme conversion as :class:`WhisperLipSync` (via
17
+ :func:`word_timings_to_visemes`), so output is shape-compatible with
18
+ the rest of the cutout pipeline.
19
+
20
+ Drop-in usage::
21
+
22
+ from cutan.audio.injectable_lipsync import (
23
+ StaticWordTimings, WordTimingsLipSync,
24
+ )
25
+
26
+ timings = [("hello", 0.5, 1.0), ("world", 1.2, 1.8)]
27
+ lipsync = WordTimingsLipSync(StaticWordTimings(timings))
28
+ track = lipsync.align(audio_clip, "hello world")
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ from typing import Sequence
34
+
35
+ from an.audio.lipsync import (
36
+ LipSyncProvider,
37
+ VisemeTrack,
38
+ WordTiming,
39
+ WordTimingProvider,
40
+ word_timings_to_visemes,
41
+ )
42
+ from cutan.audio.offline_lipsync import _CHAR_TO_VISEME, _REST_VISEME
43
+ from an.audio.tts import AudioClip
44
+
45
+
46
+ class StaticWordTimings:
47
+ """A :class:`WordTimingProvider` over a fixed list of timings."""
48
+
49
+ name: str = "static"
50
+
51
+ def __init__(self, words: Sequence[WordTiming], *, label: str = "static") -> None:
52
+ self._words = tuple(words)
53
+ self.name = label
54
+
55
+ def words_for(
56
+ self, audio: AudioClip, *, transcript: str = ""
57
+ ) -> Sequence[WordTiming]:
58
+ return self._words
59
+
60
+
61
+ class WordTimingsLipSync:
62
+ """:class:`LipSyncProvider` driven by a :class:`WordTimingProvider`.
63
+
64
+ Skips transcription entirely. Use this when the caller already has
65
+ authoritative word timings (e.g. from a separate lyric-alignment
66
+ pipeline).
67
+
68
+ Args:
69
+ provider: any :class:`WordTimingProvider`.
70
+ char_to_viseme: optional override of the character→viseme code
71
+ mapping; defaults to the one shared with
72
+ :class:`OfflineLipSync` / :class:`WhisperLipSync`.
73
+ convention: declared viseme convention string for the produced
74
+ track. Defaults to ``"rhubarb"`` for compatibility with the
75
+ existing cutout adapter.
76
+ rest_viseme: code emitted in silent gaps. Defaults to
77
+ :data:`_REST_VISEME`.
78
+ min_gap_for_rest: minimum inter-word silence (seconds) before
79
+ we insert a rest keyframe. Defaults to ``0.20``.
80
+ """
81
+
82
+ convention: str = "rhubarb"
83
+
84
+ def __init__(
85
+ self,
86
+ provider: WordTimingProvider,
87
+ *,
88
+ char_to_viseme: dict[str, str] | None = None,
89
+ convention: str = "rhubarb",
90
+ rest_viseme: str = _REST_VISEME,
91
+ min_gap_for_rest: float = 0.20,
92
+ ) -> None:
93
+ self.provider = provider
94
+ self.convention = convention
95
+ self._mapping = (
96
+ dict(char_to_viseme) if char_to_viseme else dict(_CHAR_TO_VISEME)
97
+ )
98
+ self._rest = rest_viseme
99
+ self._min_gap = min_gap_for_rest
100
+
101
+ @property
102
+ def name(self) -> str:
103
+ # Disambiguate cache keys per-provider so swapping providers
104
+ # invalidates the viseme cache.
105
+ return f"word-timings:{getattr(self.provider, 'name', 'unknown')}"
106
+
107
+ #: Built from words, so the track carries them (an#96).
108
+ emits_word_timings: bool = True
109
+
110
+ def align(self, audio: AudioClip, transcript: str) -> VisemeTrack:
111
+ words = list(self.provider.words_for(audio, transcript=transcript))
112
+ return VisemeTrack(
113
+ visemes=word_timings_to_visemes(
114
+ words,
115
+ total_duration=audio.duration,
116
+ char_to_viseme=self._mapping,
117
+ rest_viseme=self._rest,
118
+ min_gap_for_rest=self._min_gap,
119
+ ),
120
+ convention=self.convention,
121
+ duration=audio.duration,
122
+ words=words,
123
+ )
@@ -0,0 +1,126 @@
1
+ """OfflineLipSync — deterministic transcript → viseme track. No network, no binary.
2
+
3
+ The default lip-sync provider for `an`. Maps each meaningful character of the
4
+ transcript to a Rhubarb-convention viseme letter (A–H + X), then distributes
5
+ keyframes evenly across the audio's duration. Repeated visemes get collapsed
6
+ into a single keyframe so the mouth doesn't "stutter" on long vowel runs.
7
+
8
+ Crude but visible. Use ``RhubarbLipSync`` for real phoneme alignment once
9
+ you've installed the Rhubarb binary.
10
+
11
+ >>> from an.audio.tts import AudioClip
12
+ >>> ls = OfflineLipSync()
13
+ >>> track = ls.align(AudioClip(duration=1.0, transcript="hello"), "hello")
14
+ >>> track.convention
15
+ 'rhubarb'
16
+ >>> track.duration
17
+ 1.0
18
+ >>> len(track.visemes) >= 2
19
+ True
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from typing import Iterable
25
+
26
+ from an.audio.lipsync import Viseme, VisemeTrack
27
+ from an.audio.tts import AudioClip
28
+
29
+
30
+ # Map a single ASCII character (lowercased) to a Rhubarb viseme letter.
31
+ # A=closed (P/B/M, rest), B=tight open (D/S/T/Z/N/L), C=eh, D=ah/wide,
32
+ # E=ohh, F=ooh, G=teeth-on-lip (F/V), H=th, X=idle.
33
+ _CHAR_TO_VISEME: dict[str, str] = {
34
+ # closed lips
35
+ "p": "A",
36
+ "b": "A",
37
+ "m": "A",
38
+ # teeth-on-lip
39
+ "f": "G",
40
+ "v": "G",
41
+ # th-ish
42
+ "h": "C",
43
+ # vowel families
44
+ "a": "D",
45
+ "e": "C",
46
+ "i": "B",
47
+ "o": "E",
48
+ "u": "F",
49
+ "y": "B",
50
+ # alveolar consonants
51
+ "d": "B",
52
+ "t": "B",
53
+ "s": "B",
54
+ "z": "B",
55
+ "n": "B",
56
+ "l": "B",
57
+ "r": "B",
58
+ # remaining consonants — neutral semi-open
59
+ "c": "C",
60
+ "g": "C",
61
+ "j": "C",
62
+ "k": "C",
63
+ "q": "C",
64
+ "w": "F",
65
+ "x": "C",
66
+ }
67
+ _REST_VISEME: str = "X"
68
+
69
+
70
+ class OfflineLipSync:
71
+ """Default lip-sync provider: deterministic char-to-viseme mapping.
72
+
73
+ Implements the ``LipSyncProvider`` protocol.
74
+ """
75
+
76
+ name: str = "offline"
77
+ convention: str = "rhubarb"
78
+
79
+ def __init__(self, *, char_to_viseme: dict[str, str] | None = None) -> None:
80
+ self._mapping = (
81
+ dict(char_to_viseme) if char_to_viseme else dict(_CHAR_TO_VISEME)
82
+ )
83
+
84
+ def align(self, audio: AudioClip, transcript: str) -> VisemeTrack:
85
+ codes = list(self._codes_for(transcript))
86
+ if not codes:
87
+ # Empty transcript → just rest at start, rest at end.
88
+ return VisemeTrack(
89
+ visemes=[
90
+ Viseme(time=0.0, code=_REST_VISEME),
91
+ Viseme(time=max(0.0, audio.duration), code=_REST_VISEME),
92
+ ],
93
+ convention=self.convention,
94
+ duration=audio.duration,
95
+ )
96
+ # Distribute one keyframe per code over [0, duration].
97
+ n = len(codes)
98
+ duration = max(audio.duration, 1e-3)
99
+ visemes: list[Viseme] = []
100
+ # Always lead with rest at t=0
101
+ visemes.append(Viseme(time=0.0, code=_REST_VISEME))
102
+ for i, code in enumerate(codes):
103
+ t = (i + 1) / (n + 1) * duration
104
+ visemes.append(Viseme(time=t, code=code))
105
+ # Trailing rest
106
+ visemes.append(Viseme(time=duration, code=_REST_VISEME))
107
+ return VisemeTrack(
108
+ visemes=visemes, convention=self.convention, duration=audio.duration
109
+ )
110
+
111
+ def _codes_for(self, transcript: str) -> Iterable[str]:
112
+ """Per-character viseme codes; collapses adjacent duplicates."""
113
+ previous: str | None = None
114
+ for raw in transcript:
115
+ ch = raw.lower()
116
+ if not ch.isalpha():
117
+ # Spaces, punctuation → mouth rest. Insert if not redundant.
118
+ if previous != _REST_VISEME:
119
+ previous = _REST_VISEME
120
+ yield _REST_VISEME
121
+ continue
122
+ code = self._mapping.get(ch, "C")
123
+ if code == previous:
124
+ continue
125
+ previous = code
126
+ yield code
@@ -0,0 +1,152 @@
1
+ """RhubarbLipSync — calls the rhubarb-lip-sync binary for phoneme-aligned visemes.
2
+
3
+ Requires the ``rhubarb`` binary on PATH. macOS: ``brew install rhubarb-lipsync``.
4
+ Linux/Windows: download from the project's GitHub releases.
5
+
6
+ Falls back gracefully (raises a clear error) if the binary is missing — the
7
+ default ``OfflineLipSync`` keeps the pipeline functional in the meantime.
8
+
9
+ **The recognizer follows the language** (an#96, epic #9 defect 5a). Rhubarb has
10
+ two: ``pocketSphinx`` — its default, "use for English recordings", the only one
11
+ that reads ``--dialogFile`` (it builds a dialog language model and mixes it 90/10
12
+ with the default) — and ``phonetic``, "use for non-English recordings", which
13
+ ``UNUSED(dialog)``s the transcript at source. This module used to pass
14
+ ``-r phonetic`` **and** ``--dialogFile`` unconditionally: English speech from an
15
+ English transcript ran the language-independent recognizer and the transcript
16
+ it wrote to disk was never read. Now ``recognizer=None`` (the default) resolves
17
+ per ``language`` — ``"en"`` → ``pocketSphinx`` with the dialog file, anything
18
+ else → ``phonetic`` and **no transcript is written** (a file nothing reads is a
19
+ lie waiting for the next reader). An explicit ``recognizer`` still overrides.
20
+ ``name`` carries the recognizer so the viseme cache key changes with it and no
21
+ stale ``phonetic`` track replays.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ import json
27
+ import shutil
28
+ import subprocess
29
+ import tempfile
30
+ from pathlib import Path
31
+
32
+ from an.audio.lipsync import Viseme, VisemeTrack
33
+ from an.audio.tts import AudioClip
34
+
35
+
36
+ _DEFAULT_TIMEOUT_S: float = 60.0
37
+ #: Rhubarb's own default and its English recognizer — the one that reads the
38
+ #: dialog file.
39
+ _ENGLISH_RECOGNIZER: str = "pocketSphinx"
40
+ #: Language-independent; ignores the dialog file at source.
41
+ _PHONETIC_RECOGNIZER: str = "phonetic"
42
+ #: The languages `pocketSphinx` (CMU Sphinx US English acoustic model) covers.
43
+ ENGLISH_LANGUAGES: frozenset[str] = frozenset({"en"})
44
+ RECOGNIZERS: frozenset[str] = frozenset({_ENGLISH_RECOGNIZER, _PHONETIC_RECOGNIZER})
45
+
46
+
47
+ def recognizer_for(language: str) -> str:
48
+ """The Rhubarb recognizer for a BCP-47 language tag (primary subtag only).
49
+
50
+ Accepts the POSIX locale spelling too (``en_US``); an empty tag is refused
51
+ rather than read as "non-English" (an#96 review).
52
+
53
+ >>> recognizer_for("en"), recognizer_for("en-GB"), recognizer_for("en_US"), recognizer_for("fr")
54
+ ('pocketSphinx', 'pocketSphinx', 'pocketSphinx', 'phonetic')
55
+ """
56
+ primary = language.replace("_", "-").split("-", 1)[0].strip().lower()
57
+ if not primary:
58
+ raise ValueError("language must be a BCP-47 tag such as 'en' or 'fr'; got ''")
59
+ return _ENGLISH_RECOGNIZER if primary in ENGLISH_LANGUAGES else _PHONETIC_RECOGNIZER
60
+
61
+
62
+ class RhubarbLipSync:
63
+ """Wrap the rhubarb CLI. Implements the ``LipSyncProvider`` protocol.
64
+
65
+ >>> RhubarbLipSync(binary_path="/bin/rhubarb").recognizer
66
+ 'pocketSphinx'
67
+ >>> RhubarbLipSync(binary_path="/bin/rhubarb", language="de").recognizer
68
+ 'phonetic'
69
+ >>> RhubarbLipSync(binary_path="/bin/rhubarb", language="de").name
70
+ 'rhubarb:phonetic'
71
+ """
72
+
73
+ convention: str = "rhubarb"
74
+
75
+ def __init__(
76
+ self,
77
+ *,
78
+ binary_path: str | None = None,
79
+ language: str = "en",
80
+ recognizer: str | None = None,
81
+ timeout_s: float = _DEFAULT_TIMEOUT_S,
82
+ ) -> None:
83
+ self.binary_path = binary_path or shutil.which("rhubarb")
84
+ self.language = language
85
+ chosen = recognizer if recognizer is not None else recognizer_for(language)
86
+ if chosen not in RECOGNIZERS:
87
+ raise ValueError(
88
+ f"unknown rhubarb recognizer {chosen!r}; known: {sorted(RECOGNIZERS)}"
89
+ )
90
+ self.recognizer = chosen
91
+ self.timeout_s = timeout_s
92
+
93
+ @property
94
+ def name(self) -> str:
95
+ # The recognizer is part of the identity: it changes the track, so it
96
+ # must change the viseme cache key (the pipeline hashes `name`).
97
+ return f"rhubarb:{self.recognizer}"
98
+
99
+ @property
100
+ def uses_dialog_file(self) -> bool:
101
+ """Whether the chosen recognizer reads a transcript at all."""
102
+ return self.recognizer == _ENGLISH_RECOGNIZER
103
+
104
+ def align(self, audio: AudioClip, transcript: str) -> VisemeTrack:
105
+ if not self.binary_path:
106
+ raise RuntimeError(
107
+ "rhubarb binary not found on PATH. Install with: "
108
+ "brew install rhubarb-lipsync (macOS) or grab a release from "
109
+ "https://github.com/DanielSWolf/rhubarb-lip-sync/releases."
110
+ )
111
+ with tempfile.TemporaryDirectory() as d:
112
+ d = Path(d)
113
+ audio_path = audio.path
114
+ if audio_path is None:
115
+ if audio.bytes_ is None:
116
+ raise ValueError("AudioClip needs either .path or .bytes_")
117
+ audio_path = d / "audio.wav"
118
+ audio_path.write_bytes(audio.bytes_)
119
+
120
+ out_json = d / "out.json"
121
+ cmd = [self.binary_path, "-f", "json", "-r", self.recognizer]
122
+ if self.uses_dialog_file:
123
+ dialog_path = d / "transcript.txt"
124
+ dialog_path.write_text(transcript, encoding="utf-8")
125
+ cmd += ["--dialogFile", str(dialog_path)]
126
+ cmd += ["-o", str(out_json), str(audio_path)]
127
+ try:
128
+ subprocess.run(
129
+ cmd,
130
+ capture_output=True,
131
+ text=True,
132
+ timeout=self.timeout_s,
133
+ check=True,
134
+ )
135
+ except subprocess.CalledProcessError as e:
136
+ raise RuntimeError(
137
+ f"rhubarb failed (rc={e.returncode}): {e.stderr}"
138
+ ) from e
139
+ data = json.loads(out_json.read_text(encoding="utf-8"))
140
+
141
+ cues = data.get("mouthCues", [])
142
+ visemes = [Viseme(time=float(c["start"]), code=str(c["value"])) for c in cues]
143
+ if cues:
144
+ # Append a final rest at the last cue's "end" so the track matches duration.
145
+ last_end = float(cues[-1]["end"])
146
+ if not visemes or visemes[-1].code != "X":
147
+ visemes.append(Viseme(time=last_end, code="X"))
148
+ return VisemeTrack(
149
+ visemes=visemes,
150
+ convention=self.convention,
151
+ duration=audio.duration,
152
+ )
@@ -0,0 +1,143 @@
1
+ """WhisperLipSync — faster-whisper word timestamps → viseme keyframes.
2
+
3
+ Phase 9. Bridges the gap between deterministic ``OfflineLipSync`` (twitchy,
4
+ char-distributed) and the system-binary-dependent ``RhubarbLipSync``. Uses
5
+ ``faster-whisper`` to transcribe the rendered audio into word-level
6
+ timestamps, then distributes visemes within each word's [start, end] span
7
+ based on the word's letter→viseme mapping (collapsed-duplicates).
8
+
9
+ This gives ~75% of Rhubarb-quality lip-sync without any system binaries,
10
+ just a ~75 MB model download (cached after first use).
11
+
12
+ Trade-offs vs. OfflineLipSync:
13
+
14
+ - **Better**: timing is locked to actual word boundaries. Mouth holds shape
15
+ through silent gaps between words instead of cycling through visemes.
16
+ - **Same**: viseme codes per phoneme are still our simple letter mapping;
17
+ no IPA/ARPAbet awareness yet (that's a future upgrade with cmudict).
18
+ - **Cost**: ~3–5 seconds CPU inference for a 10s clip on first call;
19
+ subsequent calls in the same process re-use the cached model.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import io
25
+ import tempfile
26
+ from pathlib import Path
27
+ from typing import Iterable
28
+
29
+ from an.audio.lipsync import (
30
+ LipSyncProvider,
31
+ Viseme,
32
+ VisemeTrack,
33
+ WordTiming,
34
+ word_timings_to_visemes,
35
+ )
36
+ from an.audio.tts import AudioClip
37
+
38
+
39
+ # Reuse the offline char→viseme mapping so the two providers stay consistent.
40
+ from cutan.audio.offline_lipsync import _CHAR_TO_VISEME, _REST_VISEME
41
+
42
+
43
+ _DEFAULT_MODEL_SIZE: str = "tiny" # ~75 MB; "base" gives slightly better word-timing
44
+ _DEFAULT_DEVICE: str = "cpu"
45
+ _DEFAULT_COMPUTE_TYPE: str = "int8"
46
+ _MIN_WORD_GAP_FOR_REST: float = (
47
+ 0.20 # seconds; insert rest viseme in gaps wider than this
48
+ )
49
+
50
+
51
+ class WhisperLipSync:
52
+ """faster-whisper word timestamps → visemes.
53
+
54
+ Implements the ``LipSyncProvider`` protocol. The model is lazy-loaded on
55
+ the first call (subsequent calls in the same process reuse the instance
56
+ via the class-level ``_model`` cache).
57
+ """
58
+
59
+ #: Whisper aligns from words, so the track carries them (an#96).
60
+ emits_word_timings: bool = True
61
+
62
+ name: str = "whisper"
63
+ convention: str = "rhubarb"
64
+
65
+ _model = None # class-level cache so repeated align() calls share the model
66
+
67
+ def __init__(
68
+ self,
69
+ *,
70
+ model_size: str = _DEFAULT_MODEL_SIZE,
71
+ device: str = _DEFAULT_DEVICE,
72
+ compute_type: str = _DEFAULT_COMPUTE_TYPE,
73
+ char_to_viseme: dict[str, str] | None = None,
74
+ ) -> None:
75
+ self.model_size = model_size
76
+ self.device = device
77
+ self.compute_type = compute_type
78
+ self._mapping = (
79
+ dict(char_to_viseme) if char_to_viseme else dict(_CHAR_TO_VISEME)
80
+ )
81
+
82
+ def _get_model(self):
83
+ if WhisperLipSync._model is not None:
84
+ return WhisperLipSync._model
85
+ try:
86
+ from faster_whisper import WhisperModel # type: ignore
87
+ except ImportError as e:
88
+ raise RuntimeError(
89
+ "faster-whisper not installed. pip install faster-whisper "
90
+ "(adds ~100 MB) or use the offline lipsync provider."
91
+ ) from e
92
+ WhisperLipSync._model = WhisperModel(
93
+ self.model_size, device=self.device, compute_type=self.compute_type
94
+ )
95
+ return WhisperLipSync._model
96
+
97
+ def align(self, audio: AudioClip, transcript: str) -> VisemeTrack:
98
+ # faster-whisper takes a path or file-like. Materialize bytes to disk
99
+ # if the caller didn't pass a path.
100
+ if audio.path and Path(audio.path).exists():
101
+ audio_path: str | Path = audio.path
102
+ cleanup: Path | None = None
103
+ elif audio.bytes_:
104
+ tmp = tempfile.NamedTemporaryFile(delete=False, suffix=".bin")
105
+ tmp.write(audio.bytes_)
106
+ tmp.close()
107
+ audio_path = tmp.name
108
+ cleanup = Path(tmp.name)
109
+ else:
110
+ raise ValueError("AudioClip needs either .path or .bytes_")
111
+
112
+ try:
113
+ model = self._get_model()
114
+ segments, _info = model.transcribe(
115
+ str(audio_path),
116
+ word_timestamps=True,
117
+ # faster-whisper auto-detects language; pass the transcript as
118
+ # an "initial prompt" to bias decoding toward the known text.
119
+ initial_prompt=transcript or None,
120
+ )
121
+ words: list[WordTiming] = []
122
+ for seg in segments:
123
+ for w in seg.words or []:
124
+ words.append((w.word, float(w.start), float(w.end)))
125
+ finally:
126
+ if cleanup is not None:
127
+ try:
128
+ cleanup.unlink()
129
+ except OSError:
130
+ pass
131
+
132
+ return VisemeTrack(
133
+ visemes=word_timings_to_visemes(
134
+ words,
135
+ total_duration=audio.duration,
136
+ char_to_viseme=self._mapping,
137
+ rest_viseme=_REST_VISEME,
138
+ min_gap_for_rest=_MIN_WORD_GAP_FOR_REST,
139
+ ),
140
+ convention=self.convention,
141
+ duration=audio.duration,
142
+ words=list(words),
143
+ )