pipecat-bithuman 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,8 @@
1
+ .venv/
2
+ __pycache__/
3
+ .pytest_cache/
4
+ .ruff_cache/
5
+ *.egg-info/
6
+ dist/
7
+ build/
8
+ .live/
@@ -0,0 +1,24 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are listed here.
4
+ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
5
+ and the project uses [Semantic Versioning](https://semver.org/).
6
+
7
+ ## [Unreleased]
8
+
9
+ ## [0.1.0] - 2026-09-30
10
+
11
+ ### Added
12
+
13
+ - `BitHumanVideoService`: TTS audio in; lip-synced avatar video (`OutputImageRawFrame`, RGB)
14
+ and paired 16 kHz audio (`TTSAudioRawFrame`) out, rendered in-process by the
15
+ `bithuman` Python SDK.
16
+ - `TTSStoppedFrame` is held until the avatar has finished speaking the reply.
17
+ - Barge-in: `InterruptionFrame` drops the reply in flight and returns the avatar to idle.
18
+ - Lifecycle: the avatar opens on `StartFrame`; `EndFrame` drains queued speech, then closes;
19
+ `CancelFrame` and cleanup close at once. Teardown is idempotent.
20
+ - Errors: one `ErrorFrame` per failure with a Pipecat error category; the API secret is
21
+ scrubbed from error text and never logged; TTS audio passes through by default.
22
+ - `BitHumanRuntime` protocol and `runtime_factory` for tests and SDK wrappers.
23
+ - Minimal Daily example (`examples/bot.py`) and a fake-runtime test suite.
24
+ - Tested with Pipecat v1.12.0 and `bithuman` 2.11.18.
@@ -0,0 +1,24 @@
1
+ BSD 2-Clause License
2
+
3
+ Copyright (c) 2026, bitHuman, Inc.
4
+
5
+ Redistribution and use in source and binary forms, with or without
6
+ modification, are permitted provided that the following conditions are met:
7
+
8
+ 1. Redistributions of source code must retain the above copyright notice, this
9
+ list of conditions and the following disclaimer.
10
+
11
+ 2. Redistributions in binary form must reproduce the above copyright notice,
12
+ this list of conditions and the following disclaimer in the documentation
13
+ and/or other materials provided with the distribution.
14
+
15
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
16
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
17
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
18
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
19
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
20
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
21
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
22
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
23
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
24
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
@@ -0,0 +1,164 @@
1
+ Metadata-Version: 2.5
2
+ Name: pipecat-bithuman
3
+ Version: 0.1.0
4
+ Summary: bitHuman real-time avatar video service for Pipecat: TTS audio in, lip-synced avatar video and audio out.
5
+ Project-URL: Homepage, https://www.bithuman.ai
6
+ Project-URL: Documentation, https://docs.bithuman.ai/platforms/python
7
+ Project-URL: Source, https://github.com/bithuman-product/pipecat-bithuman
8
+ Project-URL: Issues, https://github.com/bithuman-product/pipecat-bithuman/issues
9
+ Project-URL: Changelog, https://github.com/bithuman-product/pipecat-bithuman/blob/main/CHANGELOG.md
10
+ Author: bitHuman, Inc.
11
+ License-Expression: BSD-2-Clause
12
+ License-File: LICENSE
13
+ Keywords: avatar,bithuman,lip-sync,pipecat,real-time,video,voice-agent
14
+ Classifier: Development Status :: 4 - Beta
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Operating System :: MacOS
17
+ Classifier: Operating System :: POSIX :: Linux
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3 :: Only
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Programming Language :: Python :: 3.14
24
+ Classifier: Topic :: Multimedia :: Video
25
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
26
+ Requires-Python: >=3.11
27
+ Requires-Dist: bithuman<3,>=2.11.18
28
+ Requires-Dist: numpy>=1.26
29
+ Requires-Dist: pipecat-ai>=1.12.0
30
+ Provides-Extra: dev
31
+ Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
32
+ Requires-Dist: pytest>=8; extra == 'dev'
33
+ Requires-Dist: ruff>=0.6; extra == 'dev'
34
+ Provides-Extra: expression-2
35
+ Requires-Dist: bithuman[expression-2]<3,>=2.11.18; extra == 'expression-2'
36
+ Description-Content-Type: text/markdown
37
+
38
+ # pipecat-bithuman
39
+
40
+ A [bitHuman](https://www.bithuman.ai) avatar for your [Pipecat](https://github.com/pipecat-ai/pipecat) bot.
41
+ Your TTS audio goes in. A lip-synced avatar video, plus the audio that goes with it, comes out.
42
+
43
+ The avatar renders in your own process, on your own machine, through the
44
+ `bithuman` Python SDK. Both models run live on a standard Linux PC with no GPU.
45
+ Expression 2 animates a character from one portrait. Essence 2 is for photoreal people.
46
+
47
+ Maintained by bitHuman, Inc. (community integration, not maintained by the Pipecat team).
48
+
49
+ **Tested with Pipecat v1.12.0**, `bithuman` 2.11.18, Python 3.11 to 3.14.
50
+
51
+ **Demo (30 s):** [docs/demo.mp4](docs/demo.mp4), rendered through a Pipecat pipeline by
52
+ [`examples/render_demo.py`](examples/render_demo.py) with the Expression 2 sample avatar Wise Pup.
53
+
54
+ ## Install
55
+
56
+ ```bash
57
+ pip install pipecat-bithuman
58
+ # Expression 2 avatars (any character from one portrait):
59
+ pip install "pipecat-bithuman[expression-2]"
60
+ ```
61
+
62
+ ## Environment variables
63
+
64
+ | Variable | Needed | What it is |
65
+ | --- | --- | --- |
66
+ | `BITHUMAN_API_SECRET` | yes | Your bitHuman API secret. Get one at [www.bithuman.ai](https://www.bithuman.ai). The SDK reads it. This package never logs it. |
67
+ | `BITHUMAN_MODEL_PATH` | unless you pass `model_path` | Path to the avatar's `.imx` model file. |
68
+
69
+ Session time is metered while the avatar is open (talking or idle).
70
+ It closes on `EndFrame`, `CancelFrame` or cleanup.
71
+
72
+ Rendering bills active session time: 2 credits a minute on your own machine. From 2026-10-12, SDK use requires the Creator plan or higher. See docs.bithuman.ai/pricing.
73
+
74
+ ## Usage
75
+
76
+ Put `BitHumanVideoService` after the TTS service and before `transport.output()`:
77
+
78
+ ```python
79
+ from pipecat_bithuman import BitHumanVideoService
80
+
81
+ avatar = BitHumanVideoService(model_path="avatar.imx") # secret from BITHUMAN_API_SECRET
82
+
83
+ pipeline = Pipeline([
84
+ transport.input(), stt, user_aggregator, llm, tts,
85
+ avatar, # TTS audio -> avatar video + audio
86
+ transport.output(),
87
+ assistant_aggregator,
88
+ ])
89
+ ```
90
+
91
+ Turn on video out in the transport (`video_out_enabled=True`). The service logs the
92
+ frame size on the first frame; set `video_out_width` / `video_out_height` to match.
93
+
94
+ ### What the service does with frames
95
+
96
+ | In | Out |
97
+ | --- | --- |
98
+ | `TTSAudioRawFrame` | Sent to the avatar. Not forwarded as is. |
99
+ | (avatar frame) | `OutputImageRawFrame` (RGB), while talking and while idle. |
100
+ | (avatar audio) | `TTSAudioRawFrame` (16 kHz mono), paired with each picture. |
101
+ | `TTSStoppedFrame` | Held until the avatar has finished speaking the reply. |
102
+ | `InterruptionFrame` | The avatar drops the reply in flight and goes back to idle. |
103
+ | anything else | Passed on unchanged. |
104
+
105
+ If the avatar cannot start, or fails mid-session, the service pushes one
106
+ `ErrorFrame` upstream. By default TTS audio then passes through unchanged, so the bot
107
+ keeps talking without video (`audio_passthrough_on_error=False` turns this off).
108
+ Error text is scrubbed of the API secret.
109
+
110
+ ### Options
111
+
112
+ | Argument | Default | Meaning |
113
+ | --- | --- | --- |
114
+ | `model_path` | `BITHUMAN_MODEL_PATH` | The `.imx` avatar model. |
115
+ | `api_secret` | `BITHUMAN_API_SECRET` | The API secret. |
116
+ | `sync_video_to_audio` | `True` | Sets `sync_with_audio` on each image. |
117
+ | `audio_passthrough_on_error` | `True` | Keep the voice if the avatar fails. |
118
+ | `stop_frame_timeout_s` | `2.0` | Release a held `TTSStoppedFrame` after this much quiet. |
119
+ | `end_drain_timeout_s` | `30.0` | Longest wait on `EndFrame` for queued speech. |
120
+ | `runtime_factory` | SDK | Advanced: your own `BitHumanRuntime` (tests, wrappers). |
121
+
122
+ ## How it maps to the bitHuman Python SDK
123
+
124
+ | Pipecat | `bithuman.AsyncBithuman` |
125
+ | --- | --- |
126
+ | `StartFrame` | `AsyncBithuman.create(model_path=..., api_secret=...)`, then `run()` |
127
+ | `TTSAudioRawFrame` | `push_audio(pcm, sample_rate, last_chunk=False)` |
128
+ | `TTSStoppedFrame` | `flush()` |
129
+ | `InterruptionFrame` | `interrupt()` |
130
+ | `EndFrame` / `CancelFrame` / cleanup | `shutdown()` |
131
+
132
+ SDK docs: [docs.bithuman.ai/platforms/python](https://docs.bithuman.ai/platforms/python).
133
+
134
+ ## Run the example
135
+
136
+ ```bash
137
+ pip install "pipecat-bithuman[expression-2]" "pipecat-ai[daily,deepgram,openai,cartesia,silero]"
138
+ export BITHUMAN_API_SECRET=... BITHUMAN_MODEL_PATH=avatar.imx
139
+ export DAILY_ROOM_URL=... DEEPGRAM_API_KEY=... OPENAI_API_KEY=... CARTESIA_API_KEY=... CARTESIA_VOICE_ID=...
140
+ python examples/bot.py
141
+ ```
142
+
143
+ ## Tests
144
+
145
+ ```bash
146
+ pip install -e ".[dev]"
147
+ pytest # fakes only: no network, no API secret, no model file
148
+ ```
149
+
150
+ One live test talks to the real SDK. It is skipped unless you set
151
+ `PIPECAT_BITHUMAN_LIVE=1`, `BITHUMAN_API_SECRET` and `BITHUMAN_MODEL_PATH`.
152
+ It opens the avatar for a few seconds, and that time is billed.
153
+
154
+ ## Links
155
+
156
+ - bitHuman: [www.bithuman.ai](https://www.bithuman.ai)
157
+ - Docs: [docs.bithuman.ai](https://docs.bithuman.ai)
158
+ - Examples: [github.com/bithuman-product/bithuman-examples](https://github.com/bithuman-product/bithuman-examples)
159
+ - Contact: sgu@bithuman.ai
160
+
161
+ ## Licence
162
+
163
+ BSD 2-Clause, the same as Pipecat. See [LICENSE](LICENSE).
164
+ Copyright (c) 2026, bitHuman, Inc.
@@ -0,0 +1,127 @@
1
+ # pipecat-bithuman
2
+
3
+ A [bitHuman](https://www.bithuman.ai) avatar for your [Pipecat](https://github.com/pipecat-ai/pipecat) bot.
4
+ Your TTS audio goes in. A lip-synced avatar video, plus the audio that goes with it, comes out.
5
+
6
+ The avatar renders in your own process, on your own machine, through the
7
+ `bithuman` Python SDK. Both models run live on a standard Linux PC with no GPU.
8
+ Expression 2 animates a character from one portrait. Essence 2 is for photoreal people.
9
+
10
+ Maintained by bitHuman, Inc. (community integration, not maintained by the Pipecat team).
11
+
12
+ **Tested with Pipecat v1.12.0**, `bithuman` 2.11.18, Python 3.11 to 3.14.
13
+
14
+ **Demo (30 s):** [docs/demo.mp4](docs/demo.mp4), rendered through a Pipecat pipeline by
15
+ [`examples/render_demo.py`](examples/render_demo.py) with the Expression 2 sample avatar Wise Pup.
16
+
17
+ ## Install
18
+
19
+ ```bash
20
+ pip install pipecat-bithuman
21
+ # Expression 2 avatars (any character from one portrait):
22
+ pip install "pipecat-bithuman[expression-2]"
23
+ ```
24
+
25
+ ## Environment variables
26
+
27
+ | Variable | Needed | What it is |
28
+ | --- | --- | --- |
29
+ | `BITHUMAN_API_SECRET` | yes | Your bitHuman API secret. Get one at [www.bithuman.ai](https://www.bithuman.ai). The SDK reads it. This package never logs it. |
30
+ | `BITHUMAN_MODEL_PATH` | unless you pass `model_path` | Path to the avatar's `.imx` model file. |
31
+
32
+ Session time is metered while the avatar is open (talking or idle).
33
+ It closes on `EndFrame`, `CancelFrame` or cleanup.
34
+
35
+ Rendering bills active session time: 2 credits a minute on your own machine. From 2026-10-12, SDK use requires the Creator plan or higher. See docs.bithuman.ai/pricing.
36
+
37
+ ## Usage
38
+
39
+ Put `BitHumanVideoService` after the TTS service and before `transport.output()`:
40
+
41
+ ```python
42
+ from pipecat_bithuman import BitHumanVideoService
43
+
44
+ avatar = BitHumanVideoService(model_path="avatar.imx") # secret from BITHUMAN_API_SECRET
45
+
46
+ pipeline = Pipeline([
47
+ transport.input(), stt, user_aggregator, llm, tts,
48
+ avatar, # TTS audio -> avatar video + audio
49
+ transport.output(),
50
+ assistant_aggregator,
51
+ ])
52
+ ```
53
+
54
+ Turn on video out in the transport (`video_out_enabled=True`). The service logs the
55
+ frame size on the first frame; set `video_out_width` / `video_out_height` to match.
56
+
57
+ ### What the service does with frames
58
+
59
+ | In | Out |
60
+ | --- | --- |
61
+ | `TTSAudioRawFrame` | Sent to the avatar. Not forwarded as is. |
62
+ | (avatar frame) | `OutputImageRawFrame` (RGB), while talking and while idle. |
63
+ | (avatar audio) | `TTSAudioRawFrame` (16 kHz mono), paired with each picture. |
64
+ | `TTSStoppedFrame` | Held until the avatar has finished speaking the reply. |
65
+ | `InterruptionFrame` | The avatar drops the reply in flight and goes back to idle. |
66
+ | anything else | Passed on unchanged. |
67
+
68
+ If the avatar cannot start, or fails mid-session, the service pushes one
69
+ `ErrorFrame` upstream. By default TTS audio then passes through unchanged, so the bot
70
+ keeps talking without video (`audio_passthrough_on_error=False` turns this off).
71
+ Error text is scrubbed of the API secret.
72
+
73
+ ### Options
74
+
75
+ | Argument | Default | Meaning |
76
+ | --- | --- | --- |
77
+ | `model_path` | `BITHUMAN_MODEL_PATH` | The `.imx` avatar model. |
78
+ | `api_secret` | `BITHUMAN_API_SECRET` | The API secret. |
79
+ | `sync_video_to_audio` | `True` | Sets `sync_with_audio` on each image. |
80
+ | `audio_passthrough_on_error` | `True` | Keep the voice if the avatar fails. |
81
+ | `stop_frame_timeout_s` | `2.0` | Release a held `TTSStoppedFrame` after this much quiet. |
82
+ | `end_drain_timeout_s` | `30.0` | Longest wait on `EndFrame` for queued speech. |
83
+ | `runtime_factory` | SDK | Advanced: your own `BitHumanRuntime` (tests, wrappers). |
84
+
85
+ ## How it maps to the bitHuman Python SDK
86
+
87
+ | Pipecat | `bithuman.AsyncBithuman` |
88
+ | --- | --- |
89
+ | `StartFrame` | `AsyncBithuman.create(model_path=..., api_secret=...)`, then `run()` |
90
+ | `TTSAudioRawFrame` | `push_audio(pcm, sample_rate, last_chunk=False)` |
91
+ | `TTSStoppedFrame` | `flush()` |
92
+ | `InterruptionFrame` | `interrupt()` |
93
+ | `EndFrame` / `CancelFrame` / cleanup | `shutdown()` |
94
+
95
+ SDK docs: [docs.bithuman.ai/platforms/python](https://docs.bithuman.ai/platforms/python).
96
+
97
+ ## Run the example
98
+
99
+ ```bash
100
+ pip install "pipecat-bithuman[expression-2]" "pipecat-ai[daily,deepgram,openai,cartesia,silero]"
101
+ export BITHUMAN_API_SECRET=... BITHUMAN_MODEL_PATH=avatar.imx
102
+ export DAILY_ROOM_URL=... DEEPGRAM_API_KEY=... OPENAI_API_KEY=... CARTESIA_API_KEY=... CARTESIA_VOICE_ID=...
103
+ python examples/bot.py
104
+ ```
105
+
106
+ ## Tests
107
+
108
+ ```bash
109
+ pip install -e ".[dev]"
110
+ pytest # fakes only: no network, no API secret, no model file
111
+ ```
112
+
113
+ One live test talks to the real SDK. It is skipped unless you set
114
+ `PIPECAT_BITHUMAN_LIVE=1`, `BITHUMAN_API_SECRET` and `BITHUMAN_MODEL_PATH`.
115
+ It opens the avatar for a few seconds, and that time is billed.
116
+
117
+ ## Links
118
+
119
+ - bitHuman: [www.bithuman.ai](https://www.bithuman.ai)
120
+ - Docs: [docs.bithuman.ai](https://docs.bithuman.ai)
121
+ - Examples: [github.com/bithuman-product/bithuman-examples](https://github.com/bithuman-product/bithuman-examples)
122
+ - Contact: sgu@bithuman.ai
123
+
124
+ ## Licence
125
+
126
+ BSD 2-Clause, the same as Pipecat. See [LICENSE](LICENSE).
127
+ Copyright (c) 2026, bitHuman, Inc.
@@ -0,0 +1,83 @@
1
+ """Minimal Pipecat bot with a bitHuman avatar, over a Daily room.
2
+
3
+ Env: BITHUMAN_API_SECRET, BITHUMAN_MODEL_PATH, DAILY_ROOM_URL, DEEPGRAM_API_KEY,
4
+ OPENAI_API_KEY, CARTESIA_API_KEY, CARTESIA_VOICE_ID.
5
+ """
6
+
7
+ import asyncio
8
+ import os
9
+
10
+ from pipecat.audio.vad.silero import SileroVADAnalyzer
11
+ from pipecat.frames.frames import LLMRunFrame
12
+ from pipecat.pipeline.pipeline import Pipeline
13
+ from pipecat.pipeline.worker import PipelineParams, PipelineWorker
14
+ from pipecat.processors.aggregators.llm_context import LLMContext
15
+ from pipecat.processors.aggregators.llm_response_universal import (
16
+ LLMContextAggregatorPair,
17
+ LLMUserAggregatorParams,
18
+ )
19
+ from pipecat.services.cartesia.tts import CartesiaTTSService
20
+ from pipecat.services.deepgram.stt import DeepgramSTTService
21
+ from pipecat.services.openai.llm import OpenAILLMService
22
+ from pipecat.transports.daily.transport import DailyParams, DailyTransport
23
+ from pipecat.workers.runner import WorkerRunner
24
+
25
+ from pipecat_bithuman import BitHumanVideoService
26
+
27
+ SYSTEM = "You are Pip, a friendly red panda barista. Keep replies short and warm."
28
+
29
+
30
+ async def main():
31
+ transport = DailyTransport(
32
+ os.environ["DAILY_ROOM_URL"],
33
+ None,
34
+ "Pip",
35
+ DailyParams(
36
+ audio_in_enabled=True,
37
+ audio_out_enabled=True,
38
+ video_out_enabled=True,
39
+ video_out_width=1280,
40
+ video_out_height=720,
41
+ ),
42
+ )
43
+ stt = DeepgramSTTService(api_key=os.environ["DEEPGRAM_API_KEY"])
44
+ llm = OpenAILLMService(api_key=os.environ["OPENAI_API_KEY"])
45
+ tts = CartesiaTTSService(
46
+ api_key=os.environ["CARTESIA_API_KEY"],
47
+ voice_id=os.environ["CARTESIA_VOICE_ID"],
48
+ )
49
+ avatar = BitHumanVideoService() # BITHUMAN_MODEL_PATH + BITHUMAN_API_SECRET
50
+
51
+ context = LLMContext([{"role": "system", "content": SYSTEM}])
52
+ aggregators = LLMContextAggregatorPair(
53
+ context, user_params=LLMUserAggregatorParams(vad_analyzer=SileroVADAnalyzer())
54
+ )
55
+ pipeline = Pipeline(
56
+ [
57
+ transport.input(),
58
+ stt,
59
+ aggregators.user(),
60
+ llm,
61
+ tts,
62
+ avatar,
63
+ transport.output(),
64
+ aggregators.assistant(),
65
+ ]
66
+ )
67
+ worker = PipelineWorker(pipeline, params=PipelineParams(enable_metrics=True))
68
+
69
+ @transport.event_handler("on_first_participant_joined")
70
+ async def on_joined(transport, participant):
71
+ await worker.queue_frame(LLMRunFrame()) # Pip says hello first
72
+
73
+ @transport.event_handler("on_participant_left")
74
+ async def on_left(transport, participant, reason):
75
+ await worker.cancel() # close the avatar: session time stops
76
+
77
+ runner = WorkerRunner()
78
+ await runner.add_workers(worker)
79
+ await runner.run()
80
+
81
+
82
+ if __name__ == "__main__":
83
+ asyncio.run(main())
@@ -0,0 +1,73 @@
1
+ """Render a short demo video through a Pipecat pipeline with BitHumanVideoService.
2
+
3
+ The speech in a WAV file goes in as TTSAudioRawFrame chunks (as a TTS service would send
4
+ them); the service's OutputImageRawFrame pictures and the 16 kHz speech that goes with
5
+ them come out and are written to an MP4 with ffmpeg.
6
+
7
+ BITHUMAN_API_SECRET=... BITHUMAN_MODEL_PATH=avatar.imx \
8
+ python examples/render_demo.py speech.wav demo.mp4 [--repeat 2]
9
+
10
+ Bills the avatar's active session time (see docs.bithuman.ai/pricing). Needs ffmpeg.
11
+ """
12
+ import argparse
13
+ import asyncio
14
+ import subprocess
15
+ import tempfile
16
+ import wave
17
+ from pathlib import Path
18
+
19
+ from pipecat.frames.frames import OutputImageRawFrame, TTSAudioRawFrame
20
+ from pipecat.tests.utils import SleepFrame, run_test
21
+
22
+ from pipecat_bithuman import BitHumanVideoService
23
+
24
+ CHUNK_S = 0.04 # 40 ms of speech per frame, like a streaming TTS
25
+
26
+
27
+ def speech_frames(path: Path, repeat: int) -> list[TTSAudioRawFrame]:
28
+ with wave.open(str(path)) as w:
29
+ if w.getnchannels() != 1 or w.getsampwidth() != 2:
30
+ raise SystemExit("speech must be 16-bit mono PCM WAV")
31
+ rate, pcm = w.getframerate(), w.readframes(w.getnframes())
32
+ step = int(rate * CHUNK_S) * 2
33
+ frames = []
34
+ for _ in range(repeat):
35
+ frames += [TTSAudioRawFrame(audio=pcm[i:i + step], sample_rate=rate, num_channels=1,
36
+ context_id="demo") for i in range(0, len(pcm), step)]
37
+ return frames
38
+
39
+
40
+ async def render(speech: Path, out: Path, repeat: int) -> None:
41
+ sent = speech_frames(speech, repeat)
42
+ seconds = sum(len(f.audio) / 2 / f.sample_rate for f in sent)
43
+ down, _ = await run_test(BitHumanVideoService(),
44
+ frames_to_send=sent + [SleepFrame(sleep=3.0)], start_timeout=60.0)
45
+ images = [f for f in down if isinstance(f, OutputImageRawFrame)]
46
+ audio = b"".join(f.audio for f in down if isinstance(f, TTSAudioRawFrame))
47
+ if not images:
48
+ raise SystemExit("no avatar frames came out")
49
+ w, h = images[0].size
50
+ fps = round(len(images) / max(seconds, 0.1))
51
+ with tempfile.TemporaryDirectory() as tmp:
52
+ wav = Path(tmp) / "speech.wav"
53
+ with wave.open(str(wav), "wb") as o:
54
+ o.setnchannels(1), o.setsampwidth(2), o.setframerate(16000), o.writeframes(audio)
55
+ cmd = ["ffmpeg", "-y", "-loglevel", "error", "-f", "rawvideo", "-pix_fmt", "rgb24",
56
+ "-s", f"{w}x{h}", "-r", str(fps), "-i", "-", "-i", str(wav),
57
+ "-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", "-shortest", str(out)]
58
+ proc = subprocess.Popen(cmd, stdin=subprocess.PIPE)
59
+ for f in images:
60
+ proc.stdin.write(f.image)
61
+ proc.stdin.close()
62
+ if proc.wait() != 0:
63
+ raise SystemExit("ffmpeg failed")
64
+ print(f"{out}: {len(images)} frames at {fps} fps, {w}x{h}, {seconds:.1f} s of speech")
65
+
66
+
67
+ if __name__ == "__main__":
68
+ ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
69
+ ap.add_argument("speech", type=Path)
70
+ ap.add_argument("out", type=Path)
71
+ ap.add_argument("--repeat", type=int, default=1)
72
+ a = ap.parse_args()
73
+ asyncio.run(render(a.speech, a.out, a.repeat))
@@ -0,0 +1,84 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.24"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "pipecat-bithuman"
7
+ version = "0.1.0"
8
+ description = "bitHuman real-time avatar video service for Pipecat: TTS audio in, lip-synced avatar video and audio out."
9
+ readme = "README.md"
10
+ license = "BSD-2-Clause"
11
+ license-files = ["LICENSE"]
12
+ requires-python = ">=3.11"
13
+ authors = [{ name = "bitHuman, Inc." }]
14
+ keywords = ["pipecat", "bithuman", "avatar", "video", "voice-agent", "lip-sync", "real-time"]
15
+ classifiers = [
16
+ "Development Status :: 4 - Beta",
17
+ "Intended Audience :: Developers",
18
+ "Operating System :: MacOS",
19
+ "Operating System :: POSIX :: Linux",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3 :: Only",
22
+ "Programming Language :: Python :: 3.11",
23
+ "Programming Language :: Python :: 3.12",
24
+ "Programming Language :: Python :: 3.13",
25
+ "Programming Language :: Python :: 3.14",
26
+ "Topic :: Multimedia :: Video",
27
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
28
+ ]
29
+ dependencies = [
30
+ "pipecat-ai>=1.12.0",
31
+ "bithuman>=2.11.18,<3",
32
+ "numpy>=1.26",
33
+ ]
34
+
35
+ [project.optional-dependencies]
36
+ # Expression 2 avatars (any character from one portrait) need the SDK extra.
37
+ expression-2 = ["bithuman[expression-2]>=2.11.18,<3"]
38
+ dev = [
39
+ "pytest>=8",
40
+ "pytest-asyncio>=0.23",
41
+ "ruff>=0.6",
42
+ ]
43
+
44
+ [project.urls]
45
+ Homepage = "https://www.bithuman.ai"
46
+ Documentation = "https://docs.bithuman.ai/platforms/python"
47
+ Source = "https://github.com/bithuman-product/pipecat-bithuman"
48
+ Issues = "https://github.com/bithuman-product/pipecat-bithuman/issues"
49
+ Changelog = "https://github.com/bithuman-product/pipecat-bithuman/blob/main/CHANGELOG.md"
50
+
51
+ [tool.hatch.build.targets.wheel]
52
+ packages = ["src/pipecat_bithuman"]
53
+
54
+ [tool.hatch.build.targets.sdist]
55
+ include = [
56
+ "src/pipecat_bithuman",
57
+ "tests",
58
+ "examples",
59
+ "README.md",
60
+ "CHANGELOG.md",
61
+ "LICENSE",
62
+ ]
63
+
64
+ [tool.pytest.ini_options]
65
+ asyncio_mode = "auto"
66
+ testpaths = ["tests"]
67
+ markers = [
68
+ "live: talks to the real bitHuman SDK with a real API secret (bills session time; opt-in only)",
69
+ ]
70
+
71
+ [tool.ruff]
72
+ line-length = 100
73
+ target-version = "py311"
74
+
75
+ [tool.ruff.lint]
76
+ select = ["E", "F", "I", "D", "UP", "B"]
77
+ ignore = ["D105", "D107"]
78
+
79
+ [tool.ruff.lint.pydocstyle]
80
+ convention = "google"
81
+
82
+ [tool.ruff.lint.per-file-ignores]
83
+ "tests/*" = ["D"]
84
+ "examples/*" = ["D"]
@@ -0,0 +1,41 @@
1
+ #
2
+ # Copyright (c) 2026, bitHuman, Inc.
3
+ #
4
+ # SPDX-License-Identifier: BSD-2-Clause
5
+ #
6
+
7
+ """bitHuman real-time avatar video service for Pipecat.
8
+
9
+ TTS audio goes in; lip-synced avatar video and the matching audio come out.
10
+ See ``BitHumanVideoService``.
11
+ """
12
+
13
+ from importlib.metadata import PackageNotFoundError, version
14
+
15
+ from .runtime import (
16
+ API_SECRET_ENV,
17
+ MODEL_PATH_ENV,
18
+ BitHumanRuntime,
19
+ RuntimeFactory,
20
+ resolve_model_path,
21
+ sdk_runtime_factory,
22
+ )
23
+ from .video import BitHumanServiceError, BitHumanVideoService, BitHumanVideoSettings
24
+
25
+ try:
26
+ __version__ = version("pipecat-bithuman")
27
+ except PackageNotFoundError: # running from a source tree without install
28
+ __version__ = "0.0.0"
29
+
30
+ __all__ = [
31
+ "API_SECRET_ENV",
32
+ "MODEL_PATH_ENV",
33
+ "BitHumanRuntime",
34
+ "BitHumanServiceError",
35
+ "BitHumanVideoService",
36
+ "BitHumanVideoSettings",
37
+ "RuntimeFactory",
38
+ "resolve_model_path",
39
+ "sdk_runtime_factory",
40
+ "__version__",
41
+ ]