pipecat-bithuman 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pipecat_bithuman-0.1.0/.gitignore +8 -0
- pipecat_bithuman-0.1.0/CHANGELOG.md +24 -0
- pipecat_bithuman-0.1.0/LICENSE +24 -0
- pipecat_bithuman-0.1.0/PKG-INFO +164 -0
- pipecat_bithuman-0.1.0/README.md +127 -0
- pipecat_bithuman-0.1.0/examples/bot.py +83 -0
- pipecat_bithuman-0.1.0/examples/render_demo.py +73 -0
- pipecat_bithuman-0.1.0/pyproject.toml +84 -0
- pipecat_bithuman-0.1.0/src/pipecat_bithuman/__init__.py +41 -0
- pipecat_bithuman-0.1.0/src/pipecat_bithuman/runtime.py +120 -0
- pipecat_bithuman-0.1.0/src/pipecat_bithuman/video.py +628 -0
- pipecat_bithuman-0.1.0/tests/__init__.py +0 -0
- pipecat_bithuman-0.1.0/tests/fakes.py +98 -0
- pipecat_bithuman-0.1.0/tests/test_errors.py +76 -0
- pipecat_bithuman-0.1.0/tests/test_flow.py +67 -0
- pipecat_bithuman-0.1.0/tests/test_interruption_lifecycle.py +75 -0
- pipecat_bithuman-0.1.0/tests/test_live_and_logging.py +43 -0
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are listed here.
|
|
4
|
+
The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
5
|
+
and the project uses [Semantic Versioning](https://semver.org/).
|
|
6
|
+
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [0.1.0] - 2026-09-30
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- `BitHumanVideoService`: TTS audio in; lip-synced avatar video (`OutputImageRawFrame`, RGB)
|
|
14
|
+
and paired 16 kHz audio (`TTSAudioRawFrame`) out, rendered in-process by the
|
|
15
|
+
`bithuman` Python SDK.
|
|
16
|
+
- `TTSStoppedFrame` is held until the avatar has finished speaking the reply.
|
|
17
|
+
- Barge-in: `InterruptionFrame` drops the reply in flight and returns the avatar to idle.
|
|
18
|
+
- Lifecycle: the avatar opens on `StartFrame`; `EndFrame` drains queued speech, then closes;
|
|
19
|
+
`CancelFrame` and cleanup close at once. Teardown is idempotent.
|
|
20
|
+
- Errors: one `ErrorFrame` per failure with a Pipecat error category; the API secret is
|
|
21
|
+
scrubbed from error text and never logged; TTS audio passes through by default.
|
|
22
|
+
- `BitHumanRuntime` protocol and `runtime_factory` for tests and SDK wrappers.
|
|
23
|
+
- Minimal Daily example (`examples/bot.py`) and a fake-runtime test suite.
|
|
24
|
+
- Tested with Pipecat v1.12.0 and `bithuman` 2.11.18.
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
BSD 2-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, bitHuman, Inc.
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
16
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
17
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
18
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
19
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
20
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
21
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
22
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
23
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
24
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: pipecat-bithuman
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: bitHuman real-time avatar video service for Pipecat: TTS audio in, lip-synced avatar video and audio out.
|
|
5
|
+
Project-URL: Homepage, https://www.bithuman.ai
|
|
6
|
+
Project-URL: Documentation, https://docs.bithuman.ai/platforms/python
|
|
7
|
+
Project-URL: Source, https://github.com/bithuman-product/pipecat-bithuman
|
|
8
|
+
Project-URL: Issues, https://github.com/bithuman-product/pipecat-bithuman/issues
|
|
9
|
+
Project-URL: Changelog, https://github.com/bithuman-product/pipecat-bithuman/blob/main/CHANGELOG.md
|
|
10
|
+
Author: bitHuman, Inc.
|
|
11
|
+
License-Expression: BSD-2-Clause
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Keywords: avatar,bithuman,lip-sync,pipecat,real-time,video,voice-agent
|
|
14
|
+
Classifier: Development Status :: 4 - Beta
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Operating System :: MacOS
|
|
17
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
24
|
+
Classifier: Topic :: Multimedia :: Video
|
|
25
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
26
|
+
Requires-Python: >=3.11
|
|
27
|
+
Requires-Dist: bithuman<3,>=2.11.18
|
|
28
|
+
Requires-Dist: numpy>=1.26
|
|
29
|
+
Requires-Dist: pipecat-ai>=1.12.0
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
32
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
33
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
34
|
+
Provides-Extra: expression-2
|
|
35
|
+
Requires-Dist: bithuman[expression-2]<3,>=2.11.18; extra == 'expression-2'
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# pipecat-bithuman
|
|
39
|
+
|
|
40
|
+
A [bitHuman](https://www.bithuman.ai) avatar for your [Pipecat](https://github.com/pipecat-ai/pipecat) bot.
|
|
41
|
+
Your TTS audio goes in. A lip-synced avatar video, plus the audio that goes with it, comes out.
|
|
42
|
+
|
|
43
|
+
The avatar renders in your own process, on your own machine, through the
|
|
44
|
+
`bithuman` Python SDK. Both models run live on a standard Linux PC with no GPU.
|
|
45
|
+
Expression 2 animates a character from one portrait. Essence 2 is for photoreal people.
|
|
46
|
+
|
|
47
|
+
Maintained by bitHuman, Inc. (community integration, not maintained by the Pipecat team).
|
|
48
|
+
|
|
49
|
+
**Tested with Pipecat v1.12.0**, `bithuman` 2.11.18, Python 3.11 to 3.14.
|
|
50
|
+
|
|
51
|
+
**Demo (30 s):** [docs/demo.mp4](docs/demo.mp4), rendered through a Pipecat pipeline by
|
|
52
|
+
[`examples/render_demo.py`](examples/render_demo.py) with the Expression 2 sample avatar Wise Pup.
|
|
53
|
+
|
|
54
|
+
## Install
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
pip install pipecat-bithuman
|
|
58
|
+
# Expression 2 avatars (any character from one portrait):
|
|
59
|
+
pip install "pipecat-bithuman[expression-2]"
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
## Environment variables
|
|
63
|
+
|
|
64
|
+
| Variable | Needed | What it is |
|
|
65
|
+
| --- | --- | --- |
|
|
66
|
+
| `BITHUMAN_API_SECRET` | yes | Your bitHuman API secret. Get one at [www.bithuman.ai](https://www.bithuman.ai). The SDK reads it. This package never logs it. |
|
|
67
|
+
| `BITHUMAN_MODEL_PATH` | unless you pass `model_path` | Path to the avatar's `.imx` model file. |
|
|
68
|
+
|
|
69
|
+
Session time is metered while the avatar is open (talking or idle).
|
|
70
|
+
It closes on `EndFrame`, `CancelFrame` or cleanup.
|
|
71
|
+
|
|
72
|
+
Rendering bills active session time: 2 credits a minute on your own machine. From 2026-10-12, SDK use requires the Creator plan or higher. See docs.bithuman.ai/pricing.
|
|
73
|
+
|
|
74
|
+
## Usage
|
|
75
|
+
|
|
76
|
+
Put `BitHumanVideoService` after the TTS service and before `transport.output()`:
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from pipecat_bithuman import BitHumanVideoService
|
|
80
|
+
|
|
81
|
+
avatar = BitHumanVideoService(model_path="avatar.imx") # secret from BITHUMAN_API_SECRET
|
|
82
|
+
|
|
83
|
+
pipeline = Pipeline([
|
|
84
|
+
transport.input(), stt, user_aggregator, llm, tts,
|
|
85
|
+
avatar, # TTS audio -> avatar video + audio
|
|
86
|
+
transport.output(),
|
|
87
|
+
assistant_aggregator,
|
|
88
|
+
])
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Turn on video out in the transport (`video_out_enabled=True`). The service logs the
|
|
92
|
+
frame size on the first frame; set `video_out_width` / `video_out_height` to match.
|
|
93
|
+
|
|
94
|
+
### What the service does with frames
|
|
95
|
+
|
|
96
|
+
| In | Out |
|
|
97
|
+
| --- | --- |
|
|
98
|
+
| `TTSAudioRawFrame` | Sent to the avatar. Not forwarded as is. |
|
|
99
|
+
| (avatar frame) | `OutputImageRawFrame` (RGB), while talking and while idle. |
|
|
100
|
+
| (avatar audio) | `TTSAudioRawFrame` (16 kHz mono), paired with each picture. |
|
|
101
|
+
| `TTSStoppedFrame` | Held until the avatar has finished speaking the reply. |
|
|
102
|
+
| `InterruptionFrame` | The avatar drops the reply in flight and goes back to idle. |
|
|
103
|
+
| anything else | Passed on unchanged. |
|
|
104
|
+
|
|
105
|
+
If the avatar cannot start, or fails mid-session, the service pushes one
|
|
106
|
+
`ErrorFrame` upstream. By default TTS audio then passes through unchanged, so the bot
|
|
107
|
+
keeps talking without video (`audio_passthrough_on_error=False` turns this off).
|
|
108
|
+
Error text is scrubbed of the API secret.
|
|
109
|
+
|
|
110
|
+
### Options
|
|
111
|
+
|
|
112
|
+
| Argument | Default | Meaning |
|
|
113
|
+
| --- | --- | --- |
|
|
114
|
+
| `model_path` | `BITHUMAN_MODEL_PATH` | The `.imx` avatar model. |
|
|
115
|
+
| `api_secret` | `BITHUMAN_API_SECRET` | The API secret. |
|
|
116
|
+
| `sync_video_to_audio` | `True` | Sets `sync_with_audio` on each image. |
|
|
117
|
+
| `audio_passthrough_on_error` | `True` | Keep the voice if the avatar fails. |
|
|
118
|
+
| `stop_frame_timeout_s` | `2.0` | Release a held `TTSStoppedFrame` after this much quiet. |
|
|
119
|
+
| `end_drain_timeout_s` | `30.0` | Longest wait on `EndFrame` for queued speech. |
|
|
120
|
+
| `runtime_factory` | SDK | Advanced: your own `BitHumanRuntime` (tests, wrappers). |
|
|
121
|
+
|
|
122
|
+
## How it maps to the bitHuman Python SDK
|
|
123
|
+
|
|
124
|
+
| Pipecat | `bithuman.AsyncBithuman` |
|
|
125
|
+
| --- | --- |
|
|
126
|
+
| `StartFrame` | `AsyncBithuman.create(model_path=..., api_secret=...)`, then `run()` |
|
|
127
|
+
| `TTSAudioRawFrame` | `push_audio(pcm, sample_rate, last_chunk=False)` |
|
|
128
|
+
| `TTSStoppedFrame` | `flush()` |
|
|
129
|
+
| `InterruptionFrame` | `interrupt()` |
|
|
130
|
+
| `EndFrame` / `CancelFrame` / cleanup | `shutdown()` |
|
|
131
|
+
|
|
132
|
+
SDK docs: [docs.bithuman.ai/platforms/python](https://docs.bithuman.ai/platforms/python).
|
|
133
|
+
|
|
134
|
+
## Run the example
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
pip install "pipecat-bithuman[expression-2]" "pipecat-ai[daily,deepgram,openai,cartesia,silero]"
|
|
138
|
+
export BITHUMAN_API_SECRET=... BITHUMAN_MODEL_PATH=avatar.imx
|
|
139
|
+
export DAILY_ROOM_URL=... DEEPGRAM_API_KEY=... OPENAI_API_KEY=... CARTESIA_API_KEY=... CARTESIA_VOICE_ID=...
|
|
140
|
+
python examples/bot.py
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
## Tests
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
pip install -e ".[dev]"
|
|
147
|
+
pytest # fakes only: no network, no API secret, no model file
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
One live test talks to the real SDK. It is skipped unless you set
|
|
151
|
+
`PIPECAT_BITHUMAN_LIVE=1`, `BITHUMAN_API_SECRET` and `BITHUMAN_MODEL_PATH`.
|
|
152
|
+
It opens the avatar for a few seconds, and that time is billed.
|
|
153
|
+
|
|
154
|
+
## Links
|
|
155
|
+
|
|
156
|
+
- bitHuman: [www.bithuman.ai](https://www.bithuman.ai)
|
|
157
|
+
- Docs: [docs.bithuman.ai](https://docs.bithuman.ai)
|
|
158
|
+
- Examples: [github.com/bithuman-product/bithuman-examples](https://github.com/bithuman-product/bithuman-examples)
|
|
159
|
+
- Contact: sgu@bithuman.ai
|
|
160
|
+
|
|
161
|
+
## Licence
|
|
162
|
+
|
|
163
|
+
BSD 2-Clause, the same as Pipecat. See [LICENSE](LICENSE).
|
|
164
|
+
Copyright (c) 2026, bitHuman, Inc.
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
# pipecat-bithuman
|
|
2
|
+
|
|
3
|
+
A [bitHuman](https://www.bithuman.ai) avatar for your [Pipecat](https://github.com/pipecat-ai/pipecat) bot.
|
|
4
|
+
Your TTS audio goes in. A lip-synced avatar video, plus the audio that goes with it, comes out.
|
|
5
|
+
|
|
6
|
+
The avatar renders in your own process, on your own machine, through the
|
|
7
|
+
`bithuman` Python SDK. Both models run live on a standard Linux PC with no GPU.
|
|
8
|
+
Expression 2 animates a character from one portrait. Essence 2 is for photoreal people.
|
|
9
|
+
|
|
10
|
+
Maintained by bitHuman, Inc. (community integration, not maintained by the Pipecat team).
|
|
11
|
+
|
|
12
|
+
**Tested with Pipecat v1.12.0**, `bithuman` 2.11.18, Python 3.11 to 3.14.
|
|
13
|
+
|
|
14
|
+
**Demo (30 s):** [docs/demo.mp4](docs/demo.mp4), rendered through a Pipecat pipeline by
|
|
15
|
+
[`examples/render_demo.py`](examples/render_demo.py) with the Expression 2 sample avatar Wise Pup.
|
|
16
|
+
|
|
17
|
+
## Install
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install pipecat-bithuman
|
|
21
|
+
# Expression 2 avatars (any character from one portrait):
|
|
22
|
+
pip install "pipecat-bithuman[expression-2]"
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## Environment variables
|
|
26
|
+
|
|
27
|
+
| Variable | Needed | What it is |
|
|
28
|
+
| --- | --- | --- |
|
|
29
|
+
| `BITHUMAN_API_SECRET` | yes | Your bitHuman API secret. Get one at [www.bithuman.ai](https://www.bithuman.ai). The SDK reads it. This package never logs it. |
|
|
30
|
+
| `BITHUMAN_MODEL_PATH` | unless you pass `model_path` | Path to the avatar's `.imx` model file. |
|
|
31
|
+
|
|
32
|
+
Session time is metered while the avatar is open (talking or idle).
|
|
33
|
+
It closes on `EndFrame`, `CancelFrame` or cleanup.
|
|
34
|
+
|
|
35
|
+
Rendering bills active session time: 2 credits a minute on your own machine. From 2026-10-12, SDK use requires the Creator plan or higher. See docs.bithuman.ai/pricing.
|
|
36
|
+
|
|
37
|
+
## Usage
|
|
38
|
+
|
|
39
|
+
Put `BitHumanVideoService` after the TTS service and before `transport.output()`:
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
from pipecat_bithuman import BitHumanVideoService
|
|
43
|
+
|
|
44
|
+
avatar = BitHumanVideoService(model_path="avatar.imx") # secret from BITHUMAN_API_SECRET
|
|
45
|
+
|
|
46
|
+
pipeline = Pipeline([
|
|
47
|
+
transport.input(), stt, user_aggregator, llm, tts,
|
|
48
|
+
avatar, # TTS audio -> avatar video + audio
|
|
49
|
+
transport.output(),
|
|
50
|
+
assistant_aggregator,
|
|
51
|
+
])
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Turn on video out in the transport (`video_out_enabled=True`). The service logs the
|
|
55
|
+
frame size on the first frame; set `video_out_width` / `video_out_height` to match.
|
|
56
|
+
|
|
57
|
+
### What the service does with frames
|
|
58
|
+
|
|
59
|
+
| In | Out |
|
|
60
|
+
| --- | --- |
|
|
61
|
+
| `TTSAudioRawFrame` | Sent to the avatar. Not forwarded as is. |
|
|
62
|
+
| (avatar frame) | `OutputImageRawFrame` (RGB), while talking and while idle. |
|
|
63
|
+
| (avatar audio) | `TTSAudioRawFrame` (16 kHz mono), paired with each picture. |
|
|
64
|
+
| `TTSStoppedFrame` | Held until the avatar has finished speaking the reply. |
|
|
65
|
+
| `InterruptionFrame` | The avatar drops the reply in flight and goes back to idle. |
|
|
66
|
+
| anything else | Passed on unchanged. |
|
|
67
|
+
|
|
68
|
+
If the avatar cannot start, or fails mid-session, the service pushes one
|
|
69
|
+
`ErrorFrame` upstream. By default TTS audio then passes through unchanged, so the bot
|
|
70
|
+
keeps talking without video (`audio_passthrough_on_error=False` turns this off).
|
|
71
|
+
Error text is scrubbed of the API secret.
|
|
72
|
+
|
|
73
|
+
### Options
|
|
74
|
+
|
|
75
|
+
| Argument | Default | Meaning |
|
|
76
|
+
| --- | --- | --- |
|
|
77
|
+
| `model_path` | `BITHUMAN_MODEL_PATH` | The `.imx` avatar model. |
|
|
78
|
+
| `api_secret` | `BITHUMAN_API_SECRET` | The API secret. |
|
|
79
|
+
| `sync_video_to_audio` | `True` | Sets `sync_with_audio` on each image. |
|
|
80
|
+
| `audio_passthrough_on_error` | `True` | Keep the voice if the avatar fails. |
|
|
81
|
+
| `stop_frame_timeout_s` | `2.0` | Release a held `TTSStoppedFrame` after this much quiet. |
|
|
82
|
+
| `end_drain_timeout_s` | `30.0` | Longest wait on `EndFrame` for queued speech. |
|
|
83
|
+
| `runtime_factory` | SDK | Advanced: your own `BitHumanRuntime` (tests, wrappers). |
|
|
84
|
+
|
|
85
|
+
## How it maps to the bitHuman Python SDK
|
|
86
|
+
|
|
87
|
+
| Pipecat | `bithuman.AsyncBithuman` |
|
|
88
|
+
| --- | --- |
|
|
89
|
+
| `StartFrame` | `AsyncBithuman.create(model_path=..., api_secret=...)`, then `run()` |
|
|
90
|
+
| `TTSAudioRawFrame` | `push_audio(pcm, sample_rate, last_chunk=False)` |
|
|
91
|
+
| `TTSStoppedFrame` | `flush()` |
|
|
92
|
+
| `InterruptionFrame` | `interrupt()` |
|
|
93
|
+
| `EndFrame` / `CancelFrame` / cleanup | `shutdown()` |
|
|
94
|
+
|
|
95
|
+
SDK docs: [docs.bithuman.ai/platforms/python](https://docs.bithuman.ai/platforms/python).
|
|
96
|
+
|
|
97
|
+
## Run the example
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
pip install "pipecat-bithuman[expression-2]" "pipecat-ai[daily,deepgram,openai,cartesia,silero]"
|
|
101
|
+
export BITHUMAN_API_SECRET=... BITHUMAN_MODEL_PATH=avatar.imx
|
|
102
|
+
export DAILY_ROOM_URL=... DEEPGRAM_API_KEY=... OPENAI_API_KEY=... CARTESIA_API_KEY=... CARTESIA_VOICE_ID=...
|
|
103
|
+
python examples/bot.py
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
## Tests
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
pip install -e ".[dev]"
|
|
110
|
+
pytest # fakes only: no network, no API secret, no model file
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
One live test talks to the real SDK. It is skipped unless you set
|
|
114
|
+
`PIPECAT_BITHUMAN_LIVE=1`, `BITHUMAN_API_SECRET` and `BITHUMAN_MODEL_PATH`.
|
|
115
|
+
It opens the avatar for a few seconds, and that time is billed.
|
|
116
|
+
|
|
117
|
+
## Links
|
|
118
|
+
|
|
119
|
+
- bitHuman: [www.bithuman.ai](https://www.bithuman.ai)
|
|
120
|
+
- Docs: [docs.bithuman.ai](https://docs.bithuman.ai)
|
|
121
|
+
- Examples: [github.com/bithuman-product/bithuman-examples](https://github.com/bithuman-product/bithuman-examples)
|
|
122
|
+
- Contact: sgu@bithuman.ai
|
|
123
|
+
|
|
124
|
+
## Licence
|
|
125
|
+
|
|
126
|
+
BSD 2-Clause, the same as Pipecat. See [LICENSE](LICENSE).
|
|
127
|
+
Copyright (c) 2026, bitHuman, Inc.
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Minimal Pipecat bot with a bitHuman avatar, over a Daily room.
|
|
2
|
+
|
|
3
|
+
Env: BITHUMAN_API_SECRET, BITHUMAN_MODEL_PATH, DAILY_ROOM_URL, DEEPGRAM_API_KEY,
|
|
4
|
+
OPENAI_API_KEY, CARTESIA_API_KEY, CARTESIA_VOICE_ID.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import asyncio
|
|
8
|
+
import os
|
|
9
|
+
|
|
10
|
+
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
|
11
|
+
from pipecat.frames.frames import LLMRunFrame
|
|
12
|
+
from pipecat.pipeline.pipeline import Pipeline
|
|
13
|
+
from pipecat.pipeline.worker import PipelineParams, PipelineWorker
|
|
14
|
+
from pipecat.processors.aggregators.llm_context import LLMContext
|
|
15
|
+
from pipecat.processors.aggregators.llm_response_universal import (
|
|
16
|
+
LLMContextAggregatorPair,
|
|
17
|
+
LLMUserAggregatorParams,
|
|
18
|
+
)
|
|
19
|
+
from pipecat.services.cartesia.tts import CartesiaTTSService
|
|
20
|
+
from pipecat.services.deepgram.stt import DeepgramSTTService
|
|
21
|
+
from pipecat.services.openai.llm import OpenAILLMService
|
|
22
|
+
from pipecat.transports.daily.transport import DailyParams, DailyTransport
|
|
23
|
+
from pipecat.workers.runner import WorkerRunner
|
|
24
|
+
|
|
25
|
+
from pipecat_bithuman import BitHumanVideoService
|
|
26
|
+
|
|
27
|
+
SYSTEM = "You are Pip, a friendly red panda barista. Keep replies short and warm."
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
async def main():
|
|
31
|
+
transport = DailyTransport(
|
|
32
|
+
os.environ["DAILY_ROOM_URL"],
|
|
33
|
+
None,
|
|
34
|
+
"Pip",
|
|
35
|
+
DailyParams(
|
|
36
|
+
audio_in_enabled=True,
|
|
37
|
+
audio_out_enabled=True,
|
|
38
|
+
video_out_enabled=True,
|
|
39
|
+
video_out_width=1280,
|
|
40
|
+
video_out_height=720,
|
|
41
|
+
),
|
|
42
|
+
)
|
|
43
|
+
stt = DeepgramSTTService(api_key=os.environ["DEEPGRAM_API_KEY"])
|
|
44
|
+
llm = OpenAILLMService(api_key=os.environ["OPENAI_API_KEY"])
|
|
45
|
+
tts = CartesiaTTSService(
|
|
46
|
+
api_key=os.environ["CARTESIA_API_KEY"],
|
|
47
|
+
voice_id=os.environ["CARTESIA_VOICE_ID"],
|
|
48
|
+
)
|
|
49
|
+
avatar = BitHumanVideoService() # BITHUMAN_MODEL_PATH + BITHUMAN_API_SECRET
|
|
50
|
+
|
|
51
|
+
context = LLMContext([{"role": "system", "content": SYSTEM}])
|
|
52
|
+
aggregators = LLMContextAggregatorPair(
|
|
53
|
+
context, user_params=LLMUserAggregatorParams(vad_analyzer=SileroVADAnalyzer())
|
|
54
|
+
)
|
|
55
|
+
pipeline = Pipeline(
|
|
56
|
+
[
|
|
57
|
+
transport.input(),
|
|
58
|
+
stt,
|
|
59
|
+
aggregators.user(),
|
|
60
|
+
llm,
|
|
61
|
+
tts,
|
|
62
|
+
avatar,
|
|
63
|
+
transport.output(),
|
|
64
|
+
aggregators.assistant(),
|
|
65
|
+
]
|
|
66
|
+
)
|
|
67
|
+
worker = PipelineWorker(pipeline, params=PipelineParams(enable_metrics=True))
|
|
68
|
+
|
|
69
|
+
@transport.event_handler("on_first_participant_joined")
|
|
70
|
+
async def on_joined(transport, participant):
|
|
71
|
+
await worker.queue_frame(LLMRunFrame()) # Pip says hello first
|
|
72
|
+
|
|
73
|
+
@transport.event_handler("on_participant_left")
|
|
74
|
+
async def on_left(transport, participant, reason):
|
|
75
|
+
await worker.cancel() # close the avatar: session time stops
|
|
76
|
+
|
|
77
|
+
runner = WorkerRunner()
|
|
78
|
+
await runner.add_workers(worker)
|
|
79
|
+
await runner.run()
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
if __name__ == "__main__":
|
|
83
|
+
asyncio.run(main())
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Render a short demo video through a Pipecat pipeline with BitHumanVideoService.
|
|
2
|
+
|
|
3
|
+
The speech in a WAV file goes in as TTSAudioRawFrame chunks (as a TTS service would send
|
|
4
|
+
them); the service's OutputImageRawFrame pictures and the 16 kHz speech that goes with
|
|
5
|
+
them come out and are written to an MP4 with ffmpeg.
|
|
6
|
+
|
|
7
|
+
BITHUMAN_API_SECRET=... BITHUMAN_MODEL_PATH=avatar.imx \
|
|
8
|
+
python examples/render_demo.py speech.wav demo.mp4 [--repeat 2]
|
|
9
|
+
|
|
10
|
+
Bills the avatar's active session time (see docs.bithuman.ai/pricing). Needs ffmpeg.
|
|
11
|
+
"""
|
|
12
|
+
import argparse
|
|
13
|
+
import asyncio
|
|
14
|
+
import subprocess
|
|
15
|
+
import tempfile
|
|
16
|
+
import wave
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
from pipecat.frames.frames import OutputImageRawFrame, TTSAudioRawFrame
|
|
20
|
+
from pipecat.tests.utils import SleepFrame, run_test
|
|
21
|
+
|
|
22
|
+
from pipecat_bithuman import BitHumanVideoService
|
|
23
|
+
|
|
24
|
+
CHUNK_S = 0.04 # 40 ms of speech per frame, like a streaming TTS
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def speech_frames(path: Path, repeat: int) -> list[TTSAudioRawFrame]:
|
|
28
|
+
with wave.open(str(path)) as w:
|
|
29
|
+
if w.getnchannels() != 1 or w.getsampwidth() != 2:
|
|
30
|
+
raise SystemExit("speech must be 16-bit mono PCM WAV")
|
|
31
|
+
rate, pcm = w.getframerate(), w.readframes(w.getnframes())
|
|
32
|
+
step = int(rate * CHUNK_S) * 2
|
|
33
|
+
frames = []
|
|
34
|
+
for _ in range(repeat):
|
|
35
|
+
frames += [TTSAudioRawFrame(audio=pcm[i:i + step], sample_rate=rate, num_channels=1,
|
|
36
|
+
context_id="demo") for i in range(0, len(pcm), step)]
|
|
37
|
+
return frames
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
async def render(speech: Path, out: Path, repeat: int) -> None:
|
|
41
|
+
sent = speech_frames(speech, repeat)
|
|
42
|
+
seconds = sum(len(f.audio) / 2 / f.sample_rate for f in sent)
|
|
43
|
+
down, _ = await run_test(BitHumanVideoService(),
|
|
44
|
+
frames_to_send=sent + [SleepFrame(sleep=3.0)], start_timeout=60.0)
|
|
45
|
+
images = [f for f in down if isinstance(f, OutputImageRawFrame)]
|
|
46
|
+
audio = b"".join(f.audio for f in down if isinstance(f, TTSAudioRawFrame))
|
|
47
|
+
if not images:
|
|
48
|
+
raise SystemExit("no avatar frames came out")
|
|
49
|
+
w, h = images[0].size
|
|
50
|
+
fps = round(len(images) / max(seconds, 0.1))
|
|
51
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
52
|
+
wav = Path(tmp) / "speech.wav"
|
|
53
|
+
with wave.open(str(wav), "wb") as o:
|
|
54
|
+
o.setnchannels(1), o.setsampwidth(2), o.setframerate(16000), o.writeframes(audio)
|
|
55
|
+
cmd = ["ffmpeg", "-y", "-loglevel", "error", "-f", "rawvideo", "-pix_fmt", "rgb24",
|
|
56
|
+
"-s", f"{w}x{h}", "-r", str(fps), "-i", "-", "-i", str(wav),
|
|
57
|
+
"-c:v", "libx264", "-pix_fmt", "yuv420p", "-c:a", "aac", "-shortest", str(out)]
|
|
58
|
+
proc = subprocess.Popen(cmd, stdin=subprocess.PIPE)
|
|
59
|
+
for f in images:
|
|
60
|
+
proc.stdin.write(f.image)
|
|
61
|
+
proc.stdin.close()
|
|
62
|
+
if proc.wait() != 0:
|
|
63
|
+
raise SystemExit("ffmpeg failed")
|
|
64
|
+
print(f"{out}: {len(images)} frames at {fps} fps, {w}x{h}, {seconds:.1f} s of speech")
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
if __name__ == "__main__":
|
|
68
|
+
ap = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
|
69
|
+
ap.add_argument("speech", type=Path)
|
|
70
|
+
ap.add_argument("out", type=Path)
|
|
71
|
+
ap.add_argument("--repeat", type=int, default=1)
|
|
72
|
+
a = ap.parse_args()
|
|
73
|
+
asyncio.run(render(a.speech, a.out, a.repeat))
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.24"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pipecat-bithuman"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "bitHuman real-time avatar video service for Pipecat: TTS audio in, lip-synced avatar video and audio out."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "BSD-2-Clause"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.11"
|
|
13
|
+
authors = [{ name = "bitHuman, Inc." }]
|
|
14
|
+
keywords = ["pipecat", "bithuman", "avatar", "video", "voice-agent", "lip-sync", "real-time"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 4 - Beta",
|
|
17
|
+
"Intended Audience :: Developers",
|
|
18
|
+
"Operating System :: MacOS",
|
|
19
|
+
"Operating System :: POSIX :: Linux",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
"Programming Language :: Python :: 3.13",
|
|
25
|
+
"Programming Language :: Python :: 3.14",
|
|
26
|
+
"Topic :: Multimedia :: Video",
|
|
27
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
28
|
+
]
|
|
29
|
+
dependencies = [
|
|
30
|
+
"pipecat-ai>=1.12.0",
|
|
31
|
+
"bithuman>=2.11.18,<3",
|
|
32
|
+
"numpy>=1.26",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
[project.optional-dependencies]
|
|
36
|
+
# Expression 2 avatars (any character from one portrait) need the SDK extra.
|
|
37
|
+
expression-2 = ["bithuman[expression-2]>=2.11.18,<3"]
|
|
38
|
+
dev = [
|
|
39
|
+
"pytest>=8",
|
|
40
|
+
"pytest-asyncio>=0.23",
|
|
41
|
+
"ruff>=0.6",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
[project.urls]
|
|
45
|
+
Homepage = "https://www.bithuman.ai"
|
|
46
|
+
Documentation = "https://docs.bithuman.ai/platforms/python"
|
|
47
|
+
Source = "https://github.com/bithuman-product/pipecat-bithuman"
|
|
48
|
+
Issues = "https://github.com/bithuman-product/pipecat-bithuman/issues"
|
|
49
|
+
Changelog = "https://github.com/bithuman-product/pipecat-bithuman/blob/main/CHANGELOG.md"
|
|
50
|
+
|
|
51
|
+
[tool.hatch.build.targets.wheel]
|
|
52
|
+
packages = ["src/pipecat_bithuman"]
|
|
53
|
+
|
|
54
|
+
[tool.hatch.build.targets.sdist]
|
|
55
|
+
include = [
|
|
56
|
+
"src/pipecat_bithuman",
|
|
57
|
+
"tests",
|
|
58
|
+
"examples",
|
|
59
|
+
"README.md",
|
|
60
|
+
"CHANGELOG.md",
|
|
61
|
+
"LICENSE",
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
[tool.pytest.ini_options]
|
|
65
|
+
asyncio_mode = "auto"
|
|
66
|
+
testpaths = ["tests"]
|
|
67
|
+
markers = [
|
|
68
|
+
"live: talks to the real bitHuman SDK with a real API secret (bills session time; opt-in only)",
|
|
69
|
+
]
|
|
70
|
+
|
|
71
|
+
[tool.ruff]
|
|
72
|
+
line-length = 100
|
|
73
|
+
target-version = "py311"
|
|
74
|
+
|
|
75
|
+
[tool.ruff.lint]
|
|
76
|
+
select = ["E", "F", "I", "D", "UP", "B"]
|
|
77
|
+
ignore = ["D105", "D107"]
|
|
78
|
+
|
|
79
|
+
[tool.ruff.lint.pydocstyle]
|
|
80
|
+
convention = "google"
|
|
81
|
+
|
|
82
|
+
[tool.ruff.lint.per-file-ignores]
|
|
83
|
+
"tests/*" = ["D"]
|
|
84
|
+
"examples/*" = ["D"]
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
#
|
|
2
|
+
# Copyright (c) 2026, bitHuman, Inc.
|
|
3
|
+
#
|
|
4
|
+
# SPDX-License-Identifier: BSD-2-Clause
|
|
5
|
+
#
|
|
6
|
+
|
|
7
|
+
"""bitHuman real-time avatar video service for Pipecat.
|
|
8
|
+
|
|
9
|
+
TTS audio goes in; lip-synced avatar video and the matching audio come out.
|
|
10
|
+
See ``BitHumanVideoService``.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
14
|
+
|
|
15
|
+
from .runtime import (
|
|
16
|
+
API_SECRET_ENV,
|
|
17
|
+
MODEL_PATH_ENV,
|
|
18
|
+
BitHumanRuntime,
|
|
19
|
+
RuntimeFactory,
|
|
20
|
+
resolve_model_path,
|
|
21
|
+
sdk_runtime_factory,
|
|
22
|
+
)
|
|
23
|
+
from .video import BitHumanServiceError, BitHumanVideoService, BitHumanVideoSettings
|
|
24
|
+
|
|
25
|
+
try:
|
|
26
|
+
__version__ = version("pipecat-bithuman")
|
|
27
|
+
except PackageNotFoundError: # running from a source tree without install
|
|
28
|
+
__version__ = "0.0.0"
|
|
29
|
+
|
|
30
|
+
__all__ = [
|
|
31
|
+
"API_SECRET_ENV",
|
|
32
|
+
"MODEL_PATH_ENV",
|
|
33
|
+
"BitHumanRuntime",
|
|
34
|
+
"BitHumanServiceError",
|
|
35
|
+
"BitHumanVideoService",
|
|
36
|
+
"BitHumanVideoSettings",
|
|
37
|
+
"RuntimeFactory",
|
|
38
|
+
"resolve_model_path",
|
|
39
|
+
"sdk_runtime_factory",
|
|
40
|
+
"__version__",
|
|
41
|
+
]
|