pipecat-mirai 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pipecat_mirai-0.1.0/.github/workflows/ci.yml +31 -0
- pipecat_mirai-0.1.0/.github/workflows/release.yml +49 -0
- pipecat_mirai-0.1.0/.gitignore +10 -0
- pipecat_mirai-0.1.0/CHANGELOG.md +23 -0
- pipecat_mirai-0.1.0/LICENSE +24 -0
- pipecat_mirai-0.1.0/PKG-INFO +140 -0
- pipecat_mirai-0.1.0/README.md +117 -0
- pipecat_mirai-0.1.0/benchmarks/phone-breaks/README.md +53 -0
- pipecat_mirai-0.1.0/benchmarks/phone-breaks/phone_breaks.py +315 -0
- pipecat_mirai-0.1.0/docs/pipecat-docs/README.md +12 -0
- pipecat_mirai-0.1.0/docs/pipecat-docs/mirai.mdx +115 -0
- pipecat_mirai-0.1.0/examples/foundational/01-say-hello.py +70 -0
- pipecat_mirai-0.1.0/examples/phone/twilio_bot.py +89 -0
- pipecat_mirai-0.1.0/pyproject.toml +54 -0
- pipecat_mirai-0.1.0/src/pipecat_mirai/__init__.py +21 -0
- pipecat_mirai-0.1.0/src/pipecat_mirai/pacing.py +81 -0
- pipecat_mirai-0.1.0/src/pipecat_mirai/py.typed +0 -0
- pipecat_mirai-0.1.0/src/pipecat_mirai/tts.py +241 -0
- pipecat_mirai-0.1.0/tests/test_pacing.py +76 -0
- pipecat_mirai-0.1.0/tests/test_tts.py +130 -0
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
test:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
strategy:
|
|
12
|
+
fail-fast: false
|
|
13
|
+
matrix:
|
|
14
|
+
python: ["3.11", "3.12"]
|
|
15
|
+
pipecat: ["1.8.1", "latest"]
|
|
16
|
+
steps:
|
|
17
|
+
- uses: actions/checkout@v4
|
|
18
|
+
- uses: astral-sh/setup-uv@v6
|
|
19
|
+
- name: Install
|
|
20
|
+
run: |
|
|
21
|
+
uv venv --python ${{ matrix.python }}
|
|
22
|
+
if [ "${{ matrix.pipecat }}" = "latest" ]; then PIPECAT="pipecat-ai[websocket]"; else PIPECAT="pipecat-ai[websocket]==${{ matrix.pipecat }}"; fi
|
|
23
|
+
uv pip install -e . "$PIPECAT" pytest pytest-asyncio ruff
|
|
24
|
+
- name: Lint
|
|
25
|
+
run: |
|
|
26
|
+
.venv/bin/ruff check .
|
|
27
|
+
.venv/bin/ruff format --check src tests examples benchmarks
|
|
28
|
+
- name: Test
|
|
29
|
+
run: .venv/bin/python -m pytest -q
|
|
30
|
+
- name: Examples compile
|
|
31
|
+
run: .venv/bin/python -m py_compile examples/foundational/01-say-hello.py examples/phone/twilio_bot.py benchmarks/phone-breaks/phone_breaks.py
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
name: Release
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags: ["v*"]
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
build:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
steps:
|
|
11
|
+
- uses: actions/checkout@v4
|
|
12
|
+
- uses: astral-sh/setup-uv@v6
|
|
13
|
+
- name: Check tag matches version
|
|
14
|
+
run: |
|
|
15
|
+
VERSION=$(grep -m1 '^version = ' pyproject.toml | cut -d'"' -f2)
|
|
16
|
+
test "v$VERSION" = "${GITHUB_REF_NAME}" || { echo "tag ${GITHUB_REF_NAME} != v$VERSION"; exit 1; }
|
|
17
|
+
- run: uv build
|
|
18
|
+
- uses: actions/upload-artifact@v4
|
|
19
|
+
with:
|
|
20
|
+
name: dist
|
|
21
|
+
path: dist/
|
|
22
|
+
|
|
23
|
+
publish:
|
|
24
|
+
needs: build
|
|
25
|
+
runs-on: ubuntu-latest
|
|
26
|
+
environment: pypi
|
|
27
|
+
permissions:
|
|
28
|
+
id-token: write # PyPI trusted publishing; no API token stored in GitHub
|
|
29
|
+
steps:
|
|
30
|
+
- uses: actions/download-artifact@v4
|
|
31
|
+
with:
|
|
32
|
+
name: dist
|
|
33
|
+
path: dist/
|
|
34
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
35
|
+
|
|
36
|
+
github-release:
|
|
37
|
+
needs: publish
|
|
38
|
+
runs-on: ubuntu-latest
|
|
39
|
+
permissions:
|
|
40
|
+
contents: write
|
|
41
|
+
steps:
|
|
42
|
+
- uses: actions/checkout@v4
|
|
43
|
+
- uses: actions/download-artifact@v4
|
|
44
|
+
with:
|
|
45
|
+
name: dist
|
|
46
|
+
path: dist/
|
|
47
|
+
- env:
|
|
48
|
+
GH_TOKEN: ${{ github.token }}
|
|
49
|
+
run: gh release create "$GITHUB_REF_NAME" dist/* --title "$GITHUB_REF_NAME" --notes "See CHANGELOG.md"
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented here. The format follows
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/) and the project uses
|
|
5
|
+
[Semantic Versioning](https://semver.org/).
|
|
6
|
+
|
|
7
|
+
## [0.1.0] - 2026-10-02
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- `MiraiTTSService`: streaming Mirai text-to-speech for Pipecat. Audio is
|
|
12
|
+
resampled from Mirai's 48 kHz to the pipeline's output rate per utterance, and
|
|
13
|
+
network reads that end mid-sample are handled. It reports TTFB and usage
|
|
14
|
+
metrics, supports tracing and `TTSUpdateSettingsFrame`, and closes the HTTP
|
|
15
|
+
stream on interruption.
|
|
16
|
+
- `apply_output_lead()`: lets Pipecat's websocket output transports (FastAPI,
|
|
17
|
+
websocket server and websocket client) send up to 0.4 s ahead of real time, so
|
|
18
|
+
event-loop stalls no longer break audio on phone calls.
|
|
19
|
+
- Examples: a foundational speech check and a Twilio phone bot.
|
|
20
|
+
- `benchmarks/phone-breaks`: a reproducible harness that measures breaks on phone
|
|
21
|
+
calls under event-loop stalls.
|
|
22
|
+
|
|
23
|
+
Tested with Pipecat 1.8.1 and 1.12.0.
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
BSD 2-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, Sona Labs Pvt Ltd
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
16
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
17
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
18
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
19
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
20
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
21
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
22
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
23
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
24
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: pipecat-mirai
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Mirai text-to-speech (Hindi, Hinglish, Gujarati) for Pipecat voice agents
|
|
5
|
+
Project-URL: Homepage, https://github.com/MiraiMinds/pipecat-mirai
|
|
6
|
+
Project-URL: Documentation, https://docs.miraiminds.co
|
|
7
|
+
Project-URL: Issues, https://github.com/MiraiMinds/pipecat-mirai/issues
|
|
8
|
+
Author-email: Sona Labs Pvt Ltd <sneh.mehta@miraiminds.co>
|
|
9
|
+
License-Expression: BSD-2-Clause
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: hindi,mirai,pipecat,telephony,text-to-speech,tts,voice-agent
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: BSD License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
17
|
+
Requires-Python: >=3.11
|
|
18
|
+
Requires-Dist: httpx<1,>=0.27
|
|
19
|
+
Requires-Dist: numpy>=1.26
|
|
20
|
+
Requires-Dist: pipecat-ai<2,>=1.8.1
|
|
21
|
+
Requires-Dist: soxr>=0.3
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# pipecat-mirai
|
|
25
|
+
|
|
26
|
+
[Mirai](https://miraiminds.co) text-to-speech for [Pipecat](https://github.com/pipecat-ai/pipecat)
|
|
27
|
+
voice agents: natural Hindi, Hinglish and Gujarati voices, streamed with about
|
|
28
|
+
100 ms to first audio.
|
|
29
|
+
|
|
30
|
+
- **`MiraiTTSService`**: a Pipecat TTS service for Mirai's streaming API. Audio is
|
|
31
|
+
resampled to your pipeline's rate (8 kHz for phone calls), with no voice
|
|
32
|
+
registration or sample-rate workarounds.
|
|
33
|
+
- **`apply_output_lead()`**: stops audio breaking up on phone calls when your
|
|
34
|
+
server is busy (see [Phone calls](#phone-calls)). This works with any TTS service.
|
|
35
|
+
|
|
36
|
+
## Installation
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
uv add pipecat-mirai # or: pip install pipecat-mirai
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Prerequisites
|
|
43
|
+
|
|
44
|
+
- A Mirai API key (`sk_live_…`) from the [Mirai console](https://sandbox.voice.miraiminds.co).
|
|
45
|
+
Set it as `MIRAI_API_KEY`, or pass `api_key=`.
|
|
46
|
+
- Pipecat 1.8.1 or newer and Python 3.11+.
|
|
47
|
+
|
|
48
|
+
## Usage
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from pipecat_mirai import MiraiTTSService
|
|
52
|
+
|
|
53
|
+
tts = MiraiTTSService(settings=MiraiTTSService.Settings(voice="shruti"))
|
|
54
|
+
|
|
55
|
+
pipeline = Pipeline([transport.input(), stt, user_aggregator, llm, tts, transport.output(), assistant_aggregator])
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
| Voice | |
|
|
59
|
+
|---|---|
|
|
60
|
+
| `ashu` | male |
|
|
61
|
+
| `neha` | female (default) |
|
|
62
|
+
| `shruti` | female |
|
|
63
|
+
| `sameer` | male |
|
|
64
|
+
|
|
65
|
+
Write Hindi in Devanagari and English words in Latin script, as people actually
|
|
66
|
+
type Hinglish: `आपका order कल deliver होगा।`
|
|
67
|
+
|
|
68
|
+
Change the voice mid-call with Pipecat's standard settings frame:
|
|
69
|
+
|
|
70
|
+
```python
|
|
71
|
+
await task.queue_frame(TTSUpdateSettingsFrame(delta=MiraiTTSService.Settings(voice="sameer")))
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
### Parameters
|
|
75
|
+
|
|
76
|
+
| Argument | Default | |
|
|
77
|
+
|---|---|---|
|
|
78
|
+
| `api_key` | `$MIRAI_API_KEY` | Your Mirai API key |
|
|
79
|
+
| `settings` | `voice="neha"`, `model="mira-tts"` | `MiraiTTSService.Settings(...)` |
|
|
80
|
+
| `voice`, `model` | | Shortcuts for the same settings |
|
|
81
|
+
| `sample_rate` | pipeline `audio_out_sample_rate` | Output rate; Mirai's 48 kHz audio is resampled to it |
|
|
82
|
+
| `base_url` | `https://sandbox.voice.miraiminds.co/v1` | API base URL |
|
|
83
|
+
| `http_client` | own client | An `httpx.AsyncClient` you manage |
|
|
84
|
+
|
|
85
|
+
The service reports time-to-first-byte and character usage metrics and supports
|
|
86
|
+
Pipecat tracing. Interrupting the bot closes the HTTP stream at once.
|
|
87
|
+
|
|
88
|
+
## Phone calls
|
|
89
|
+
|
|
90
|
+
Pipecat's websocket transports (`FastAPIWebsocketTransport` with Twilio, Plivo,
|
|
91
|
+
Exotel or Telnyx serializers, `WebsocketServerTransport`, `WebsocketClientTransport`)
|
|
92
|
+
send audio at exactly real time. The phone provider then never holds more than
|
|
93
|
+
about 40 ms of audio. If your server's event loop stalls for longer (a blocking
|
|
94
|
+
call, a heavy parse, many calls on one CPU), the provider runs out and the caller
|
|
95
|
+
hears the voice break up. This happens with every TTS vendor, and it gets worse
|
|
96
|
+
as you add concurrent calls.
|
|
97
|
+
|
|
98
|
+
`apply_output_lead()` lets the transport send up to 0.4 s ahead, so short stalls
|
|
99
|
+
go unnoticed:
|
|
100
|
+
|
|
101
|
+
```python
|
|
102
|
+
from pipecat_mirai import MiraiTTSService, apply_output_lead
|
|
103
|
+
|
|
104
|
+
transport = FastAPIWebsocketTransport(websocket, FastAPIWebsocketParams(
|
|
105
|
+
audio_out_enabled=True, add_wav_header=False, serializer=serializer, ...))
|
|
106
|
+
apply_output_lead(transport) # default 0.4 s; apply_output_lead(transport, 0.6) for more
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
We measured this on 8 kHz Twilio-protocol calls, with Pipecat 1.8.1, real Mirai audio
|
|
110
|
+
and 50–250 ms event-loop stalls every ~3 s per call
|
|
111
|
+
([benchmarks/phone-breaks](benchmarks/phone-breaks)):
|
|
112
|
+
|
|
113
|
+
| Concurrent calls | Stock Pipecat: speech stretched by breaks | With `apply_output_lead` |
|
|
114
|
+
|---|---|---|
|
|
115
|
+
| 3 | 6.6% (3.5 breaks per sentence) | **0.08%** |
|
|
116
|
+
| 10 | 29.1% (14.5 breaks per sentence) | **0.01%** |
|
|
117
|
+
|
|
118
|
+
When the caller interrupts, Pipecat still tells the provider to clear its queue, so
|
|
119
|
+
at most the lead (0.4 s) of already-sent audio is discarded. One side effect:
|
|
120
|
+
Pipecat's "bot stopped speaking" event fires up to the lead earlier than the caller
|
|
121
|
+
actually stops hearing the bot.
|
|
122
|
+
|
|
123
|
+
## Examples
|
|
124
|
+
|
|
125
|
+
- [`examples/foundational/01-say-hello.py`](examples/foundational/01-say-hello.py):
|
|
126
|
+
a minimal Pipecat pipeline that speaks one line and saves `hello.wav`.
|
|
127
|
+
- [`examples/phone/twilio_bot.py`](examples/phone/twilio_bot.py): a Twilio Media
|
|
128
|
+
Streams bot with `apply_output_lead`.
|
|
129
|
+
|
|
130
|
+
## Compatibility
|
|
131
|
+
|
|
132
|
+
Tested with Pipecat 1.8.1 and 1.12.0 on Python 3.11–3.12. `apply_output_lead` changes how
|
|
133
|
+
Pipecat's websocket output transports time their writes. A test in this repo
|
|
134
|
+
fails if a new Pipecat release changes that, and the function logs a warning and
|
|
135
|
+
does nothing for transports it doesn't support.
|
|
136
|
+
|
|
137
|
+
## License
|
|
138
|
+
|
|
139
|
+
BSD-2-Clause. Maintained by [Mirai](https://miraiminds.co) (Sona Labs Pvt Ltd).
|
|
140
|
+
This is a community integration, not maintained by the Pipecat team.
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
# pipecat-mirai
|
|
2
|
+
|
|
3
|
+
[Mirai](https://miraiminds.co) text-to-speech for [Pipecat](https://github.com/pipecat-ai/pipecat)
|
|
4
|
+
voice agents: natural Hindi, Hinglish and Gujarati voices, streamed with about
|
|
5
|
+
100 ms to first audio.
|
|
6
|
+
|
|
7
|
+
- **`MiraiTTSService`**: a Pipecat TTS service for Mirai's streaming API. Audio is
|
|
8
|
+
resampled to your pipeline's rate (8 kHz for phone calls), with no voice
|
|
9
|
+
registration or sample-rate workarounds.
|
|
10
|
+
- **`apply_output_lead()`**: stops audio breaking up on phone calls when your
|
|
11
|
+
server is busy (see [Phone calls](#phone-calls)). This works with any TTS service.
|
|
12
|
+
|
|
13
|
+
## Installation
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
uv add pipecat-mirai # or: pip install pipecat-mirai
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
## Prerequisites
|
|
20
|
+
|
|
21
|
+
- A Mirai API key (`sk_live_…`) from the [Mirai console](https://sandbox.voice.miraiminds.co).
|
|
22
|
+
Set it as `MIRAI_API_KEY`, or pass `api_key=`.
|
|
23
|
+
- Pipecat 1.8.1 or newer and Python 3.11+.
|
|
24
|
+
|
|
25
|
+
## Usage
|
|
26
|
+
|
|
27
|
+
```python
|
|
28
|
+
from pipecat_mirai import MiraiTTSService
|
|
29
|
+
|
|
30
|
+
tts = MiraiTTSService(settings=MiraiTTSService.Settings(voice="shruti"))
|
|
31
|
+
|
|
32
|
+
pipeline = Pipeline([transport.input(), stt, user_aggregator, llm, tts, transport.output(), assistant_aggregator])
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
| Voice | |
|
|
36
|
+
|---|---|
|
|
37
|
+
| `ashu` | male |
|
|
38
|
+
| `neha` | female (default) |
|
|
39
|
+
| `shruti` | female |
|
|
40
|
+
| `sameer` | male |
|
|
41
|
+
|
|
42
|
+
Write Hindi in Devanagari and English words in Latin script, as people actually
|
|
43
|
+
type Hinglish: `आपका order कल deliver होगा।`
|
|
44
|
+
|
|
45
|
+
Change the voice mid-call with Pipecat's standard settings frame:
|
|
46
|
+
|
|
47
|
+
```python
|
|
48
|
+
await task.queue_frame(TTSUpdateSettingsFrame(delta=MiraiTTSService.Settings(voice="sameer")))
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
### Parameters
|
|
52
|
+
|
|
53
|
+
| Argument | Default | |
|
|
54
|
+
|---|---|---|
|
|
55
|
+
| `api_key` | `$MIRAI_API_KEY` | Your Mirai API key |
|
|
56
|
+
| `settings` | `voice="neha"`, `model="mira-tts"` | `MiraiTTSService.Settings(...)` |
|
|
57
|
+
| `voice`, `model` | | Shortcuts for the same settings |
|
|
58
|
+
| `sample_rate` | pipeline `audio_out_sample_rate` | Output rate; Mirai's 48 kHz audio is resampled to it |
|
|
59
|
+
| `base_url` | `https://sandbox.voice.miraiminds.co/v1` | API base URL |
|
|
60
|
+
| `http_client` | own client | An `httpx.AsyncClient` you manage |
|
|
61
|
+
|
|
62
|
+
The service reports time-to-first-byte and character usage metrics and supports
|
|
63
|
+
Pipecat tracing. Interrupting the bot closes the HTTP stream at once.
|
|
64
|
+
|
|
65
|
+
## Phone calls
|
|
66
|
+
|
|
67
|
+
Pipecat's websocket transports (`FastAPIWebsocketTransport` with Twilio, Plivo,
|
|
68
|
+
Exotel or Telnyx serializers, `WebsocketServerTransport`, `WebsocketClientTransport`)
|
|
69
|
+
send audio at exactly real time. The phone provider then never holds more than
|
|
70
|
+
about 40 ms of audio. If your server's event loop stalls for longer (a blocking
|
|
71
|
+
call, a heavy parse, many calls on one CPU), the provider runs out and the caller
|
|
72
|
+
hears the voice break up. This happens with every TTS vendor, and it gets worse
|
|
73
|
+
as you add concurrent calls.
|
|
74
|
+
|
|
75
|
+
`apply_output_lead()` lets the transport send up to 0.4 s ahead, so short stalls
|
|
76
|
+
go unnoticed:
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
from pipecat_mirai import MiraiTTSService, apply_output_lead
|
|
80
|
+
|
|
81
|
+
transport = FastAPIWebsocketTransport(websocket, FastAPIWebsocketParams(
|
|
82
|
+
audio_out_enabled=True, add_wav_header=False, serializer=serializer, ...))
|
|
83
|
+
apply_output_lead(transport) # default 0.4 s; apply_output_lead(transport, 0.6) for more
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
We measured this on 8 kHz Twilio-protocol calls, with Pipecat 1.8.1, real Mirai audio
|
|
87
|
+
and 50–250 ms event-loop stalls every ~3 s per call
|
|
88
|
+
([benchmarks/phone-breaks](benchmarks/phone-breaks)):
|
|
89
|
+
|
|
90
|
+
| Concurrent calls | Stock Pipecat: speech stretched by breaks | With `apply_output_lead` |
|
|
91
|
+
|---|---|---|
|
|
92
|
+
| 3 | 6.6% (3.5 breaks per sentence) | **0.08%** |
|
|
93
|
+
| 10 | 29.1% (14.5 breaks per sentence) | **0.01%** |
|
|
94
|
+
|
|
95
|
+
When the caller interrupts, Pipecat still tells the provider to clear its queue, so
|
|
96
|
+
at most the lead (0.4 s) of already-sent audio is discarded. One side effect:
|
|
97
|
+
Pipecat's "bot stopped speaking" event fires up to the lead earlier than the caller
|
|
98
|
+
actually stops hearing the bot.
|
|
99
|
+
|
|
100
|
+
## Examples
|
|
101
|
+
|
|
102
|
+
- [`examples/foundational/01-say-hello.py`](examples/foundational/01-say-hello.py):
|
|
103
|
+
a minimal Pipecat pipeline that speaks one line and saves `hello.wav`.
|
|
104
|
+
- [`examples/phone/twilio_bot.py`](examples/phone/twilio_bot.py): a Twilio Media
|
|
105
|
+
Streams bot with `apply_output_lead`.
|
|
106
|
+
|
|
107
|
+
## Compatibility
|
|
108
|
+
|
|
109
|
+
Tested with Pipecat 1.8.1 and 1.12.0 on Python 3.11–3.12. `apply_output_lead` changes how
|
|
110
|
+
Pipecat's websocket output transports time their writes. A test in this repo
|
|
111
|
+
fails if a new Pipecat release changes that, and the function logs a warning and
|
|
112
|
+
does nothing for transports it doesn't support.
|
|
113
|
+
|
|
114
|
+
## License
|
|
115
|
+
|
|
116
|
+
BSD-2-Clause. Maintained by [Mirai](https://miraiminds.co) (Sona Labs Pvt Ltd).
|
|
117
|
+
This is a community integration, not maintained by the Pipecat team.
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# Phone-call breaks benchmark
|
|
2
|
+
|
|
3
|
+
Do callers hear the bot's voice break up when the bot's server is busy? This
|
|
4
|
+
harness measures it end to end. A real Pipecat bot (`FastAPIWebsocketTransport` +
|
|
5
|
+
`TwilioFrameSerializer`, 8 kHz μ-law, `MiraiTTSService`) talks to a simulated phone
|
|
6
|
+
provider in a separate process. The provider plays the received audio in real
|
|
7
|
+
time from a 60 ms jitter buffer and records every moment it has nothing to play.
|
|
8
|
+
|
|
9
|
+
The bot's event loop gets two kinds of load, per call:
|
|
10
|
+
|
|
11
|
+
- steady work: 1 ms of CPU every 20 ms, like VAD or STT processing
|
|
12
|
+
- stalls: on average every 3 s, the loop is blocked for 50–250 ms, the way a
|
|
13
|
+
synchronous HTTP call, a large JSON parse or a CPU-heavy step blocks a real bot
|
|
14
|
+
|
|
15
|
+
Each call speaks seven lines (about 45 s of speech) at fixed offsets over 80 s.
|
|
16
|
+
|
|
17
|
+
## Results
|
|
18
|
+
|
|
19
|
+
Pipecat 1.8.1, real Mirai audio (recorded engine streams replayed on their
|
|
20
|
+
original timing), 4-core arm64 Linux host:
|
|
21
|
+
|
|
22
|
+
| Concurrent calls | Output pacing | Breaks per sentence | Silence inserted per sentence | Speech stretched |
|
|
23
|
+
|---|---|---|---|---|
|
|
24
|
+
| 3 | stock Pipecat | 3.46 | 365 ms | 6.6% |
|
|
25
|
+
| 3 | `apply_output_lead(0.4)` | 0.04 | 4 ms | **0.08%** |
|
|
26
|
+
| 10 | stock Pipecat | 14.54 | 1,608 ms | 29.1% |
|
|
27
|
+
| 10 | `apply_output_lead(0.4)` | 0.01 | 1 ms | **0.01%** |
|
|
28
|
+
| 3 | stock, with the stand-in server below | 3.88 | 383 ms | 6.3% |
|
|
29
|
+
| 3 | `apply_output_lead(0.4)`, with the stand-in | 0 | 0 ms | **0%** |
|
|
30
|
+
|
|
31
|
+
Stock Pipecat sends audio at exactly real time, so the provider never holds more
|
|
32
|
+
than one 40 ms chunk, and every stall longer than that is heard as a gap. The
|
|
33
|
+
damage grows with the number of calls, because each call adds stalls to the same
|
|
34
|
+
event loop. This doesn't depend on the TTS vendor.
|
|
35
|
+
|
|
36
|
+
## Running it
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
uv venv && uv pip install -e ../.. "pipecat-ai[websocket]" uvicorn aiohttp websockets
|
|
40
|
+
|
|
41
|
+
# Mirai stand-in: streams a 48 kHz mono WAV (10 s or more) with Mirai's delivery timing.
|
|
42
|
+
python phone_breaks.py standin --wav speech48k.wav &
|
|
43
|
+
# Or skip the stand-in: export MIRAI_API_KEY=... and pass --tts-url https://sandbox.voice.miraiminds.co/v1
|
|
44
|
+
|
|
45
|
+
python phone_breaks.py bot --port 8765 --lead 0 & # stock Pipecat pacing
|
|
46
|
+
python phone_breaks.py bot --port 8766 --lead 0.4 & # with apply_output_lead
|
|
47
|
+
|
|
48
|
+
python phone_breaks.py phone --ws-url ws://127.0.0.1:8765/ws --calls 3 --out results/stock_c3
|
|
49
|
+
python phone_breaks.py phone --ws-url ws://127.0.0.1:8766/ws --calls 3 --out results/lead_c3
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Each run prints a JSON summary, writes `result.json` with per-sentence numbers,
|
|
53
|
+
and saves `caller_hears_call0.wav`, which is what the first caller heard.
|