pipecat-duplexjev 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pipecat_duplexjev-0.1.0/.gitignore +7 -0
- pipecat_duplexjev-0.1.0/CHANGELOG.md +7 -0
- pipecat_duplexjev-0.1.0/LICENSE +24 -0
- pipecat_duplexjev-0.1.0/PKG-INFO +111 -0
- pipecat_duplexjev-0.1.0/README.md +92 -0
- pipecat_duplexjev-0.1.0/examples/live_check.py +38 -0
- pipecat_duplexjev-0.1.0/examples/wiring.py +34 -0
- pipecat_duplexjev-0.1.0/pyproject.toml +27 -0
- pipecat_duplexjev-0.1.0/src/pipecat_duplexjev/__init__.py +5 -0
- pipecat_duplexjev-0.1.0/src/pipecat_duplexjev/turn.py +136 -0
- pipecat_duplexjev-0.1.0/tests/test_turn.py +63 -0
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0 (2026-10-10)
|
|
4
|
+
|
|
5
|
+
- First release: `DuplexJevTurnAnalyzer`, a Pipecat `BaseSmartTurn` that asks a DuplexJev gateway for the turn state.
|
|
6
|
+
- Extra questions (emotion, non-verbal sound, barge-in, or your own) answered in the same request; see `last_answers`.
|
|
7
|
+
- Tested with pipecat-ai 1.12.0.
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
BSD 2-Clause License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, Adventists.ai (AI降临派)
|
|
4
|
+
|
|
5
|
+
Redistribution and use in source and binary forms, with or without
|
|
6
|
+
modification, are permitted provided that the following conditions are met:
|
|
7
|
+
|
|
8
|
+
1. Redistributions of source code must retain the above copyright notice, this
|
|
9
|
+
list of conditions and the following disclaimer.
|
|
10
|
+
|
|
11
|
+
2. Redistributions in binary form must reproduce the above copyright notice,
|
|
12
|
+
this list of conditions and the following disclaimer in the documentation
|
|
13
|
+
and/or other materials provided with the distribution.
|
|
14
|
+
|
|
15
|
+
THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
|
|
16
|
+
AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
|
17
|
+
IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
|
|
18
|
+
DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
|
|
19
|
+
FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
|
|
20
|
+
DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
|
|
21
|
+
SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
|
|
22
|
+
CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
|
|
23
|
+
OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
24
|
+
OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: pipecat-duplexjev
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Audio end-of-turn detection for Pipecat with DuplexJev: turn state, emotion and non-verbal sounds from one forward pass, no transcript needed.
|
|
5
|
+
Project-URL: Homepage, https://github.com/adventists-ai/duplexjev/tree/main/integrations/pipecat
|
|
6
|
+
Project-URL: Models, https://huggingface.co/adventists-ai/DuplexJev-4B-Para
|
|
7
|
+
Project-URL: Paper, https://arxiv.org/abs/2610.02638
|
|
8
|
+
Author-email: Jie Jin <jiejin@adventists.ai>
|
|
9
|
+
License: BSD-2-Clause
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: duplexjev,end-of-turn,pipecat,speech-llm,turn-detection,voice-agent
|
|
12
|
+
Classifier: License :: OSI Approved :: BSD License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Requires-Dist: numpy
|
|
17
|
+
Requires-Dist: pipecat-ai>=1.0
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# pipecat-duplexjev
|
|
21
|
+
|
|
22
|
+
**Audio end-of-turn detection for [Pipecat](https://github.com/pipecat-ai/pipecat), powered by
|
|
23
|
+
[DuplexJev](https://github.com/adventists-ai/duplexjev).** It listens to *how* the user speaks, not only what they say,
|
|
24
|
+
and tells your bot whether the user is **finished**, **still talking**, **just backchanneling** ("mm-hmm") or
|
|
25
|
+
**asking you to wait**. The same request can also return **emotion**, **laughter / breathing / coughs / sighs** and
|
|
26
|
+
**barge-in** at no extra cost.
|
|
27
|
+
|
|
28
|
+
- **Four turn states, not a yes/no.** A backchannel or a "hold on" is not a reason to start talking.
|
|
29
|
+
- **Hears beyond the transcript.** Every answer is read from the audio as one constrained token of a speech LLM.
|
|
30
|
+
Nothing is decoded and no STT is needed for the decision.
|
|
31
|
+
- **More than turns, same forward pass.** Ask for emotion, non-verbal sounds or your own multiple-choice questions.
|
|
32
|
+
They come back in `last_answers` with the turn decision.
|
|
33
|
+
- **~55 ms on the server** for a turn decision with DuplexJev-4B-Para on one GPU.
|
|
34
|
+
- **Chinese and English.**
|
|
35
|
+
- On the CoDeTT turn-taking benchmark (English, zero-shot), DuplexJev-32B-Turn scores **70.0**, against **51.4** for
|
|
36
|
+
Smart Turn v3. See the [paper](https://arxiv.org/abs/2610.02638) and the
|
|
37
|
+
[project page](https://adventists-ai.github.io/duplexjev/).
|
|
38
|
+
|
|
39
|
+
Tested with `pipecat-ai` 1.12.0. Maintained by Adventists.ai (AI降临派).
|
|
40
|
+
|
|
41
|
+
## Install
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
pip install pipecat-duplexjev
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## Use
|
|
48
|
+
|
|
49
|
+
Drop it in wherever Pipecat takes a turn analyzer (Pipecat >= 1.0):
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
from pipecat.processors.aggregators.llm_response_universal import LLMContextAggregatorPair, LLMUserAggregatorParams
|
|
53
|
+
from pipecat.turns.user_stop import TurnAnalyzerUserTurnStopStrategy
|
|
54
|
+
from pipecat.turns.user_turn_strategies import UserTurnStrategies
|
|
55
|
+
from pipecat_duplexjev import DuplexJevTurnAnalyzer
|
|
56
|
+
|
|
57
|
+
turn = DuplexJevTurnAnalyzer(
|
|
58
|
+
url="http://localhost:8420", # your own gateway; omit to use the free hosted trial API
|
|
59
|
+
extra_questions=["emotion", "sound"],
|
|
60
|
+
on_answers=lambda a: print({k: v["answer"] for k, v in a.items()}),
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
aggregators = LLMContextAggregatorPair(context, user_params=LLMUserAggregatorParams(
|
|
64
|
+
vad_analyzer=SileroVADAnalyzer(),
|
|
65
|
+
user_turn_strategies=UserTurnStrategies(stop=[TurnAnalyzerUserTurnStopStrategy(turn_analyzer=turn)]),
|
|
66
|
+
))
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
A complete snippet is in [`examples/wiring.py`](examples/wiring.py). [`examples/live_check.py`](examples/live_check.py)
|
|
70
|
+
feeds real clips through the analyzer, frame by frame, the way a transport does:
|
|
71
|
+
|
|
72
|
+
```text
|
|
73
|
+
ex4.wav: COMPLETE p(finished)=1.00 turn='finished, the assistant can reply' emotion=neutral server 59.6 ms
|
|
74
|
+
ex3.wav: INCOMPLETE p(finished)=0.11 turn='not finished, still talking' emotion=sad server 53.2 ms
|
|
75
|
+
ex5.wav: INCOMPLETE p(finished)=0.28 turn='hesitating or asking to wait' emotion=angry server 56.0 ms
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
## Run the model yourself (recommended for real-time)
|
|
79
|
+
|
|
80
|
+
The default URL is our free trial API. It is rate limited (30 requests per minute) and served from mainland China,
|
|
81
|
+
so the network round trip from elsewhere can be over a second. For a live bot, serve the model next to your agent.
|
|
82
|
+
DuplexJev-4B-Para needs about 10 GB of GPU memory:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install "vllm[audio]>=0.29" duplexjev-vllm "duplexjev[server]>=0.4.1"
|
|
86
|
+
vllm serve adventists-ai/DuplexJev-4B-Para --max-model-len 4096 --port 8000 &
|
|
87
|
+
duplexjev gateway --vllm http://localhost:8000/v1 --port 8420
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Then pass `url="http://localhost:8420"`.
|
|
91
|
+
|
|
92
|
+
## Options
|
|
93
|
+
|
|
94
|
+
| argument | default | |
|
|
95
|
+
|---|---|---|
|
|
96
|
+
| `url` | hosted trial API | base URL of a `duplexjev gateway` |
|
|
97
|
+
| `api_key` | `None` | bearer key, if your gateway requires one |
|
|
98
|
+
| `lang` | `"en"` | `"en"` or `"zh"`: wording of the questions (the model hears both languages either way) |
|
|
99
|
+
| `threshold` | `0.5` | probability of "finished" from which the turn is complete |
|
|
100
|
+
| `extra_questions` | `[]` | `"emotion"`, `"sound"`, `"barge_in"`, or dicts `{"id", "text", "options"}` |
|
|
101
|
+
| `on_answers` | `None` | callback with every answer dict |
|
|
102
|
+
| `timeout` | `params.stop_secs` | HTTP timeout; on timeout Pipecat treats the turn as complete |
|
|
103
|
+
| `params` | `SmartTurnParams()` | Pipecat's usual `stop_secs`, `pre_speech_ms`, `max_duration_secs` |
|
|
104
|
+
|
|
105
|
+
If the gateway can't be reached, the analyzer logs the error and reports "not finished", so Pipecat falls back to its
|
|
106
|
+
silence timeout and the conversation never blocks.
|
|
107
|
+
|
|
108
|
+
## License
|
|
109
|
+
|
|
110
|
+
The plugin code is BSD-2-Clause. DuplexJev model weights are CC BY-NC 4.0 (non-commercial); see the
|
|
111
|
+
[model cards](https://huggingface.co/adventists-ai). For commercial use, contact jiejin@adventists.ai.
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# pipecat-duplexjev
|
|
2
|
+
|
|
3
|
+
**Audio end-of-turn detection for [Pipecat](https://github.com/pipecat-ai/pipecat), powered by
|
|
4
|
+
[DuplexJev](https://github.com/adventists-ai/duplexjev).** It listens to *how* the user speaks, not only what they say,
|
|
5
|
+
and tells your bot whether the user is **finished**, **still talking**, **just backchanneling** ("mm-hmm") or
|
|
6
|
+
**asking you to wait**. The same request can also return **emotion**, **laughter / breathing / coughs / sighs** and
|
|
7
|
+
**barge-in** at no extra cost.
|
|
8
|
+
|
|
9
|
+
- **Four turn states, not a yes/no.** A backchannel or a "hold on" is not a reason to start talking.
|
|
10
|
+
- **Hears beyond the transcript.** Every answer is read from the audio as one constrained token of a speech LLM.
|
|
11
|
+
Nothing is decoded and no STT is needed for the decision.
|
|
12
|
+
- **More than turns, same forward pass.** Ask for emotion, non-verbal sounds or your own multiple-choice questions.
|
|
13
|
+
They come back in `last_answers` with the turn decision.
|
|
14
|
+
- **~55 ms on the server** for a turn decision with DuplexJev-4B-Para on one GPU.
|
|
15
|
+
- **Chinese and English.**
|
|
16
|
+
- On the CoDeTT turn-taking benchmark (English, zero-shot), DuplexJev-32B-Turn scores **70.0**, against **51.4** for
|
|
17
|
+
Smart Turn v3. See the [paper](https://arxiv.org/abs/2610.02638) and the
|
|
18
|
+
[project page](https://adventists-ai.github.io/duplexjev/).
|
|
19
|
+
|
|
20
|
+
Tested with `pipecat-ai` 1.12.0. Maintained by Adventists.ai (AI降临派).
|
|
21
|
+
|
|
22
|
+
## Install
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install pipecat-duplexjev
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Use
|
|
29
|
+
|
|
30
|
+
Drop it in wherever Pipecat takes a turn analyzer (Pipecat >= 1.0):
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from pipecat.processors.aggregators.llm_response_universal import LLMContextAggregatorPair, LLMUserAggregatorParams
|
|
34
|
+
from pipecat.turns.user_stop import TurnAnalyzerUserTurnStopStrategy
|
|
35
|
+
from pipecat.turns.user_turn_strategies import UserTurnStrategies
|
|
36
|
+
from pipecat_duplexjev import DuplexJevTurnAnalyzer
|
|
37
|
+
|
|
38
|
+
turn = DuplexJevTurnAnalyzer(
|
|
39
|
+
url="http://localhost:8420", # your own gateway; omit to use the free hosted trial API
|
|
40
|
+
extra_questions=["emotion", "sound"],
|
|
41
|
+
on_answers=lambda a: print({k: v["answer"] for k, v in a.items()}),
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
aggregators = LLMContextAggregatorPair(context, user_params=LLMUserAggregatorParams(
|
|
45
|
+
vad_analyzer=SileroVADAnalyzer(),
|
|
46
|
+
user_turn_strategies=UserTurnStrategies(stop=[TurnAnalyzerUserTurnStopStrategy(turn_analyzer=turn)]),
|
|
47
|
+
))
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
A complete snippet is in [`examples/wiring.py`](examples/wiring.py). [`examples/live_check.py`](examples/live_check.py)
|
|
51
|
+
feeds real clips through the analyzer, frame by frame, the way a transport does:
|
|
52
|
+
|
|
53
|
+
```text
|
|
54
|
+
ex4.wav: COMPLETE p(finished)=1.00 turn='finished, the assistant can reply' emotion=neutral server 59.6 ms
|
|
55
|
+
ex3.wav: INCOMPLETE p(finished)=0.11 turn='not finished, still talking' emotion=sad server 53.2 ms
|
|
56
|
+
ex5.wav: INCOMPLETE p(finished)=0.28 turn='hesitating or asking to wait' emotion=angry server 56.0 ms
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Run the model yourself (recommended for real-time)
|
|
60
|
+
|
|
61
|
+
The default URL is our free trial API. It is rate limited (30 requests per minute) and served from mainland China,
|
|
62
|
+
so the network round trip from elsewhere can be over a second. For a live bot, serve the model next to your agent.
|
|
63
|
+
DuplexJev-4B-Para needs about 10 GB of GPU memory:
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
pip install "vllm[audio]>=0.29" duplexjev-vllm "duplexjev[server]>=0.4.1"
|
|
67
|
+
vllm serve adventists-ai/DuplexJev-4B-Para --max-model-len 4096 --port 8000 &
|
|
68
|
+
duplexjev gateway --vllm http://localhost:8000/v1 --port 8420
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Then pass `url="http://localhost:8420"`.
|
|
72
|
+
|
|
73
|
+
## Options
|
|
74
|
+
|
|
75
|
+
| argument | default | |
|
|
76
|
+
|---|---|---|
|
|
77
|
+
| `url` | hosted trial API | base URL of a `duplexjev gateway` |
|
|
78
|
+
| `api_key` | `None` | bearer key, if your gateway requires one |
|
|
79
|
+
| `lang` | `"en"` | `"en"` or `"zh"`: wording of the questions (the model hears both languages either way) |
|
|
80
|
+
| `threshold` | `0.5` | probability of "finished" from which the turn is complete |
|
|
81
|
+
| `extra_questions` | `[]` | `"emotion"`, `"sound"`, `"barge_in"`, or dicts `{"id", "text", "options"}` |
|
|
82
|
+
| `on_answers` | `None` | callback with every answer dict |
|
|
83
|
+
| `timeout` | `params.stop_secs` | HTTP timeout; on timeout Pipecat treats the turn as complete |
|
|
84
|
+
| `params` | `SmartTurnParams()` | Pipecat's usual `stop_secs`, `pre_speech_ms`, `max_duration_secs` |
|
|
85
|
+
|
|
86
|
+
If the gateway can't be reached, the analyzer logs the error and reports "not finished", so Pipecat falls back to its
|
|
87
|
+
silence timeout and the conversation never blocks.
|
|
88
|
+
|
|
89
|
+
## License
|
|
90
|
+
|
|
91
|
+
The plugin code is BSD-2-Clause. DuplexJev model weights are CC BY-NC 4.0 (non-commercial); see the
|
|
92
|
+
[model cards](https://huggingface.co/adventists-ai). For commercial use, contact jiejin@adventists.ai.
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Feed real clips through the analyzer the way Pipecat does (20 ms frames, then 200 ms of silence)."""
|
|
2
|
+
import asyncio
|
|
3
|
+
import sys
|
|
4
|
+
import time
|
|
5
|
+
import wave
|
|
6
|
+
|
|
7
|
+
import numpy as np
|
|
8
|
+
|
|
9
|
+
from pipecat_duplexjev import DuplexJevTurnAnalyzer
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def load16k(path):
|
|
13
|
+
w = wave.open(path)
|
|
14
|
+
x = np.frombuffer(w.readframes(w.getnframes()), np.int16).astype(np.float32)
|
|
15
|
+
sr = w.getframerate()
|
|
16
|
+
n = int(len(x) * 16000 / sr)
|
|
17
|
+
return np.interp(np.linspace(0, len(x) - 1, n), np.arange(len(x)), x).astype(np.int16)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
async def main(paths):
|
|
21
|
+
a = DuplexJevTurnAnalyzer(sample_rate=16000, lang="en", extra_questions=["emotion"])
|
|
22
|
+
a.set_sample_rate(16000) # done by the transport in a real pipeline
|
|
23
|
+
for p in paths:
|
|
24
|
+
x = load16k(p)
|
|
25
|
+
frame = 320
|
|
26
|
+
for i in range(0, len(x), frame):
|
|
27
|
+
a.append_audio(x[i:i + frame].tobytes(), is_speech=True)
|
|
28
|
+
for _ in range(10):
|
|
29
|
+
a.append_audio(np.zeros(frame, np.int16).tobytes(), is_speech=False)
|
|
30
|
+
t = time.time()
|
|
31
|
+
state, m = await a.analyze_end_of_turn()
|
|
32
|
+
print(f"{p}: {state.name:10s} p(finished)={m.probability:.2f} "
|
|
33
|
+
f"turn={a.last_answers['turn']['answer']!r} emotion={a.last_answers['emotion']['answer']} "
|
|
34
|
+
f"server {a.last_server_ms} ms, round trip {1000 * (time.time() - t):.0f} ms")
|
|
35
|
+
a.clear()
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
asyncio.run(main(sys.argv[1:]))
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Minimal Pipecat (>= 1.0) wiring: DuplexJev decides when the user's turn is over.
|
|
2
|
+
|
|
3
|
+
Use it in place of the default smart-turn model in any Pipecat bot. Put the aggregator pair in your pipeline as usual:
|
|
4
|
+
Pipeline([transport.input(), stt, aggregators.user(), llm, tts, transport.output(), aggregators.assistant()])
|
|
5
|
+
"""
|
|
6
|
+
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
|
7
|
+
from pipecat.processors.aggregators.llm_context import LLMContext
|
|
8
|
+
from pipecat.processors.aggregators.llm_response_universal import LLMContextAggregatorPair, LLMUserAggregatorParams
|
|
9
|
+
from pipecat.turns.user_stop import TurnAnalyzerUserTurnStopStrategy
|
|
10
|
+
from pipecat.turns.user_turn_strategies import UserTurnStrategies
|
|
11
|
+
|
|
12
|
+
from pipecat_duplexjev import DuplexJevTurnAnalyzer
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def on_answers(answers):
|
|
16
|
+
# Same forward pass, free extra signals: route an angry caller, react to laughter, ...
|
|
17
|
+
print({k: v["answer"] for k, v in answers.items()})
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
turn = DuplexJevTurnAnalyzer(
|
|
21
|
+
url="http://localhost:8420", # your `duplexjev gateway`; omit to use the free hosted trial API
|
|
22
|
+
lang="en",
|
|
23
|
+
extra_questions=["emotion", "sound"],
|
|
24
|
+
on_answers=on_answers,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
context = LLMContext()
|
|
28
|
+
aggregators = LLMContextAggregatorPair(
|
|
29
|
+
context,
|
|
30
|
+
user_params=LLMUserAggregatorParams(
|
|
31
|
+
vad_analyzer=SileroVADAnalyzer(),
|
|
32
|
+
user_turn_strategies=UserTurnStrategies(stop=[TurnAnalyzerUserTurnStopStrategy(turn_analyzer=turn)]),
|
|
33
|
+
),
|
|
34
|
+
)
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pipecat-duplexjev"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Audio end-of-turn detection for Pipecat with DuplexJev: turn state, emotion and non-verbal sounds from one forward pass, no transcript needed."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "BSD-2-Clause" }
|
|
11
|
+
authors = [{ name = "Jie Jin", email = "jiejin@adventists.ai" }]
|
|
12
|
+
requires-python = ">=3.10"
|
|
13
|
+
dependencies = ["pipecat-ai>=1.0", "numpy"]
|
|
14
|
+
keywords = ["pipecat", "turn-detection", "end-of-turn", "voice-agent", "speech-llm", "duplexjev"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"License :: OSI Approved :: BSD License",
|
|
18
|
+
"Topic :: Multimedia :: Sound/Audio :: Speech",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
[project.urls]
|
|
22
|
+
Homepage = "https://github.com/adventists-ai/duplexjev/tree/main/integrations/pipecat"
|
|
23
|
+
Models = "https://huggingface.co/adventists-ai/DuplexJev-4B-Para"
|
|
24
|
+
Paper = "https://arxiv.org/abs/2610.02638"
|
|
25
|
+
|
|
26
|
+
[tool.hatch.build.targets.wheel]
|
|
27
|
+
packages = ["src/pipecat_duplexjev"]
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
"""DuplexJev end-of-turn analyzer for Pipecat.
|
|
2
|
+
|
|
3
|
+
DuplexJev reads turn state (finished / still talking / backchannel / asking to wait) straight from the audio as one
|
|
4
|
+
constrained token of a speech LLM, so no transcript and no decoding are needed. Optional extra questions (emotion,
|
|
5
|
+
non-verbal sounds, intent, ...) are answered in the same forward pass and exposed on ``last_answers``.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import base64
|
|
10
|
+
import io
|
|
11
|
+
import json
|
|
12
|
+
import urllib.error
|
|
13
|
+
import urllib.request
|
|
14
|
+
import wave
|
|
15
|
+
from typing import Any, Callable
|
|
16
|
+
|
|
17
|
+
import numpy as np
|
|
18
|
+
from loguru import logger
|
|
19
|
+
from pipecat.audio.turn.smart_turn.base_smart_turn import BaseSmartTurn, SmartTurnParams, SmartTurnTimeoutException
|
|
20
|
+
|
|
21
|
+
DEFAULT_URL = "https://api.adventists.cn/duplexjev"
|
|
22
|
+
|
|
23
|
+
TURN_QUESTION = {
|
|
24
|
+
"en": {"id": "turn", "text": "Has the user finished speaking?",
|
|
25
|
+
"options": ["finished, the assistant can reply", "not finished, still talking",
|
|
26
|
+
"just a backchannel, not taking the turn", "hesitating or asking to wait"], "lang": "en"},
|
|
27
|
+
"zh": {"id": "turn", "text": "用户现在处于什么话轮状态?",
|
|
28
|
+
"options": ["话说完了,可以接话", "句子没说完,还在继续", "简短附和,不是要接话", "还在组织语言,或要求先等一下"],
|
|
29
|
+
"lang": "zh"},
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
PRESETS = {
|
|
33
|
+
"emotion": {"en": ("What is the speaker's emotional state?", ["neutral", "happy", "angry", "sad"]),
|
|
34
|
+
"zh": ("说话人当时的情绪状态是?", ["中性", "高兴", "生气", "伤心"])},
|
|
35
|
+
"sound": {"en": ("Besides speech, which sound can be heard in this audio?",
|
|
36
|
+
["laughter", "breathing", "coughing", "a sigh", "None of these"]),
|
|
37
|
+
"zh": ("除了说话,这段音频里还有哪种声音?", ["笑声", "呼吸声", "咳嗽", "叹气", "都没有"])},
|
|
38
|
+
"barge_in": {"en": ("Is the user trying to interrupt or take over the conversation?", ["yes", "no"]),
|
|
39
|
+
"zh": ("用户是不是想打断或者抢话?", ["是", "不是"])},
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _wav16k(audio: np.ndarray, sample_rate: int) -> bytes:
|
|
44
|
+
"""Mono float32 in [-1, 1] -> 16 kHz PCM16 WAV bytes."""
|
|
45
|
+
x = np.asarray(audio, dtype=np.float32).reshape(-1)
|
|
46
|
+
if sample_rate != 16000 and len(x):
|
|
47
|
+
n = max(1, int(round(len(x) * 16000 / sample_rate)))
|
|
48
|
+
x = np.interp(np.linspace(0, len(x) - 1, n), np.arange(len(x)), x).astype(np.float32)
|
|
49
|
+
pcm = (np.clip(x, -1.0, 1.0) * 32767).astype("<i2").tobytes()
|
|
50
|
+
buf = io.BytesIO()
|
|
51
|
+
with wave.open(buf, "wb") as w:
|
|
52
|
+
w.setnchannels(1)
|
|
53
|
+
w.setsampwidth(2)
|
|
54
|
+
w.setframerate(16000)
|
|
55
|
+
w.writeframes(pcm)
|
|
56
|
+
return buf.getvalue()
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class DuplexJevTurnAnalyzer(BaseSmartTurn):
|
|
60
|
+
"""Audio end-of-turn detection with DuplexJev, via the hosted API or your own ``duplexjev gateway``.
|
|
61
|
+
|
|
62
|
+
Args:
|
|
63
|
+
url: base URL of a DuplexJev gateway. Defaults to the free hosted trial API (rate limited, served from China).
|
|
64
|
+
For production run your own: ``vllm serve adventists-ai/DuplexJev-4B-Para`` + ``duplexjev gateway``.
|
|
65
|
+
api_key: bearer key, if the gateway requires one.
|
|
66
|
+
lang: ``"en"`` or ``"zh"``: language of the question wording (the model hears both languages either way).
|
|
67
|
+
threshold: probability of "finished" at or above which the turn is complete.
|
|
68
|
+
extra_questions: names from ``PRESETS`` (``"emotion"``, ``"sound"``, ``"barge_in"``) or question dicts
|
|
69
|
+
``{"id", "text", "options"}``; answered in the same pass, see ``last_answers``.
|
|
70
|
+
on_answers: optional callback ``f(answers: dict)`` called after every prediction.
|
|
71
|
+
timeout: HTTP timeout in seconds (defaults to ``params.stop_secs``).
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
def __init__(self, *, url: str = DEFAULT_URL, api_key: str | None = None, lang: str = "en",
|
|
75
|
+
threshold: float = 0.5, extra_questions: list | None = None,
|
|
76
|
+
on_answers: Callable[[dict], Any] | None = None, timeout: float | None = None,
|
|
77
|
+
sample_rate: int | None = None, params: SmartTurnParams | None = None):
|
|
78
|
+
super().__init__(sample_rate=sample_rate, params=params)
|
|
79
|
+
if lang not in TURN_QUESTION:
|
|
80
|
+
raise ValueError("lang must be 'en' or 'zh'")
|
|
81
|
+
self._url = url.rstrip("/") + "/v1/decide"
|
|
82
|
+
self._headers = {"Content-Type": "application/json"}
|
|
83
|
+
if api_key:
|
|
84
|
+
self._headers["Authorization"] = f"Bearer {api_key}"
|
|
85
|
+
self._lang = lang
|
|
86
|
+
self._threshold = threshold
|
|
87
|
+
self._timeout = timeout
|
|
88
|
+
self._on_answers = on_answers
|
|
89
|
+
self._questions = [TURN_QUESTION[lang]] + [self._as_question(q) for q in (extra_questions or [])]
|
|
90
|
+
self.last_answers: dict = {}
|
|
91
|
+
self.last_server_ms: float | None = None
|
|
92
|
+
|
|
93
|
+
def _as_question(self, q) -> dict:
|
|
94
|
+
if isinstance(q, str):
|
|
95
|
+
if q not in PRESETS:
|
|
96
|
+
raise ValueError(f"unknown preset {q!r}; choose from {sorted(PRESETS)} or pass a dict")
|
|
97
|
+
text, options = PRESETS[q][self._lang]
|
|
98
|
+
return {"id": q, "text": text, "options": options, "lang": self._lang}
|
|
99
|
+
d = dict(q)
|
|
100
|
+
d.setdefault("lang", self._lang)
|
|
101
|
+
return d
|
|
102
|
+
|
|
103
|
+
def _request(self, wav: bytes) -> dict:
|
|
104
|
+
payload = {"audio_b64": base64.b64encode(wav).decode(), "format": "wav", "lang": self._lang,
|
|
105
|
+
"questions": self._questions}
|
|
106
|
+
req = urllib.request.Request(self._url, data=json.dumps(payload).encode(), headers=self._headers)
|
|
107
|
+
timeout = self._timeout if self._timeout is not None else self.params.stop_secs
|
|
108
|
+
try:
|
|
109
|
+
with urllib.request.urlopen(req, timeout=timeout) as r:
|
|
110
|
+
return json.load(r)
|
|
111
|
+
except TimeoutError as e:
|
|
112
|
+
raise SmartTurnTimeoutException(str(e)) from e
|
|
113
|
+
except urllib.error.URLError as e:
|
|
114
|
+
if isinstance(e.reason, TimeoutError):
|
|
115
|
+
raise SmartTurnTimeoutException(str(e)) from e
|
|
116
|
+
raise
|
|
117
|
+
|
|
118
|
+
def _predict_endpoint(self, audio_array: np.ndarray) -> dict[str, Any]:
|
|
119
|
+
try:
|
|
120
|
+
r = self._request(_wav16k(audio_array, self.sample_rate or 16000))
|
|
121
|
+
except SmartTurnTimeoutException:
|
|
122
|
+
raise
|
|
123
|
+
except Exception as e: # network or server error: do not block the conversation
|
|
124
|
+
logger.error(f"DuplexJev request failed: {e}")
|
|
125
|
+
return {"prediction": 0, "probability": 0.0}
|
|
126
|
+
answers = r.get("answers", {})
|
|
127
|
+
self.last_answers = answers
|
|
128
|
+
self.last_server_ms = r.get("ms")
|
|
129
|
+
if self._on_answers:
|
|
130
|
+
try:
|
|
131
|
+
self._on_answers(answers)
|
|
132
|
+
except Exception as e:
|
|
133
|
+
logger.warning(f"on_answers callback raised: {e}")
|
|
134
|
+
turn = answers.get("turn", {})
|
|
135
|
+
p = float(turn.get("probs", {}).get(TURN_QUESTION[self._lang]["options"][0], 0.0))
|
|
136
|
+
return {"prediction": 1 if p >= self._threshold else 0, "probability": p}
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import threading
|
|
3
|
+
from http.server import BaseHTTPRequestHandler, HTTPServer
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pytest
|
|
7
|
+
|
|
8
|
+
from pipecat_duplexjev import DuplexJevTurnAnalyzer
|
|
9
|
+
from pipecat_duplexjev.turn import _wav16k
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class Fake(BaseHTTPRequestHandler):
|
|
13
|
+
p_finished = 0.9
|
|
14
|
+
last = None
|
|
15
|
+
|
|
16
|
+
def do_POST(self):
|
|
17
|
+
body = json.loads(self.rfile.read(int(self.headers["Content-Length"])))
|
|
18
|
+
Fake.last = body
|
|
19
|
+
opts = body["questions"][0]["options"]
|
|
20
|
+
probs = {o: 0.0 for o in opts}
|
|
21
|
+
probs[opts[0]] = Fake.p_finished
|
|
22
|
+
probs[opts[1]] = 1 - Fake.p_finished
|
|
23
|
+
ans = {"turn": {"answer": max(probs, key=probs.get), "confidence": max(probs.values()), "probs": probs}}
|
|
24
|
+
for q in body["questions"][1:]:
|
|
25
|
+
ans[q["id"]] = {"answer": q["options"][0], "confidence": 1.0, "probs": {q["options"][0]: 1.0}}
|
|
26
|
+
out = json.dumps({"answers": ans, "ms": 42.0}).encode()
|
|
27
|
+
self.send_response(200)
|
|
28
|
+
self.send_header("Content-Type", "application/json")
|
|
29
|
+
self.end_headers()
|
|
30
|
+
self.wfile.write(out)
|
|
31
|
+
|
|
32
|
+
def log_message(self, *a):
|
|
33
|
+
pass
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@pytest.fixture()
|
|
37
|
+
def server():
|
|
38
|
+
s = HTTPServer(("127.0.0.1", 0), Fake)
|
|
39
|
+
t = threading.Thread(target=s.serve_forever, daemon=True)
|
|
40
|
+
t.start()
|
|
41
|
+
yield f"http://127.0.0.1:{s.server_port}"
|
|
42
|
+
s.shutdown()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def test_wav16k_resamples():
|
|
46
|
+
wav = _wav16k(np.zeros(48000, np.float32), 48000)
|
|
47
|
+
assert wav[:4] == b"RIFF" and abs(len(wav) - 44 - 32000) <= 2
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def test_complete_and_incomplete(server):
|
|
51
|
+
a = DuplexJevTurnAnalyzer(url=server, sample_rate=16000, extra_questions=["emotion"])
|
|
52
|
+
Fake.p_finished = 0.9
|
|
53
|
+
r = a._predict_endpoint(np.zeros(16000, np.float32))
|
|
54
|
+
assert r == {"prediction": 1, "probability": 0.9}
|
|
55
|
+
assert a.last_answers["emotion"]["answer"] == "neutral" and a.last_server_ms == 42.0
|
|
56
|
+
assert [q["id"] for q in Fake.last["questions"]] == ["turn", "emotion"]
|
|
57
|
+
Fake.p_finished = 0.2
|
|
58
|
+
assert a._predict_endpoint(np.zeros(16000, np.float32))["prediction"] == 0
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def test_server_down_does_not_raise():
|
|
62
|
+
a = DuplexJevTurnAnalyzer(url="http://127.0.0.1:9", sample_rate=16000)
|
|
63
|
+
assert a._predict_endpoint(np.zeros(1600, np.float32)) == {"prediction": 0, "probability": 0.0}
|