pipecat-duplexjev 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ """Pipecat integration for DuplexJev speech decision models."""
2
+ from .turn import DEFAULT_URL, PRESETS, TURN_QUESTION, DuplexJevTurnAnalyzer
3
+
4
+ __all__ = ["DuplexJevTurnAnalyzer", "DEFAULT_URL", "PRESETS", "TURN_QUESTION"]
5
+ __version__ = "0.1.0"
@@ -0,0 +1,136 @@
1
+ """DuplexJev end-of-turn analyzer for Pipecat.
2
+
3
+ DuplexJev reads turn state (finished / still talking / backchannel / asking to wait) straight from the audio as one
4
+ constrained token of a speech LLM, so no transcript and no decoding are needed. Optional extra questions (emotion,
5
+ non-verbal sounds, intent, ...) are answered in the same forward pass and exposed on ``last_answers``.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import base64
10
+ import io
11
+ import json
12
+ import urllib.error
13
+ import urllib.request
14
+ import wave
15
+ from typing import Any, Callable
16
+
17
+ import numpy as np
18
+ from loguru import logger
19
+ from pipecat.audio.turn.smart_turn.base_smart_turn import BaseSmartTurn, SmartTurnParams, SmartTurnTimeoutException
20
+
21
+ DEFAULT_URL = "https://api.adventists.cn/duplexjev"
22
+
23
+ TURN_QUESTION = {
24
+ "en": {"id": "turn", "text": "Has the user finished speaking?",
25
+ "options": ["finished, the assistant can reply", "not finished, still talking",
26
+ "just a backchannel, not taking the turn", "hesitating or asking to wait"], "lang": "en"},
27
+ "zh": {"id": "turn", "text": "用户现在处于什么话轮状态?",
28
+ "options": ["话说完了,可以接话", "句子没说完,还在继续", "简短附和,不是要接话", "还在组织语言,或要求先等一下"],
29
+ "lang": "zh"},
30
+ }
31
+
32
+ PRESETS = {
33
+ "emotion": {"en": ("What is the speaker's emotional state?", ["neutral", "happy", "angry", "sad"]),
34
+ "zh": ("说话人当时的情绪状态是?", ["中性", "高兴", "生气", "伤心"])},
35
+ "sound": {"en": ("Besides speech, which sound can be heard in this audio?",
36
+ ["laughter", "breathing", "coughing", "a sigh", "None of these"]),
37
+ "zh": ("除了说话,这段音频里还有哪种声音?", ["笑声", "呼吸声", "咳嗽", "叹气", "都没有"])},
38
+ "barge_in": {"en": ("Is the user trying to interrupt or take over the conversation?", ["yes", "no"]),
39
+ "zh": ("用户是不是想打断或者抢话?", ["是", "不是"])},
40
+ }
41
+
42
+
43
+ def _wav16k(audio: np.ndarray, sample_rate: int) -> bytes:
44
+ """Mono float32 in [-1, 1] -> 16 kHz PCM16 WAV bytes."""
45
+ x = np.asarray(audio, dtype=np.float32).reshape(-1)
46
+ if sample_rate != 16000 and len(x):
47
+ n = max(1, int(round(len(x) * 16000 / sample_rate)))
48
+ x = np.interp(np.linspace(0, len(x) - 1, n), np.arange(len(x)), x).astype(np.float32)
49
+ pcm = (np.clip(x, -1.0, 1.0) * 32767).astype("<i2").tobytes()
50
+ buf = io.BytesIO()
51
+ with wave.open(buf, "wb") as w:
52
+ w.setnchannels(1)
53
+ w.setsampwidth(2)
54
+ w.setframerate(16000)
55
+ w.writeframes(pcm)
56
+ return buf.getvalue()
57
+
58
+
59
+ class DuplexJevTurnAnalyzer(BaseSmartTurn):
60
+ """Audio end-of-turn detection with DuplexJev, via the hosted API or your own ``duplexjev gateway``.
61
+
62
+ Args:
63
+ url: base URL of a DuplexJev gateway. Defaults to the free hosted trial API (rate limited, served from China).
64
+ For production run your own: ``vllm serve adventists-ai/DuplexJev-4B-Para`` + ``duplexjev gateway``.
65
+ api_key: bearer key, if the gateway requires one.
66
+ lang: ``"en"`` or ``"zh"``: language of the question wording (the model hears both languages either way).
67
+ threshold: probability of "finished" at or above which the turn is complete.
68
+ extra_questions: names from ``PRESETS`` (``"emotion"``, ``"sound"``, ``"barge_in"``) or question dicts
69
+ ``{"id", "text", "options"}``; answered in the same pass, see ``last_answers``.
70
+ on_answers: optional callback ``f(answers: dict)`` called after every prediction.
71
+ timeout: HTTP timeout in seconds (defaults to ``params.stop_secs``).
72
+ """
73
+
74
+ def __init__(self, *, url: str = DEFAULT_URL, api_key: str | None = None, lang: str = "en",
75
+ threshold: float = 0.5, extra_questions: list | None = None,
76
+ on_answers: Callable[[dict], Any] | None = None, timeout: float | None = None,
77
+ sample_rate: int | None = None, params: SmartTurnParams | None = None):
78
+ super().__init__(sample_rate=sample_rate, params=params)
79
+ if lang not in TURN_QUESTION:
80
+ raise ValueError("lang must be 'en' or 'zh'")
81
+ self._url = url.rstrip("/") + "/v1/decide"
82
+ self._headers = {"Content-Type": "application/json"}
83
+ if api_key:
84
+ self._headers["Authorization"] = f"Bearer {api_key}"
85
+ self._lang = lang
86
+ self._threshold = threshold
87
+ self._timeout = timeout
88
+ self._on_answers = on_answers
89
+ self._questions = [TURN_QUESTION[lang]] + [self._as_question(q) for q in (extra_questions or [])]
90
+ self.last_answers: dict = {}
91
+ self.last_server_ms: float | None = None
92
+
93
+ def _as_question(self, q) -> dict:
94
+ if isinstance(q, str):
95
+ if q not in PRESETS:
96
+ raise ValueError(f"unknown preset {q!r}; choose from {sorted(PRESETS)} or pass a dict")
97
+ text, options = PRESETS[q][self._lang]
98
+ return {"id": q, "text": text, "options": options, "lang": self._lang}
99
+ d = dict(q)
100
+ d.setdefault("lang", self._lang)
101
+ return d
102
+
103
+ def _request(self, wav: bytes) -> dict:
104
+ payload = {"audio_b64": base64.b64encode(wav).decode(), "format": "wav", "lang": self._lang,
105
+ "questions": self._questions}
106
+ req = urllib.request.Request(self._url, data=json.dumps(payload).encode(), headers=self._headers)
107
+ timeout = self._timeout if self._timeout is not None else self.params.stop_secs
108
+ try:
109
+ with urllib.request.urlopen(req, timeout=timeout) as r:
110
+ return json.load(r)
111
+ except TimeoutError as e:
112
+ raise SmartTurnTimeoutException(str(e)) from e
113
+ except urllib.error.URLError as e:
114
+ if isinstance(e.reason, TimeoutError):
115
+ raise SmartTurnTimeoutException(str(e)) from e
116
+ raise
117
+
118
+ def _predict_endpoint(self, audio_array: np.ndarray) -> dict[str, Any]:
119
+ try:
120
+ r = self._request(_wav16k(audio_array, self.sample_rate or 16000))
121
+ except SmartTurnTimeoutException:
122
+ raise
123
+ except Exception as e: # network or server error: do not block the conversation
124
+ logger.error(f"DuplexJev request failed: {e}")
125
+ return {"prediction": 0, "probability": 0.0}
126
+ answers = r.get("answers", {})
127
+ self.last_answers = answers
128
+ self.last_server_ms = r.get("ms")
129
+ if self._on_answers:
130
+ try:
131
+ self._on_answers(answers)
132
+ except Exception as e:
133
+ logger.warning(f"on_answers callback raised: {e}")
134
+ turn = answers.get("turn", {})
135
+ p = float(turn.get("probs", {}).get(TURN_QUESTION[self._lang]["options"][0], 0.0))
136
+ return {"prediction": 1 if p >= self._threshold else 0, "probability": p}
@@ -0,0 +1,111 @@
1
+ Metadata-Version: 2.5
2
+ Name: pipecat-duplexjev
3
+ Version: 0.1.0
4
+ Summary: Audio end-of-turn detection for Pipecat with DuplexJev: turn state, emotion and non-verbal sounds from one forward pass, no transcript needed.
5
+ Project-URL: Homepage, https://github.com/adventists-ai/duplexjev/tree/main/integrations/pipecat
6
+ Project-URL: Models, https://huggingface.co/adventists-ai/DuplexJev-4B-Para
7
+ Project-URL: Paper, https://arxiv.org/abs/2610.02638
8
+ Author-email: Jie Jin <jiejin@adventists.ai>
9
+ License: BSD-2-Clause
10
+ License-File: LICENSE
11
+ Keywords: duplexjev,end-of-turn,pipecat,speech-llm,turn-detection,voice-agent
12
+ Classifier: License :: OSI Approved :: BSD License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
15
+ Requires-Python: >=3.10
16
+ Requires-Dist: numpy
17
+ Requires-Dist: pipecat-ai>=1.0
18
+ Description-Content-Type: text/markdown
19
+
20
+ # pipecat-duplexjev
21
+
22
+ **Audio end-of-turn detection for [Pipecat](https://github.com/pipecat-ai/pipecat), powered by
23
+ [DuplexJev](https://github.com/adventists-ai/duplexjev).** It listens to *how* the user speaks, not only what they say,
24
+ and tells your bot whether the user is **finished**, **still talking**, **just backchanneling** ("mm-hmm") or
25
+ **asking you to wait**. The same request can also return **emotion**, **laughter / breathing / coughs / sighs** and
26
+ **barge-in** at no extra cost.
27
+
28
+ - **Four turn states, not a yes/no.** A backchannel or a "hold on" is not a reason to start talking.
29
+ - **Hears beyond the transcript.** Every answer is read from the audio as one constrained token of a speech LLM.
30
+ Nothing is decoded and no STT is needed for the decision.
31
+ - **More than turns, same forward pass.** Ask for emotion, non-verbal sounds or your own multiple-choice questions.
32
+ They come back in `last_answers` with the turn decision.
33
+ - **~55 ms on the server** for a turn decision with DuplexJev-4B-Para on one GPU.
34
+ - **Chinese and English.**
35
+ - On the CoDeTT turn-taking benchmark (English, zero-shot), DuplexJev-32B-Turn scores **70.0**, against **51.4** for
36
+ Smart Turn v3. See the [paper](https://arxiv.org/abs/2610.02638) and the
37
+ [project page](https://adventists-ai.github.io/duplexjev/).
38
+
39
+ Tested with `pipecat-ai` 1.12.0. Maintained by Adventists.ai (AI降临派).
40
+
41
+ ## Install
42
+
43
+ ```bash
44
+ pip install pipecat-duplexjev
45
+ ```
46
+
47
+ ## Use
48
+
49
+ Drop it in wherever Pipecat takes a turn analyzer (Pipecat >= 1.0):
50
+
51
+ ```python
52
+ from pipecat.processors.aggregators.llm_response_universal import LLMContextAggregatorPair, LLMUserAggregatorParams
53
+ from pipecat.turns.user_stop import TurnAnalyzerUserTurnStopStrategy
54
+ from pipecat.turns.user_turn_strategies import UserTurnStrategies
55
+ from pipecat_duplexjev import DuplexJevTurnAnalyzer
56
+
57
+ turn = DuplexJevTurnAnalyzer(
58
+ url="http://localhost:8420", # your own gateway; omit to use the free hosted trial API
59
+ extra_questions=["emotion", "sound"],
60
+ on_answers=lambda a: print({k: v["answer"] for k, v in a.items()}),
61
+ )
62
+
63
+ aggregators = LLMContextAggregatorPair(context, user_params=LLMUserAggregatorParams(
64
+ vad_analyzer=SileroVADAnalyzer(),
65
+ user_turn_strategies=UserTurnStrategies(stop=[TurnAnalyzerUserTurnStopStrategy(turn_analyzer=turn)]),
66
+ ))
67
+ ```
68
+
69
+ A complete snippet is in [`examples/wiring.py`](examples/wiring.py). [`examples/live_check.py`](examples/live_check.py)
70
+ feeds real clips through the analyzer, frame by frame, the way a transport does:
71
+
72
+ ```text
73
+ ex4.wav: COMPLETE p(finished)=1.00 turn='finished, the assistant can reply' emotion=neutral server 59.6 ms
74
+ ex3.wav: INCOMPLETE p(finished)=0.11 turn='not finished, still talking' emotion=sad server 53.2 ms
75
+ ex5.wav: INCOMPLETE p(finished)=0.28 turn='hesitating or asking to wait' emotion=angry server 56.0 ms
76
+ ```
77
+
78
+ ## Run the model yourself (recommended for real-time)
79
+
80
+ The default URL is our free trial API. It is rate limited (30 requests per minute) and served from mainland China,
81
+ so the network round trip from elsewhere can be over a second. For a live bot, serve the model next to your agent.
82
+ DuplexJev-4B-Para needs about 10 GB of GPU memory:
83
+
84
+ ```bash
85
+ pip install "vllm[audio]>=0.29" duplexjev-vllm "duplexjev[server]>=0.4.1"
86
+ vllm serve adventists-ai/DuplexJev-4B-Para --max-model-len 4096 --port 8000 &
87
+ duplexjev gateway --vllm http://localhost:8000/v1 --port 8420
88
+ ```
89
+
90
+ Then pass `url="http://localhost:8420"`.
91
+
92
+ ## Options
93
+
94
+ | argument | default | |
95
+ |---|---|---|
96
+ | `url` | hosted trial API | base URL of a `duplexjev gateway` |
97
+ | `api_key` | `None` | bearer key, if your gateway requires one |
98
+ | `lang` | `"en"` | `"en"` or `"zh"`: wording of the questions (the model hears both languages either way) |
99
+ | `threshold` | `0.5` | probability of "finished" from which the turn is complete |
100
+ | `extra_questions` | `[]` | `"emotion"`, `"sound"`, `"barge_in"`, or dicts `{"id", "text", "options"}` |
101
+ | `on_answers` | `None` | callback with every answer dict |
102
+ | `timeout` | `params.stop_secs` | HTTP timeout; on timeout Pipecat treats the turn as complete |
103
+ | `params` | `SmartTurnParams()` | Pipecat's usual `stop_secs`, `pre_speech_ms`, `max_duration_secs` |
104
+
105
+ If the gateway can't be reached, the analyzer logs the error and reports "not finished", so Pipecat falls back to its
106
+ silence timeout and the conversation never blocks.
107
+
108
+ ## License
109
+
110
+ The plugin code is BSD-2-Clause. DuplexJev model weights are CC BY-NC 4.0 (non-commercial); see the
111
+ [model cards](https://huggingface.co/adventists-ai). For commercial use, contact jiejin@adventists.ai.
@@ -0,0 +1,6 @@
1
+ pipecat_duplexjev/__init__.py,sha256=mPSbS_djP21XvieyTYp6iSl53h2pGPcP-qvh3oHq4gU,243
2
+ pipecat_duplexjev/turn.py,sha256=o5GV7-boEpcmREql7hE9BQOSfG7a6wukUYO3r3SL1to,6900
3
+ pipecat_duplexjev-0.1.0.dist-info/METADATA,sha256=tyDLSGrYzRqHk0JBp9pnWuKLpO2YiLdYCJnGqqgP24U,5483
4
+ pipecat_duplexjev-0.1.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
5
+ pipecat_duplexjev-0.1.0.dist-info/licenses/LICENSE,sha256=nmfjvVoTdFz0Qzyh6vHkBLTW-YpHWApu0IhvpZlJEs0,1316
6
+ pipecat_duplexjev-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,24 @@
1
+ BSD 2-Clause License
2
+
3
+ Copyright (c) 2026, Adventists.ai (AI降临派)
4
+
5
+ Redistribution and use in source and binary forms, with or without
6
+ modification, are permitted provided that the following conditions are met:
7
+
8
+ 1. Redistributions of source code must retain the above copyright notice, this
9
+ list of conditions and the following disclaimer.
10
+
11
+ 2. Redistributions in binary form must reproduce the above copyright notice,
12
+ this list of conditions and the following disclaimer in the documentation
13
+ and/or other materials provided with the distribution.
14
+
15
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS"
16
+ AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
17
+ IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
18
+ DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR CONTRIBUTORS BE LIABLE
19
+ FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
20
+ DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR
21
+ SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER
22
+ CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY,
23
+ OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
24
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.