voicemaster 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- voicemaster-0.1.0/PKG-INFO +63 -0
- voicemaster-0.1.0/README.md +40 -0
- voicemaster-0.1.0/pyproject.toml +41 -0
- voicemaster-0.1.0/setup.cfg +4 -0
- voicemaster-0.1.0/src/voicemaster/__init__.py +28 -0
- voicemaster-0.1.0/src/voicemaster/audio.py +162 -0
- voicemaster-0.1.0/src/voicemaster/call_events.py +178 -0
- voicemaster-0.1.0/src/voicemaster/engines/__init__.py +0 -0
- voicemaster-0.1.0/src/voicemaster/engines/twilio.py +1142 -0
- voicemaster-0.1.0/src/voicemaster/engines/vobiz.py +1232 -0
- voicemaster-0.1.0/src/voicemaster/errors.py +11 -0
- voicemaster-0.1.0/src/voicemaster/knowledge.py +112 -0
- voicemaster-0.1.0/src/voicemaster/live.py +37 -0
- voicemaster-0.1.0/src/voicemaster/logs.py +21 -0
- voicemaster-0.1.0/src/voicemaster/pending.py +49 -0
- voicemaster-0.1.0/src/voicemaster/prewarm.py +57 -0
- voicemaster-0.1.0/src/voicemaster/prompts.py +38 -0
- voicemaster-0.1.0/src/voicemaster/providers/__init__.py +0 -0
- voicemaster-0.1.0/src/voicemaster/providers/twilio.py +202 -0
- voicemaster-0.1.0/src/voicemaster/providers/vobiz.py +202 -0
- voicemaster-0.1.0/src/voicemaster/request.py +123 -0
- voicemaster-0.1.0/src/voicemaster/server.py +70 -0
- voicemaster-0.1.0/src/voicemaster/settings.py +75 -0
- voicemaster-0.1.0/src/voicemaster/static/dialer.html +299 -0
- voicemaster-0.1.0/src/voicemaster/tools.py +89 -0
- voicemaster-0.1.0/src/voicemaster/validation.py +44 -0
- voicemaster-0.1.0/src/voicemaster.egg-info/PKG-INFO +63 -0
- voicemaster-0.1.0/src/voicemaster.egg-info/SOURCES.txt +39 -0
- voicemaster-0.1.0/src/voicemaster.egg-info/dependency_links.txt +1 -0
- voicemaster-0.1.0/src/voicemaster.egg-info/entry_points.txt +2 -0
- voicemaster-0.1.0/src/voicemaster.egg-info/requires.txt +9 -0
- voicemaster-0.1.0/src/voicemaster.egg-info/top_level.txt +1 -0
- voicemaster-0.1.0/tests/test_call_events.py +171 -0
- voicemaster-0.1.0/tests/test_engine_events_flow.py +275 -0
- voicemaster-0.1.0/tests/test_hangup_mechanism.py +176 -0
- voicemaster-0.1.0/tests/test_knowledge.py +162 -0
- voicemaster-0.1.0/tests/test_request.py +127 -0
- voicemaster-0.1.0/tests/test_server.py +89 -0
- voicemaster-0.1.0/tests/test_time_context.py +57 -0
- voicemaster-0.1.0/tests/test_tools_and_cleanup.py +154 -0
- voicemaster-0.1.0/tests/test_transfer_destinations.py +282 -0
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: voicemaster
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Real-time AI phone calls: Gemini Live bridged to Vobiz and Twilio, with call reports, tools and knowledge search.
|
|
5
|
+
Author: Rahul Wale
|
|
6
|
+
Keywords: voice,phone,gemini,gemini-live,vobiz,twilio,ai-calling,telephony
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
11
|
+
Classifier: Framework :: FastAPI
|
|
12
|
+
Classifier: Topic :: Communications :: Telephony
|
|
13
|
+
Requires-Python: >=3.10
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: fastapi<1,>=0.141.1
|
|
16
|
+
Requires-Dist: uvicorn[standard]<1,>=0.52.4
|
|
17
|
+
Requires-Dist: aiohttp<4,>=3.14.3
|
|
18
|
+
Requires-Dist: google-genai<2,>=1.75.0
|
|
19
|
+
Requires-Dist: loguru<1,>=0.7.3
|
|
20
|
+
Requires-Dist: tzdata>=2025.2
|
|
21
|
+
Provides-Extra: test
|
|
22
|
+
Requires-Dist: httpx; extra == "test"
|
|
23
|
+
|
|
24
|
+
# voicemaster
|
|
25
|
+
|
|
26
|
+
Real-time AI phone calls. A caller (or a person being called) talks to a Gemini Live assistant over a Vobiz or Twilio phone
|
|
27
|
+
line. The library is the whole calling engine: the HTTP routes the telephony provider calls, the live audio stream in both
|
|
28
|
+
directions, the Gemini Live session, the tools the assistant may use (end the call, transfer it, search the customer's
|
|
29
|
+
documents), and the signed reports of how each call went.
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
pip install voicemaster
|
|
33
|
+
python -m voicemaster.server # Vobiz at /vobiz/..., Twilio at /twilio/..., one port (PORT, default 8760)
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
## Settings (environment variables)
|
|
37
|
+
|
|
38
|
+
| Variable | Default | Meaning |
|
|
39
|
+
|---|---|---|
|
|
40
|
+
| `PUBLIC_URL` | – | This server's public address **ending in `/vobiz`** (Vobiz). |
|
|
41
|
+
| `TWILIO_PUBLIC_URL` | – | The same, **ending in `/twilio`** (Twilio). |
|
|
42
|
+
| `API_TOKEN` | empty | If set, `POST /start` and `POST /calls/{id}/config` need the header `x-api-token`. |
|
|
43
|
+
| `EVENTS_URL`, `EVENTS_SECRET` | empty | Where call reports go, and the secret that signs them (HMAC-SHA256). |
|
|
44
|
+
| `KNOWLEDGE_SEARCH_URL` | derived from `EVENTS_URL` | Where the `search_knowledge` tool asks for passages. |
|
|
45
|
+
| `MAX_CONCURRENT_CALLS` | 10 | Calls accepted at once. |
|
|
46
|
+
| `RECORD_CALLS`, `RECORD_FORMAT`, `RECORD_MAX_LENGTH_SECONDS` | false, mp3, 300 | Recording (a call is recorded only if this *and* the call both ask). |
|
|
47
|
+
| `GEMINI_LIVE_MODEL` | `gemini-3.1-flash-live-preview` | The Gemini Live model. |
|
|
48
|
+
| `TIME_CONTEXT_TIMEZONE` | `Asia/Kolkata` | Time zone of the "what time is it" line given to the model. |
|
|
49
|
+
| `LOG_LEVEL`, `LOG_DIR` | INFO, `logs` | Logging; files rotate at 50 MB and are kept 14 days. |
|
|
50
|
+
|
|
51
|
+
The library never loads a `.env` file; load it yourself (`dotenv.load_dotenv()`) before importing `voicemaster`.
|
|
52
|
+
Calling credentials (Vobiz / Twilio accounts, the Gemini key, the assistant's prompt) arrive **per call**, never from settings.
|
|
53
|
+
|
|
54
|
+
## Building blocks
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
from voicemaster.request import VOBIZ, parse_call_request # check a POST /start body
|
|
58
|
+
from voicemaster.tools import build_tools # end_call / transfer_call / search_knowledge
|
|
59
|
+
from voicemaster.live import live_config # the Gemini Live settings for a call
|
|
60
|
+
from voicemaster.call_events import sign # the HMAC signature of a report
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
See `voicemaster/__init__.py` for the full map of modules.
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# voicemaster
|
|
2
|
+
|
|
3
|
+
Real-time AI phone calls. A caller (or a person being called) talks to a Gemini Live assistant over a Vobiz or Twilio phone
|
|
4
|
+
line. The library is the whole calling engine: the HTTP routes the telephony provider calls, the live audio stream in both
|
|
5
|
+
directions, the Gemini Live session, the tools the assistant may use (end the call, transfer it, search the customer's
|
|
6
|
+
documents), and the signed reports of how each call went.
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
pip install voicemaster
|
|
10
|
+
python -m voicemaster.server # Vobiz at /vobiz/..., Twilio at /twilio/..., one port (PORT, default 8760)
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Settings (environment variables)
|
|
14
|
+
|
|
15
|
+
| Variable | Default | Meaning |
|
|
16
|
+
|---|---|---|
|
|
17
|
+
| `PUBLIC_URL` | – | This server's public address **ending in `/vobiz`** (Vobiz). |
|
|
18
|
+
| `TWILIO_PUBLIC_URL` | – | The same, **ending in `/twilio`** (Twilio). |
|
|
19
|
+
| `API_TOKEN` | empty | If set, `POST /start` and `POST /calls/{id}/config` need the header `x-api-token`. |
|
|
20
|
+
| `EVENTS_URL`, `EVENTS_SECRET` | empty | Where call reports go, and the secret that signs them (HMAC-SHA256). |
|
|
21
|
+
| `KNOWLEDGE_SEARCH_URL` | derived from `EVENTS_URL` | Where the `search_knowledge` tool asks for passages. |
|
|
22
|
+
| `MAX_CONCURRENT_CALLS` | 10 | Calls accepted at once. |
|
|
23
|
+
| `RECORD_CALLS`, `RECORD_FORMAT`, `RECORD_MAX_LENGTH_SECONDS` | false, mp3, 300 | Recording (a call is recorded only if this *and* the call both ask). |
|
|
24
|
+
| `GEMINI_LIVE_MODEL` | `gemini-3.1-flash-live-preview` | The Gemini Live model. |
|
|
25
|
+
| `TIME_CONTEXT_TIMEZONE` | `Asia/Kolkata` | Time zone of the "what time is it" line given to the model. |
|
|
26
|
+
| `LOG_LEVEL`, `LOG_DIR` | INFO, `logs` | Logging; files rotate at 50 MB and are kept 14 days. |
|
|
27
|
+
|
|
28
|
+
The library never loads a `.env` file; load it yourself (`dotenv.load_dotenv()`) before importing `voicemaster`.
|
|
29
|
+
Calling credentials (Vobiz / Twilio accounts, the Gemini key, the assistant's prompt) arrive **per call**, never from settings.
|
|
30
|
+
|
|
31
|
+
## Building blocks
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from voicemaster.request import VOBIZ, parse_call_request # check a POST /start body
|
|
35
|
+
from voicemaster.tools import build_tools # end_call / transfer_call / search_knowledge
|
|
36
|
+
from voicemaster.live import live_config # the Gemini Live settings for a call
|
|
37
|
+
from voicemaster.call_events import sign # the HMAC signature of a report
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
See `voicemaster/__init__.py` for the full map of modules.
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "voicemaster"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Real-time AI phone calls: Gemini Live bridged to Vobiz and Twilio, with call reports, tools and knowledge search."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
authors = [{ name = "Rahul Wale" }]
|
|
12
|
+
keywords = ["voice", "phone", "gemini", "gemini-live", "vobiz", "twilio", "ai-calling", "telephony"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Programming Language :: Python :: 3",
|
|
15
|
+
"Programming Language :: Python :: 3.10",
|
|
16
|
+
"Programming Language :: Python :: 3.11",
|
|
17
|
+
"Programming Language :: Python :: 3.12",
|
|
18
|
+
"Framework :: FastAPI",
|
|
19
|
+
"Topic :: Communications :: Telephony",
|
|
20
|
+
]
|
|
21
|
+
dependencies = [
|
|
22
|
+
"fastapi>=0.141.1,<1",
|
|
23
|
+
"uvicorn[standard]>=0.52.4,<1",
|
|
24
|
+
"aiohttp>=3.14.3,<4",
|
|
25
|
+
"google-genai>=1.75.0,<2",
|
|
26
|
+
"loguru>=0.7.3,<1",
|
|
27
|
+
# The engine tells the model the current time in an Indian timezone (zoneinfo); a slim image has no system tz database.
|
|
28
|
+
"tzdata>=2025.2",
|
|
29
|
+
]
|
|
30
|
+
|
|
31
|
+
[project.optional-dependencies]
|
|
32
|
+
test = ["httpx"]
|
|
33
|
+
|
|
34
|
+
[project.scripts]
|
|
35
|
+
voicemaster-server = "voicemaster.server:main"
|
|
36
|
+
|
|
37
|
+
[tool.setuptools.packages.find]
|
|
38
|
+
where = ["src"]
|
|
39
|
+
|
|
40
|
+
[tool.setuptools.package-data]
|
|
41
|
+
voicemaster = ["static/*.html"]
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""voicemaster -- real-time AI phone calls: Gemini Live bridged to Vobiz and Twilio.
|
|
2
|
+
|
|
3
|
+
The whole calling engine lives in this package. A thin application only has to start it:
|
|
4
|
+
|
|
5
|
+
python -m voicemaster.server # both providers on one port (PORT, default 8760)
|
|
6
|
+
python -m voicemaster.engines.vobiz # the Vobiz program on its own
|
|
7
|
+
python -m voicemaster.engines.twilio # the Twilio program on its own
|
|
8
|
+
|
|
9
|
+
The parts, so each can be used and tested on its own:
|
|
10
|
+
|
|
11
|
+
voicemaster.engines the Vobiz and Twilio programs: HTTP routes, audio streams and the Gemini sessions
|
|
12
|
+
voicemaster.audio audio format helpers (PCM / mu-law, resampling)
|
|
13
|
+
voicemaster.server both programs behind one port
|
|
14
|
+
voicemaster.request check a call request (POST /start, POST /calls/{id}/config) parse_call_request
|
|
15
|
+
voicemaster.validation phone numbers, language codes, transfer destinations
|
|
16
|
+
voicemaster.tools the tools the model may call: end_call, transfer_call, search_knowledge
|
|
17
|
+
voicemaster.prompts the base instructions and the "what time is it" line given to the model
|
|
18
|
+
voicemaster.live the Gemini Live connection settings for a call
|
|
19
|
+
voicemaster.prewarm connect to Gemini before the caller is on the line
|
|
20
|
+
voicemaster.pending clean up calls that were placed but never connected
|
|
21
|
+
voicemaster.call_events report what happened on a call to the backend (signed)
|
|
22
|
+
voicemaster.knowledge the search_knowledge tool and its signed call to the backend
|
|
23
|
+
voicemaster.providers the Vobiz and Twilio REST / XML helpers
|
|
24
|
+
|
|
25
|
+
Settings are environment variables, read once by `voicemaster.settings`.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""
|
|
2
|
+
vobiz_gemini_audio.py -- format conversion between the Vobiz stream and Gemini Live
|
|
3
|
+
|
|
4
|
+
Ported verbatim from Vobiz's own official reference implementation
|
|
5
|
+
(vobiz-ai/Vobiz-Gemini-Live-Streaming, audio.py) -- this is proven, tested
|
|
6
|
+
audio math, not something to second-guess or rewrite.
|
|
7
|
+
|
|
8
|
+
Two independent directions, each with its own format:
|
|
9
|
+
|
|
10
|
+
Vobiz -> app whatever `<Stream contentType>` asked for, echoed back
|
|
11
|
+
in the WebSocket `start.mediaFormat` event:
|
|
12
|
+
L16/8000, L16/16000 or mulaw/8000
|
|
13
|
+
app -> Vobiz `playAudio.media`: L16 at 8/16/24 kHz, or mulaw/8000
|
|
14
|
+
|
|
15
|
+
Gemini Live is fixed on both sides: it accepts 16-bit PCM mono at 16 kHz and
|
|
16
|
+
emits 16-bit PCM mono at 24 kHz. So the default configuration --
|
|
17
|
+
`audio/x-l16;rate=16000` in, L16/24000 out -- needs no conversion at all in
|
|
18
|
+
either direction. Everything below exists for the other combinations (e.g. if
|
|
19
|
+
Vobiz negotiates mulaw/8000 instead of what we asked for).
|
|
20
|
+
|
|
21
|
+
Pure Python on purpose: `audioop` was removed in Python 3.13.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import struct
|
|
27
|
+
|
|
28
|
+
GEMINI_INPUT_RATE = 16000 # what Gemini Live expects on realtime input
|
|
29
|
+
GEMINI_OUTPUT_RATE = 24000 # what Gemini Live emits
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# ---------------------------------------------------------------------------
|
|
33
|
+
# G.711 mu-law
|
|
34
|
+
# ---------------------------------------------------------------------------
|
|
35
|
+
|
|
36
|
+
_BIAS = 0x84
|
|
37
|
+
_CLIP = 32635
|
|
38
|
+
|
|
39
|
+
_EXP_LUT = bytes(
|
|
40
|
+
[
|
|
41
|
+
0, 0, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 3,
|
|
42
|
+
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
|
43
|
+
5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5,
|
|
44
|
+
5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5,
|
|
45
|
+
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
|
|
46
|
+
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
|
|
47
|
+
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
|
|
48
|
+
6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6, 6,
|
|
49
|
+
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
|
50
|
+
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
|
51
|
+
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
|
52
|
+
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
|
53
|
+
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
|
54
|
+
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
|
55
|
+
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
|
56
|
+
7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7, 7,
|
|
57
|
+
]
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _encode_mulaw_sample(sample: int) -> int:
|
|
62
|
+
sign = 0x80 if sample < 0 else 0x00
|
|
63
|
+
if sample < 0:
|
|
64
|
+
sample = -sample
|
|
65
|
+
if sample > _CLIP:
|
|
66
|
+
sample = _CLIP
|
|
67
|
+
sample += _BIAS
|
|
68
|
+
exponent = _EXP_LUT[(sample >> 7) & 0xFF]
|
|
69
|
+
mantissa = (sample >> (exponent + 3)) & 0x0F
|
|
70
|
+
return ~(sign | (exponent << 4) | mantissa) & 0xFF
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _decode_mulaw_sample(byte: int) -> int:
|
|
74
|
+
byte = ~byte & 0xFF
|
|
75
|
+
t = ((byte & 0x0F) << 3) + _BIAS
|
|
76
|
+
t <<= (byte & 0x70) >> 4
|
|
77
|
+
return (_BIAS - t) if (byte & 0x80) else (t - _BIAS)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
_ENCODE_TABLE = bytes(_encode_mulaw_sample(s) for s in range(-32768, 32768))
|
|
81
|
+
_DECODE_TABLE = [_decode_mulaw_sample(b) for b in range(256)]
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def pcm16_to_mulaw(pcm: bytes) -> bytes:
|
|
85
|
+
"""Signed 16-bit little-endian PCM -> G.711 mu-law."""
|
|
86
|
+
samples = struct.unpack_from(f"<{len(pcm) // 2}h", pcm)
|
|
87
|
+
return bytes(_ENCODE_TABLE[s + 32768] for s in samples)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def mulaw_to_pcm16(mulaw: bytes) -> bytes:
|
|
91
|
+
"""G.711 mu-law -> signed 16-bit little-endian PCM."""
|
|
92
|
+
return struct.pack(f"<{len(mulaw)}h", *(_DECODE_TABLE[b] for b in mulaw))
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
# ---------------------------------------------------------------------------
|
|
96
|
+
# Resampling
|
|
97
|
+
# ---------------------------------------------------------------------------
|
|
98
|
+
|
|
99
|
+
def resample_pcm16(pcm: bytes, from_rate: int, to_rate: int) -> bytes:
|
|
100
|
+
"""
|
|
101
|
+
Linear-interpolation resample of mono 16-bit PCM.
|
|
102
|
+
|
|
103
|
+
Good enough for telephony speech and dependency-free. Only exercised for
|
|
104
|
+
the non-default format combos (see module docstring) -- the default path
|
|
105
|
+
(L16/16000 in, L16/24000 out) never calls this.
|
|
106
|
+
"""
|
|
107
|
+
if from_rate == to_rate or not pcm:
|
|
108
|
+
return pcm
|
|
109
|
+
|
|
110
|
+
count = len(pcm) // 2
|
|
111
|
+
if count == 0:
|
|
112
|
+
return b""
|
|
113
|
+
samples = struct.unpack_from(f"<{count}h", pcm)
|
|
114
|
+
|
|
115
|
+
out_count = max(1, int(count * to_rate / from_rate))
|
|
116
|
+
step = (count - 1) / out_count if out_count > 1 and count > 1 else 0.0
|
|
117
|
+
|
|
118
|
+
out = []
|
|
119
|
+
for i in range(out_count):
|
|
120
|
+
pos = i * step
|
|
121
|
+
left = int(pos)
|
|
122
|
+
right = min(left + 1, count - 1)
|
|
123
|
+
frac = pos - left
|
|
124
|
+
out.append(int(samples[left] + (samples[right] - samples[left]) * frac))
|
|
125
|
+
return struct.pack(f"<{out_count}h", *out)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
# ---------------------------------------------------------------------------
|
|
129
|
+
# Vobiz <-> Gemini
|
|
130
|
+
# ---------------------------------------------------------------------------
|
|
131
|
+
|
|
132
|
+
def parse_content_type(content_type: str) -> tuple[str, int]:
|
|
133
|
+
"""
|
|
134
|
+
`audio/x-l16;rate=16000` -> ("l16", 16000). Accepts the `encoding` value
|
|
135
|
+
from `start.mediaFormat` too, which carries no `;rate=` suffix.
|
|
136
|
+
"""
|
|
137
|
+
head, _, tail = content_type.partition(";")
|
|
138
|
+
encoding = "mulaw" if "mulaw" in head.lower() else "l16"
|
|
139
|
+
rate = 8000
|
|
140
|
+
for part in tail.split(";"):
|
|
141
|
+
key, _, value = part.partition("=")
|
|
142
|
+
if key.strip().lower() == "rate" and value.strip().isdigit():
|
|
143
|
+
rate = int(value.strip())
|
|
144
|
+
return encoding, rate
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def to_gemini(payload: bytes, encoding: str, sample_rate: int) -> bytes:
|
|
148
|
+
"""Caller audio in the stream's own format -> PCM16 mono 16 kHz."""
|
|
149
|
+
pcm = mulaw_to_pcm16(payload) if encoding == "mulaw" else payload
|
|
150
|
+
return resample_pcm16(pcm, sample_rate, GEMINI_INPUT_RATE)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def from_gemini(pcm24: bytes, encoding: str, sample_rate: int) -> bytes:
|
|
154
|
+
"""Gemini PCM16 mono 24 kHz -> the format declared in `playAudio.media`."""
|
|
155
|
+
pcm = resample_pcm16(pcm24, GEMINI_OUTPUT_RATE, sample_rate)
|
|
156
|
+
return pcm16_to_mulaw(pcm) if encoding == "mulaw" else pcm
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def frame_bytes(encoding: str, sample_rate: int, ms: int) -> int:
|
|
160
|
+
"""Raw bytes in one `playAudio` frame of `ms` milliseconds."""
|
|
161
|
+
width = 1 if encoding == "mulaw" else 2
|
|
162
|
+
return int(sample_rate * width * ms / 1000)
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Tells the backend how each call went: answered, and ended.
|
|
3
|
+
|
|
4
|
+
The engine stores nothing, so it can not be asked afterwards. Instead, each call reports two facts to the backend as
|
|
5
|
+
they happen (see docs/voice-engine-events.md in the backend repo for the contract):
|
|
6
|
+
|
|
7
|
+
call.answered the callee picked up
|
|
8
|
+
call.ended the call is over: completed, no_answer or failed, with how long it lasted
|
|
9
|
+
|
|
10
|
+
Sending never touches the call itself. An event is handed to a background task and `emit` returns at once; if the
|
|
11
|
+
backend is slow, down or refuses, the call carries on exactly as before and the failure is only logged. Events are kept
|
|
12
|
+
in memory only while they are being retried (nothing is written to disk); a restart of the engine loses the ones still
|
|
13
|
+
waiting, which is the price of storing nothing.
|
|
14
|
+
|
|
15
|
+
Off unless EVENTS_URL and EVENTS_SECRET are both set.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
import asyncio
|
|
19
|
+
import hashlib
|
|
20
|
+
import hmac
|
|
21
|
+
import json
|
|
22
|
+
import time
|
|
23
|
+
from datetime import datetime, timezone
|
|
24
|
+
|
|
25
|
+
import aiohttp
|
|
26
|
+
from loguru import logger
|
|
27
|
+
|
|
28
|
+
# The backend answers 404 for a call it does not know YET (the event can beat its own bookkeeping of a new call by a
|
|
29
|
+
# moment), and 5xx or nothing when it has a problem: all worth trying again. Seconds to wait before each retry.
|
|
30
|
+
RETRY_DELAYS = (1, 3, 10, 30)
|
|
31
|
+
SEND_TIMEOUT_SECONDS = 10
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def iso(timestamp: float) -> str:
|
|
35
|
+
"""Seconds since 1970 -> "2026-03-01T10:00:05Z" (always with a time zone, as the backend requires)."""
|
|
36
|
+
return datetime.fromtimestamp(timestamp, tz=timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def sign(secret: str, timestamp: str, body: bytes) -> str:
|
|
40
|
+
"""The signature the backend checks: HMAC-SHA256 over "<timestamp>." + the exact body, as sha256=<hex>."""
|
|
41
|
+
digest = hmac.new(secret.encode(), timestamp.encode() + b"." + body, hashlib.sha256).hexdigest()
|
|
42
|
+
return f"sha256={digest}"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def answered_event(call_uuid: str, answered_at: float) -> dict:
|
|
46
|
+
return {"event": "call.answered", "call_uuid": call_uuid, "answered_at": iso(answered_at)}
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def transcript_event(call_uuid: str, entries: list[dict]) -> dict:
|
|
50
|
+
"""entries: [{"speaker": "agent"|"caller", "text": "...", "at": <epoch seconds>}, ...], oldest first.
|
|
51
|
+
One entry per turn (what the model actually heard/said was already assembled turn by turn), not per audio chunk."""
|
|
52
|
+
return {
|
|
53
|
+
"event": "call.transcript",
|
|
54
|
+
"call_uuid": call_uuid,
|
|
55
|
+
"entries": [{"speaker": entry["speaker"], "content": entry["text"], "spoken_at": iso(entry["at"])} for entry in entries],
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def recording_event(call_uuid: str, recording_url: str, duration_seconds: int | None, format: str) -> dict:
|
|
60
|
+
event = {"event": "call.recording", "call_uuid": call_uuid, "recording_url": recording_url, "format": format}
|
|
61
|
+
if duration_seconds is not None:
|
|
62
|
+
event["duration_seconds"] = duration_seconds
|
|
63
|
+
return event
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
class GeminiUsageMeter:
|
|
67
|
+
"""Adds up the tokens Gemini Live reports it used, by modality, so the backend can work out what the call cost.
|
|
68
|
+
|
|
69
|
+
Gemini sends `usage_metadata` once per turn, and each one covers that turn alone (its prompt already includes the
|
|
70
|
+
conversation so far, which Google bills again every turn) -- so the call's total is the SUM of all of them, not the
|
|
71
|
+
last one. Only reads messages; it never touches the audio path.
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
def __init__(self) -> None:
|
|
75
|
+
self.input: dict[str, int] = {}
|
|
76
|
+
self.output: dict[str, int] = {}
|
|
77
|
+
|
|
78
|
+
def add(self, message) -> None:
|
|
79
|
+
usage = getattr(message, "usage_metadata", None)
|
|
80
|
+
if not usage:
|
|
81
|
+
return
|
|
82
|
+
try:
|
|
83
|
+
self._count(self.input, getattr(usage, "prompt_tokens_details", None), getattr(usage, "prompt_token_count", None))
|
|
84
|
+
self._count(self.output, getattr(usage, "response_tokens_details", None), getattr(usage, "response_token_count", None))
|
|
85
|
+
except Exception: # metering must never be able to break a live call
|
|
86
|
+
pass
|
|
87
|
+
|
|
88
|
+
@staticmethod
|
|
89
|
+
def _count(into: dict[str, int], details, total) -> None:
|
|
90
|
+
if details:
|
|
91
|
+
for item in details:
|
|
92
|
+
modality = getattr(getattr(item, "modality", None), "name", None) or "UNKNOWN"
|
|
93
|
+
into[modality] = into.get(modality, 0) + int(getattr(item, "token_count", 0) or 0)
|
|
94
|
+
elif total:
|
|
95
|
+
# No per-modality split: count it as AUDIO, the dearer rate, rather than undercount.
|
|
96
|
+
into["AUDIO"] = into.get("AUDIO", 0) + int(total)
|
|
97
|
+
|
|
98
|
+
def as_event_field(self) -> dict | None:
|
|
99
|
+
if not self.input and not self.output:
|
|
100
|
+
return None
|
|
101
|
+
return {"gemini": {"input": dict(self.input), "output": dict(self.output)}}
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def ended_event(call_uuid: str, end_reason: str, ended_at: float, answered_at: float | None, transferred: bool = False, ai_usage: dict | None = None) -> dict:
|
|
105
|
+
"""end_reason: completed | no_answer | busy | failed | cancelled. A call nobody answered has no answered_at and no duration.
|
|
106
|
+
|
|
107
|
+
transferred: the call was handed to a person (Vobiz accepted the transfer). The backend charges a transfer fee for it.
|
|
108
|
+
"""
|
|
109
|
+
event = {"event": "call.ended", "call_uuid": call_uuid, "end_reason": end_reason, "ended_at": iso(ended_at), "transferred": transferred}
|
|
110
|
+
if answered_at is not None:
|
|
111
|
+
event["answered_at"] = iso(answered_at)
|
|
112
|
+
event["duration_seconds"] = max(0, int(ended_at - answered_at))
|
|
113
|
+
if ai_usage:
|
|
114
|
+
event["ai_usage"] = ai_usage
|
|
115
|
+
return event
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class CallEventSender:
|
|
119
|
+
def __init__(self, url: str, secret: str, session: aiohttp.ClientSession):
|
|
120
|
+
self.url = url.strip()
|
|
121
|
+
self.secret = secret.strip()
|
|
122
|
+
self.session = session
|
|
123
|
+
self._tasks: set[asyncio.Task] = set()
|
|
124
|
+
|
|
125
|
+
@property
|
|
126
|
+
def enabled(self) -> bool:
|
|
127
|
+
return bool(self.url and self.secret)
|
|
128
|
+
|
|
129
|
+
def emit(self, event: dict) -> None:
|
|
130
|
+
"""Hands the event to a background task and returns at once. Never raises, never waits: safe inside a live call."""
|
|
131
|
+
if not self.enabled:
|
|
132
|
+
return
|
|
133
|
+
try:
|
|
134
|
+
task = asyncio.get_running_loop().create_task(self._deliver(event))
|
|
135
|
+
except RuntimeError:
|
|
136
|
+
return
|
|
137
|
+
self._tasks.add(task)
|
|
138
|
+
task.add_done_callback(self._tasks.discard)
|
|
139
|
+
|
|
140
|
+
async def drain(self, timeout: float = 5.0) -> None:
|
|
141
|
+
"""At shutdown: gives events that are still being sent a moment to arrive."""
|
|
142
|
+
if self._tasks:
|
|
143
|
+
await asyncio.wait(list(self._tasks), timeout=timeout)
|
|
144
|
+
|
|
145
|
+
async def _deliver(self, event: dict) -> None:
|
|
146
|
+
tag = f"{event['event']} {event['call_uuid'][:8]}"
|
|
147
|
+
body = json.dumps(event, separators=(",", ":")).encode() # sent exactly as signed
|
|
148
|
+
for attempt in range(len(RETRY_DELAYS) + 1):
|
|
149
|
+
status = await self._post(body)
|
|
150
|
+
if status == 200:
|
|
151
|
+
return
|
|
152
|
+
retryable = status is None or status == 404 or status >= 500
|
|
153
|
+
if not retryable:
|
|
154
|
+
# 401: wrong secret or a clock that is off. 400: the event is not valid. Trying again changes nothing.
|
|
155
|
+
logger.error(f"Call event {tag} refused by the backend (HTTP {status}); not retrying")
|
|
156
|
+
return
|
|
157
|
+
if attempt == len(RETRY_DELAYS):
|
|
158
|
+
logger.warning(f"Call event {tag} not delivered after {attempt + 1} tries (last: {status or 'no answer'}); giving up")
|
|
159
|
+
return
|
|
160
|
+
await asyncio.sleep(RETRY_DELAYS[attempt])
|
|
161
|
+
|
|
162
|
+
async def _post(self, body: bytes) -> int | None:
|
|
163
|
+
timestamp = str(int(time.time()))
|
|
164
|
+
headers = {
|
|
165
|
+
"Content-Type": "application/json",
|
|
166
|
+
"X-Engine-Timestamp": timestamp,
|
|
167
|
+
"X-Engine-Signature": sign(self.secret, timestamp, body),
|
|
168
|
+
}
|
|
169
|
+
try:
|
|
170
|
+
async with self.session.post(
|
|
171
|
+
self.url, data=body, headers=headers, timeout=aiohttp.ClientTimeout(total=SEND_TIMEOUT_SECONDS)
|
|
172
|
+
) as response:
|
|
173
|
+
await response.read()
|
|
174
|
+
return response.status
|
|
175
|
+
except Exception as exc:
|
|
176
|
+
# Only the kind of failure is logged: never the address parameters, the body or the signature.
|
|
177
|
+
logger.debug(f"Call event not sent: {type(exc).__name__}")
|
|
178
|
+
return None
|
|
File without changes
|