asr-concurrent-stream 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- asr_concurrent_stream-1.0.0.dist-info/METADATA +168 -0
- asr_concurrent_stream-1.0.0.dist-info/RECORD +18 -0
- asr_concurrent_stream-1.0.0.dist-info/WHEEL +5 -0
- asr_concurrent_stream-1.0.0.dist-info/entry_points.txt +3 -0
- asr_concurrent_stream-1.0.0.dist-info/top_level.txt +3 -0
- client/__init__.py +1 -0
- client/example_client.py +409 -0
- proto/__init__.py +46 -0
- proto/asr_pb2.py +60 -0
- proto/asr_pb2_grpc.py +269 -0
- server/__init__.py +1 -0
- server/_launch_server.py +73 -0
- server/asr_grpc_server.py +459 -0
- server/asr_model.py +213 -0
- server/asr_processor.py +39 -0
- server/asr_utils.py +133 -0
- server/inference_coordinator.py +480 -0
- server/stream_manager.py +337 -0
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: asr-concurrent-stream
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Concurrent streaming gRPC server and client for Qwen3-ASR
|
|
5
|
+
Author: Alibaba Qwen Team
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/Qwen/Qwen3-ASR
|
|
8
|
+
Keywords: asr,speech-recognition,grpc,streaming,qwen3
|
|
9
|
+
Requires-Python: >=3.9
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
Requires-Dist: grpcio
|
|
12
|
+
Requires-Dist: numpy
|
|
13
|
+
Requires-Dist: protobuf
|
|
14
|
+
Requires-Dist: vllm
|
|
15
|
+
Requires-Dist: transformers
|
|
16
|
+
Provides-Extra: client
|
|
17
|
+
Requires-Dist: soundfile; extra == "client"
|
|
18
|
+
Requires-Dist: torch; extra == "client"
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: grpcio-tools; extra == "dev"
|
|
21
|
+
|
|
22
|
+
# ASR Concurrent Stream Server
|
|
23
|
+
|
|
24
|
+
Concurrent streaming gRPC server and client for Qwen3-ASR.
|
|
25
|
+
|
|
26
|
+
## Installation
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install asr-concurrent-stream
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
For client-side audio loading and VAD segmentation:
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install asr-concurrent-stream[client]
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
For development (regenerating protobuf code):
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
pip install asr-concurrent-stream[dev]
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
**Note:** The server requires `vllm` and `transformers` to run the Qwen3-ASR model. These are installed automatically with the package.
|
|
45
|
+
|
|
46
|
+
## Quick Start
|
|
47
|
+
|
|
48
|
+
### Start the server
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
asr-concurrent-server
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
The server reads configuration from environment variables:
|
|
55
|
+
|
|
56
|
+
| Variable | Default | Description |
|
|
57
|
+
|----------|---------|-------------|
|
|
58
|
+
| `ASR_PORT` | `8000` | gRPC server port |
|
|
59
|
+
| `ASR_MODEL_PATH` | `Qwen/Qwen3-ASR-1.7B` | Model path or HuggingFace ID |
|
|
60
|
+
| `ASR_GPU_MEMORY_UTILIZATION` | `0.30` | GPU memory fraction |
|
|
61
|
+
| `ASR_MAX_MODEL_LEN` | `4096` | Maximum model sequence length |
|
|
62
|
+
| `ASR_MAX_CONCURRENT_STREAMS` | `15` | Maximum concurrent streams |
|
|
63
|
+
| `ASR_MAX_BATCH_SIZE` | `8` | Maximum batch size |
|
|
64
|
+
| `ASR_BATCH_TIMEOUT_MS` | `50` | Batch timeout in milliseconds |
|
|
65
|
+
| `ASR_WORKER_THREADS` | `4` | Number of worker threads |
|
|
66
|
+
| `ASR_HEALTH_PORT` | `8080` | HTTP health check port |
|
|
67
|
+
|
|
68
|
+
### Run the client
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
asr-concurrent-client --audio path/to/audio.wav --language Italian
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
## Dependencies
|
|
75
|
+
|
|
76
|
+
| Package | Purpose |
|
|
77
|
+
|---------|---------|
|
|
78
|
+
| `grpcio` | gRPC framework |
|
|
79
|
+
| `numpy` | Audio array handling |
|
|
80
|
+
| `protobuf` | Protocol buffer serialization |
|
|
81
|
+
| `vllm` | vLLM inference engine |
|
|
82
|
+
| `transformers` | Model processor and tokenizer |
|
|
83
|
+
|
|
84
|
+
Optional client dependencies: `soundfile`, `torch` (for Silero VAD segmentation).
|
|
85
|
+
|
|
86
|
+
---
|
|
87
|
+
|
|
88
|
+
*Docker image for the Qwen3-ASR concurrent streaming gRPC server.*
|
|
89
|
+
|
|
90
|
+
## Image Composition
|
|
91
|
+
|
|
92
|
+
| Layer | Detail |
|
|
93
|
+
|-------|--------|
|
|
94
|
+
| **Base OS** | Ubuntu 24.04 LTS |
|
|
95
|
+
| **CUDA** | NVIDIA CUDA 12.8.0 (`nvidia/cuda:12.8.0-runtime-ubuntu24.04`) |
|
|
96
|
+
| **Python** | 3.12 (deadsnakes PPA), installed into `/opt/venv` |
|
|
97
|
+
| **Model** | `Qwen/Qwen3-ASR-1.7B` baked into the image |
|
|
98
|
+
| **Server port** | Container `8002` (map to any host port at runtime) |
|
|
99
|
+
|
|
100
|
+
## Installation Sequence
|
|
101
|
+
|
|
102
|
+
1. **System dependencies** — `curl`, `git`, `wget`, `build-essential`, `ffmpeg`, `libsox-*`, `libsndfile1`, `libopus0`, `libffi-dev`.
|
|
103
|
+
2. **Python 3.12** — installed via deadsnakes PPA, venv created at `/opt/venv`, binaries symlinked to `/usr/local/bin`.
|
|
104
|
+
3. **pip** — upgraded to latest `pip`, `setuptools`, `wheel`.
|
|
105
|
+
4. **`qwen-asr[vllm]`** — installs the full Qwen3-ASR stack with vLLM extras (transformers, vllm, torch, accelerate, librosa, soundfile, etc. with pinned versions).
|
|
106
|
+
5. **flash-attention** — prebuilt wheel for CUDA 12.8 + PyTorch 2.9 + Python 3.12.
|
|
107
|
+
6. **Project files** — `ASR_Concurrent_Stream/` copied to `/app/ASR_Concurrent_Stream/`.
|
|
108
|
+
7. **Model download** — `Qwen/Qwen3-ASR-1.7B` downloaded from HuggingFace to `/app/models/Qwen3-ASR-1.7B`.
|
|
109
|
+
8. **Launcher patch** — `run_asr_grpc_server.sh` is updated:
|
|
110
|
+
- `PORT` → `8002`
|
|
111
|
+
- `MODEL_PATH` → `/app/models/Qwen3-ASR-1.7B`
|
|
112
|
+
- `QWEN3_ASR_PATH` → site-packages `qwen_asr` path
|
|
113
|
+
|
|
114
|
+
## Build & Run
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
# Build
|
|
118
|
+
docker build -t qwen3-asr-server:latest .
|
|
119
|
+
|
|
120
|
+
# Run (map host port 8002 to container port 8002)
|
|
121
|
+
docker run -d \
|
|
122
|
+
--name qwen3-asr-server \
|
|
123
|
+
--gpus all \
|
|
124
|
+
-p 8002:8002 \
|
|
125
|
+
--shm-size=16g \
|
|
126
|
+
--ulimit memlock=-1 \
|
|
127
|
+
--restart unless-stopped \
|
|
128
|
+
qwen3-asr-server:latest
|
|
129
|
+
|
|
130
|
+
# View logs
|
|
131
|
+
docker logs -f qwen3-asr-server
|
|
132
|
+
|
|
133
|
+
# Stop & remove
|
|
134
|
+
docker stop qwen3-asr-server && docker rm qwen3-asr-server
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
## Server Configuration
|
|
138
|
+
|
|
139
|
+
Hard-coded defaults in `server/run_asr_grpc_server.sh`:
|
|
140
|
+
|
|
141
|
+
| Parameter | Value |
|
|
142
|
+
|-----------|-------|
|
|
143
|
+
| Port | `8002` |
|
|
144
|
+
| Model | `/app/models/Qwen3-ASR-1.7B` |
|
|
145
|
+
| GPU memory utilization | `0.35` |
|
|
146
|
+
| Max model length | `4096` |
|
|
147
|
+
| Max concurrent streams | `10` |
|
|
148
|
+
| Max batch size | `8` |
|
|
149
|
+
| Batch timeout | `50` ms |
|
|
150
|
+
| Worker threads | `4` |
|
|
151
|
+
|
|
152
|
+
## Project Structure (in image)
|
|
153
|
+
|
|
154
|
+
```
|
|
155
|
+
/app/
|
|
156
|
+
├── ASR_Concurrent_Stream/
|
|
157
|
+
│ ├── proto/ # gRPC protobuf definitions
|
|
158
|
+
│ ├── server/
|
|
159
|
+
│ │ ├── asr_grpc_server.py
|
|
160
|
+
│ │ ├── _launch_server.py
|
|
161
|
+
│ │ ├── stream_manager.py
|
|
162
|
+
│ │ ├── inference_coordinator.py
|
|
163
|
+
│ │ └── run_asr_grpc_server.sh
|
|
164
|
+
│ └── client/
|
|
165
|
+
│ └── example_client.py
|
|
166
|
+
└── models/
|
|
167
|
+
└── Qwen3-ASR-1.7B/ # Pre-downloaded model
|
|
168
|
+
```
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
client/__init__.py,sha256=ldoglfYZaoF7WwiByvSc8X6NjjlAM70W3EsawatR-JI,31
|
|
2
|
+
client/example_client.py,sha256=inb2GxiOoXIjRUIi0mSSaGlY0Lsg-ny2tkxdVoA7RZ0,13843
|
|
3
|
+
proto/__init__.py,sha256=GtM4B-bLPBm0dQD4jZlYzr2pV5DaEXO9yG7rPsXiFf8,998
|
|
4
|
+
proto/asr_pb2.py,sha256=VGvE-zDdnjxIRusPRYgQ8N0e60McpXT0MP3m5ewLt-Y,5214
|
|
5
|
+
proto/asr_pb2_grpc.py,sha256=orH4z9ZF8RxyS7JcKQSVtgp71TyQvoiYtGmyixv80pE,10737
|
|
6
|
+
server/__init__.py,sha256=JE4vCAhKQtmX8OMBIVywu5TyIX8onYOc62Rqqu5JC8k,31
|
|
7
|
+
server/_launch_server.py,sha256=CmiuWOtv13yWuvhYSHuDL_GagftSZIsnvYACtI5gxKo,2644
|
|
8
|
+
server/asr_grpc_server.py,sha256=IvdA0aWYeLg2jargCh1h_-JNwpDoLRso3rs3bLQDlSo,16792
|
|
9
|
+
server/asr_model.py,sha256=IOebywqCpgwTd0HAtrb-PmeHqNDbhzNTVRzYPzAL6_E,7116
|
|
10
|
+
server/asr_processor.py,sha256=bMUkqUW3AXM58viaDyVwsHUuZZtCeiLHXU8eSmI23lM,1311
|
|
11
|
+
server/asr_utils.py,sha256=QS8o6KmclSN7nDYYwBFHWSIkxNeNjYbEtUHCvcyFLwY,3933
|
|
12
|
+
server/inference_coordinator.py,sha256=TrYEaTUZ7TVyPD_EKtwvSUWKOXqqnGOv56q1VxorCBo,17361
|
|
13
|
+
server/stream_manager.py,sha256=xqNGCAB1OcaQClLWsszAZZpurEk-G461fPDxfrbROLw,11218
|
|
14
|
+
asr_concurrent_stream-1.0.0.dist-info/METADATA,sha256=FKI8SMX2JR56oLA01xExVutRLAK4dVxMibTQ439G9Mk,5100
|
|
15
|
+
asr_concurrent_stream-1.0.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
16
|
+
asr_concurrent_stream-1.0.0.dist-info/entry_points.txt,sha256=MI9vgH2fQSENeYdsQx6K8jjspqZOpkRssUqdT2ZMAcU,120
|
|
17
|
+
asr_concurrent_stream-1.0.0.dist-info/top_level.txt,sha256=-uYCXuMnc3babcVgmckBI2woPaBoxPWEF1QfKOV2l3U,20
|
|
18
|
+
asr_concurrent_stream-1.0.0.dist-info/RECORD,,
|
client/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# ASR Concurrent Stream Client
|
client/example_client.py
ADDED
|
@@ -0,0 +1,409 @@
|
|
|
1
|
+
"""
|
|
2
|
+
gRPC Client for ASR Concurrent Stream
|
|
3
|
+
|
|
4
|
+
Multi-segment continuous streaming with Silero VAD:
|
|
5
|
+
- Loads one or more audio files
|
|
6
|
+
- Segments audio using Silero VAD (configurable silence threshold)
|
|
7
|
+
- Streams each segment to a dedicated gRPC stream (new stream_id per segment)
|
|
8
|
+
- Prints partial transcripts in real-time as they arrive from the server
|
|
9
|
+
- Optional concurrent mode: all segments processed in parallel across streams
|
|
10
|
+
|
|
11
|
+
Usage examples:
|
|
12
|
+
# Single file, sequential segments
|
|
13
|
+
python example_client.py --audio my_audio.wav
|
|
14
|
+
|
|
15
|
+
# Single file, custom silence gap and language
|
|
16
|
+
python example_client.py --audio my_audio.wav --silence-ms 600 --language Italian
|
|
17
|
+
|
|
18
|
+
# Multiple files, all segments in parallel
|
|
19
|
+
python example_client.py --audio a.wav b.wav --concurrent
|
|
20
|
+
|
|
21
|
+
# Faster-than-realtime (e.g. 4x) for stress testing
|
|
22
|
+
python example_client.py --audio my_audio.wav --realtime 4.0
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
import asyncio
|
|
26
|
+
import logging
|
|
27
|
+
import math
|
|
28
|
+
import os
|
|
29
|
+
import sys
|
|
30
|
+
import time
|
|
31
|
+
import uuid
|
|
32
|
+
from typing import List, Optional, Tuple
|
|
33
|
+
|
|
34
|
+
import grpc
|
|
35
|
+
import grpc.aio
|
|
36
|
+
import numpy as np
|
|
37
|
+
|
|
38
|
+
CLIENT_DIR = os.path.dirname(os.path.abspath(__file__))
|
|
39
|
+
BASE_DIR = os.path.dirname(CLIENT_DIR)
|
|
40
|
+
sys.path.insert(0, BASE_DIR)
|
|
41
|
+
|
|
42
|
+
from proto import asr_pb2
|
|
43
|
+
from proto import asr_pb2_grpc
|
|
44
|
+
|
|
45
|
+
logger = logging.getLogger(__name__)
|
|
46
|
+
|
|
47
|
+
SAMPLE_RATE = 16000
|
|
48
|
+
CHUNK_SIZE = 1600 # 100 ms per chunk at 16 kHz
|
|
49
|
+
VAD_WINDOW = 512 # Silero VAD internal window size at 16 kHz
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# ---------------------------------------------------------------------------
|
|
53
|
+
# Audio helpers
|
|
54
|
+
# ---------------------------------------------------------------------------
|
|
55
|
+
|
|
56
|
+
def load_audio(audio_file: str) -> np.ndarray:
|
|
57
|
+
"""Load audio as float32 mono at 16 kHz."""
|
|
58
|
+
try:
|
|
59
|
+
import soundfile as sf
|
|
60
|
+
audio, sr = sf.read(audio_file, dtype="float32")
|
|
61
|
+
except Exception:
|
|
62
|
+
import librosa
|
|
63
|
+
return librosa.load(audio_file, sr=SAMPLE_RATE, mono=True)[0]
|
|
64
|
+
|
|
65
|
+
if audio.ndim > 1:
|
|
66
|
+
audio = audio.mean(axis=1)
|
|
67
|
+
if sr != SAMPLE_RATE:
|
|
68
|
+
import librosa
|
|
69
|
+
audio = librosa.resample(audio, orig_sr=sr, target_sr=SAMPLE_RATE)
|
|
70
|
+
return audio.astype(np.float32)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def segment_audio_with_vad(
|
|
74
|
+
audio: np.ndarray,
|
|
75
|
+
silence_duration_ms: int = 800,
|
|
76
|
+
speech_threshold: float = 0.5,
|
|
77
|
+
pad_ms: int = 150,
|
|
78
|
+
) -> List[np.ndarray]:
|
|
79
|
+
"""
|
|
80
|
+
Segment float32 audio into speech chunks using Silero VAD.
|
|
81
|
+
|
|
82
|
+
Each returned segment starts `pad_ms` before the detected speech onset
|
|
83
|
+
(for model context) and ends at the detected speech offset. If no
|
|
84
|
+
segments are found (all silence, or threshold too strict), the entire
|
|
85
|
+
audio is returned as a single segment.
|
|
86
|
+
"""
|
|
87
|
+
import torch
|
|
88
|
+
|
|
89
|
+
model, utils = torch.hub.load(
|
|
90
|
+
repo_or_dir="snakers4/silero-vad",
|
|
91
|
+
model="silero_vad",
|
|
92
|
+
force_reload=False,
|
|
93
|
+
verbose=False,
|
|
94
|
+
)
|
|
95
|
+
_, _, _, VADIterator, _ = utils
|
|
96
|
+
vad_iter = VADIterator(
|
|
97
|
+
model,
|
|
98
|
+
sampling_rate=SAMPLE_RATE,
|
|
99
|
+
threshold=speech_threshold,
|
|
100
|
+
min_silence_duration_ms=silence_duration_ms,
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
pad_samples = int(pad_ms * SAMPLE_RATE / 1000)
|
|
104
|
+
segments: List[Tuple[int, int]] = []
|
|
105
|
+
current_start: Optional[int] = None
|
|
106
|
+
|
|
107
|
+
for i in range(0, len(audio), VAD_WINDOW):
|
|
108
|
+
window = audio[i : i + VAD_WINDOW].astype(np.float32)
|
|
109
|
+
if len(window) < VAD_WINDOW:
|
|
110
|
+
window = np.pad(window, (0, VAD_WINDOW - len(window)))
|
|
111
|
+
|
|
112
|
+
ev = vad_iter(torch.from_numpy(window), return_seconds=False)
|
|
113
|
+
if ev is not None:
|
|
114
|
+
if "start" in ev:
|
|
115
|
+
current_start = max(0, i - pad_samples)
|
|
116
|
+
if "end" in ev and current_start is not None:
|
|
117
|
+
segments.append((current_start, min(i + VAD_WINDOW, len(audio))))
|
|
118
|
+
current_start = None
|
|
119
|
+
|
|
120
|
+
# Audio ends while still in speech
|
|
121
|
+
if current_start is not None:
|
|
122
|
+
segments.append((current_start, len(audio)))
|
|
123
|
+
|
|
124
|
+
if not segments:
|
|
125
|
+
logger.warning("VAD found no speech; treating entire audio as one segment")
|
|
126
|
+
return [audio]
|
|
127
|
+
|
|
128
|
+
return [audio[s:e] for s, e in segments]
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
# ---------------------------------------------------------------------------
|
|
132
|
+
# Health check
|
|
133
|
+
# ---------------------------------------------------------------------------
|
|
134
|
+
|
|
135
|
+
async def health_check(
|
|
136
|
+
host: str = "localhost",
|
|
137
|
+
port: int = 50051,
|
|
138
|
+
) -> bool:
|
|
139
|
+
"""Call the server HealthCheck RPC and return True if serving."""
|
|
140
|
+
channel = grpc.aio.insecure_channel(f"{host}:{port}")
|
|
141
|
+
try:
|
|
142
|
+
stub = asr_pb2_grpc.ASRServiceStub(channel)
|
|
143
|
+
response = await stub.HealthCheck(asr_pb2.HealthCheckRequest())
|
|
144
|
+
print(f"Health check: serving={response.serving}, version={response.version}")
|
|
145
|
+
return response.serving
|
|
146
|
+
except grpc.aio.AioRpcError as e:
|
|
147
|
+
print(f"Health check failed: {e.code()} – {e.details()}")
|
|
148
|
+
return False
|
|
149
|
+
finally:
|
|
150
|
+
await channel.close()
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
# ---------------------------------------------------------------------------
|
|
154
|
+
# Per-segment streaming
|
|
155
|
+
# ---------------------------------------------------------------------------
|
|
156
|
+
|
|
157
|
+
async def stream_segment(
|
|
158
|
+
stub: asr_pb2_grpc.ASRServiceStub,
|
|
159
|
+
audio_segment: np.ndarray,
|
|
160
|
+
segment_idx: int,
|
|
161
|
+
language: str = "Italian",
|
|
162
|
+
realtime_factor: float = 1.0,
|
|
163
|
+
label: str = "",
|
|
164
|
+
) -> Tuple[int, str, float]:
|
|
165
|
+
"""
|
|
166
|
+
Stream a single audio segment and return (segment_idx, final_transcript, total_ms).
|
|
167
|
+
|
|
168
|
+
A new stream_id is created for every call. Partial transcripts are
|
|
169
|
+
printed to stdout as they arrive. The function returns once the server
|
|
170
|
+
sends is_final=True (triggered when the last chunk's is_final flag
|
|
171
|
+
flows through the coordinator and calls finish_streaming_transcribe).
|
|
172
|
+
"""
|
|
173
|
+
stream_id = str(uuid.uuid4())
|
|
174
|
+
prefix = label or f"seg {segment_idx}"
|
|
175
|
+
sleep_per_chunk = (
|
|
176
|
+
CHUNK_SIZE / SAMPLE_RATE / max(realtime_factor, 1e-3)
|
|
177
|
+
if realtime_factor > 0
|
|
178
|
+
else 0.0
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
# Initialise the stream (sets language and chunking config on the server)
|
|
182
|
+
config = asr_pb2.StreamConfig(
|
|
183
|
+
chunk_size_sec=0.5, unfixed_chunk_num=2, unfixed_token_num=5
|
|
184
|
+
)
|
|
185
|
+
start_resp = await stub.StartStream(
|
|
186
|
+
asr_pb2.StreamStartRequest(
|
|
187
|
+
stream_id=stream_id, language=language, config=config
|
|
188
|
+
)
|
|
189
|
+
)
|
|
190
|
+
if not start_resp.success:
|
|
191
|
+
logger.error(f"[{prefix}] StartStream failed: {start_resp.error}")
|
|
192
|
+
return segment_idx, "", 0.0
|
|
193
|
+
|
|
194
|
+
audio_int16 = (audio_segment * 32767).astype(np.int16)
|
|
195
|
+
num_chunks = math.ceil(len(audio_int16) / CHUNK_SIZE)
|
|
196
|
+
t_start = time.perf_counter()
|
|
197
|
+
final_text = ""
|
|
198
|
+
|
|
199
|
+
async def chunk_generator():
|
|
200
|
+
for i in range(num_chunks):
|
|
201
|
+
s = i * CHUNK_SIZE
|
|
202
|
+
e = min(s + CHUNK_SIZE, len(audio_int16))
|
|
203
|
+
is_final = i == num_chunks - 1
|
|
204
|
+
yield asr_pb2.AudioChunkRequest(
|
|
205
|
+
stream_id=stream_id,
|
|
206
|
+
chunk_id=i,
|
|
207
|
+
audio_data=audio_int16[s:e].tobytes(),
|
|
208
|
+
sample_rate=SAMPLE_RATE,
|
|
209
|
+
is_final=is_final,
|
|
210
|
+
)
|
|
211
|
+
if is_final:
|
|
212
|
+
return
|
|
213
|
+
if sleep_per_chunk > 0:
|
|
214
|
+
await asyncio.sleep(sleep_per_chunk)
|
|
215
|
+
|
|
216
|
+
try:
|
|
217
|
+
async for response in stub.StreamTranscribe(chunk_generator()):
|
|
218
|
+
text = response.partial_transcript
|
|
219
|
+
status = "FINAL " if response.is_final else "partial"
|
|
220
|
+
display = (text[:72] + "\u2026") if len(text) > 73 else text
|
|
221
|
+
print(
|
|
222
|
+
f"\r[{prefix}] [{status}] {display:<76} | {response.latency_ms}ms",
|
|
223
|
+
end="",
|
|
224
|
+
flush=True,
|
|
225
|
+
)
|
|
226
|
+
if response.is_final:
|
|
227
|
+
final_text = text
|
|
228
|
+
print() # newline after final
|
|
229
|
+
break
|
|
230
|
+
except grpc.aio.AioRpcError as e:
|
|
231
|
+
logger.error(f"[{prefix}] gRPC error: {e.code()} \u2013 {e.details()}")
|
|
232
|
+
print()
|
|
233
|
+
except Exception as e:
|
|
234
|
+
logger.error(f"[{prefix}] Error: {e}", exc_info=True)
|
|
235
|
+
print()
|
|
236
|
+
|
|
237
|
+
total_ms = (time.perf_counter() - t_start) * 1000
|
|
238
|
+
return segment_idx, final_text, total_ms
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
# ---------------------------------------------------------------------------
|
|
242
|
+
# Main orchestrator
|
|
243
|
+
# ---------------------------------------------------------------------------
|
|
244
|
+
|
|
245
|
+
async def run_continuous_streaming(
|
|
246
|
+
audio_files: List[str],
|
|
247
|
+
host: str = "localhost",
|
|
248
|
+
port: int = 50051,
|
|
249
|
+
language: str = "Italian",
|
|
250
|
+
silence_duration_ms: int = 800,
|
|
251
|
+
concurrent: bool = False,
|
|
252
|
+
realtime_factor: float = 1.0,
|
|
253
|
+
) -> None:
|
|
254
|
+
"""
|
|
255
|
+
Load one or more audio files, segment them with Silero VAD, and stream
|
|
256
|
+
each segment to a dedicated gRPC stream on the ASR server.
|
|
257
|
+
|
|
258
|
+
concurrent=False — segments are processed one after another (default)
|
|
259
|
+
concurrent=True — all segments fire simultaneously (multi-stream load test)
|
|
260
|
+
"""
|
|
261
|
+
print("\n" + "=" * 65)
|
|
262
|
+
print("ASR Concurrent Stream \u2014 Continuous Streaming with Silero VAD")
|
|
263
|
+
print("=" * 65)
|
|
264
|
+
print(f" Files: {len(audio_files)}")
|
|
265
|
+
print(f" Language: {language}")
|
|
266
|
+
print(f" Silence gap: {silence_duration_ms} ms")
|
|
267
|
+
print(f" Mode: {'Concurrent (parallel streams)' if concurrent else 'Sequential'}")
|
|
268
|
+
print(f" Realtime: {realtime_factor}\u00d7")
|
|
269
|
+
print("=" * 65 + "\n")
|
|
270
|
+
|
|
271
|
+
channel = grpc.aio.insecure_channel(
|
|
272
|
+
f"{host}:{port}",
|
|
273
|
+
options=[("grpc.max_receive_message_length", 10 * 1024 * 1024)],
|
|
274
|
+
)
|
|
275
|
+
stub = asr_pb2_grpc.ASRServiceStub(channel)
|
|
276
|
+
logger.info(f"Connected to {host}:{port}")
|
|
277
|
+
|
|
278
|
+
try:
|
|
279
|
+
# Build task list: (label, segment_audio, global_idx)
|
|
280
|
+
tasks: List[Tuple[str, np.ndarray, int]] = []
|
|
281
|
+
global_idx = 0
|
|
282
|
+
|
|
283
|
+
for f_idx, audio_file in enumerate(audio_files):
|
|
284
|
+
print(f"\u25ba Loading {os.path.basename(audio_file)}")
|
|
285
|
+
audio = load_audio(audio_file)
|
|
286
|
+
duration_s = len(audio) / SAMPLE_RATE
|
|
287
|
+
print(f" Duration : {duration_s:.2f}s")
|
|
288
|
+
|
|
289
|
+
print(f" Segmenting with Silero VAD (silence={silence_duration_ms}ms)...")
|
|
290
|
+
segments = segment_audio_with_vad(
|
|
291
|
+
audio, silence_duration_ms=silence_duration_ms
|
|
292
|
+
)
|
|
293
|
+
print(f" Segments : {len(segments)}\n")
|
|
294
|
+
|
|
295
|
+
for s_idx, seg in enumerate(segments):
|
|
296
|
+
seg_dur = len(seg) / SAMPLE_RATE
|
|
297
|
+
lbl = (
|
|
298
|
+
f"f{f_idx}/s{s_idx}" if len(audio_files) > 1 else f"seg {s_idx}"
|
|
299
|
+
)
|
|
300
|
+
print(f" [{lbl}] {seg_dur:.2f}s")
|
|
301
|
+
tasks.append((lbl, seg, global_idx))
|
|
302
|
+
global_idx += 1
|
|
303
|
+
|
|
304
|
+
if not tasks:
|
|
305
|
+
print("No segments to process.")
|
|
306
|
+
return
|
|
307
|
+
|
|
308
|
+
print(f"\nStreaming {len(tasks)} segment(s)...\n")
|
|
309
|
+
|
|
310
|
+
if concurrent:
|
|
311
|
+
coros = [
|
|
312
|
+
stream_segment(stub, seg, idx, language, realtime_factor, lbl)
|
|
313
|
+
for lbl, seg, idx in tasks
|
|
314
|
+
]
|
|
315
|
+
results = await asyncio.gather(*coros, return_exceptions=True)
|
|
316
|
+
else:
|
|
317
|
+
results = []
|
|
318
|
+
for lbl, seg, idx in tasks:
|
|
319
|
+
r = await stream_segment(stub, seg, idx, language, realtime_factor, lbl)
|
|
320
|
+
results.append(r)
|
|
321
|
+
|
|
322
|
+
print("\n" + "=" * 65)
|
|
323
|
+
print("TRANSCRIPTION COMPLETE")
|
|
324
|
+
print("=" * 65)
|
|
325
|
+
for res in results:
|
|
326
|
+
if isinstance(res, Exception):
|
|
327
|
+
print(f" ERROR: {res}")
|
|
328
|
+
else:
|
|
329
|
+
seg_idx, text, lat = res
|
|
330
|
+
print(f" [{seg_idx:>2}] {(text or '(empty)'):<60} ({lat:.0f} ms)")
|
|
331
|
+
print("=" * 65)
|
|
332
|
+
|
|
333
|
+
finally:
|
|
334
|
+
await channel.close()
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
# ---------------------------------------------------------------------------
|
|
338
|
+
# CLI entry point
|
|
339
|
+
# ---------------------------------------------------------------------------
|
|
340
|
+
|
|
341
|
+
def main() -> None:
|
|
342
|
+
import argparse
|
|
343
|
+
|
|
344
|
+
parser = argparse.ArgumentParser(
|
|
345
|
+
description="ASR Concurrent Stream Client \u2014 multi-segment Silero VAD streaming",
|
|
346
|
+
formatter_class=argparse.ArgumentDefaultsHelpFormatter,
|
|
347
|
+
)
|
|
348
|
+
parser.add_argument(
|
|
349
|
+
"--audio",
|
|
350
|
+
nargs="+",
|
|
351
|
+
default=["/mnt/c/Users/pierc/Qwen3_asr/audios/Audio 1-2.wav"],
|
|
352
|
+
metavar="FILE",
|
|
353
|
+
help="Path(s) to audio file(s) to transcribe",
|
|
354
|
+
)
|
|
355
|
+
parser.add_argument("--host", default="localhost", help="gRPC server host")
|
|
356
|
+
parser.add_argument("--port", type=int, default=50051, help="gRPC server port")
|
|
357
|
+
parser.add_argument(
|
|
358
|
+
"--language",
|
|
359
|
+
default="Italian",
|
|
360
|
+
help="Transcription language (e.g. Italian, English)",
|
|
361
|
+
)
|
|
362
|
+
parser.add_argument(
|
|
363
|
+
"--silence-ms",
|
|
364
|
+
type=int,
|
|
365
|
+
default=400,
|
|
366
|
+
help="Minimum silence (ms) between speech segments for VAD segmentation",
|
|
367
|
+
)
|
|
368
|
+
parser.add_argument(
|
|
369
|
+
"--concurrent",
|
|
370
|
+
action="store_true",
|
|
371
|
+
help="Process all segments in parallel (multi-stream concurrency test)",
|
|
372
|
+
)
|
|
373
|
+
parser.add_argument(
|
|
374
|
+
"--realtime",
|
|
375
|
+
type=float,
|
|
376
|
+
default=2.0,
|
|
377
|
+
help=(
|
|
378
|
+
"Realtime streaming factor: 1.0 = real-time, "
|
|
379
|
+
"2.0 = 2x faster, 0.0 = no delay (stress test)"
|
|
380
|
+
),
|
|
381
|
+
)
|
|
382
|
+
parser.add_argument(
|
|
383
|
+
"--health",
|
|
384
|
+
action="store_true",
|
|
385
|
+
help="Run a health check against the server and exit",
|
|
386
|
+
)
|
|
387
|
+
args = parser.parse_args()
|
|
388
|
+
|
|
389
|
+
logging.basicConfig(level=logging.INFO, format="%(levelname)s - %(message)s")
|
|
390
|
+
|
|
391
|
+
if args.health:
|
|
392
|
+
healthy = asyncio.run(health_check(host=args.host, port=args.port))
|
|
393
|
+
sys.exit(0 if healthy else 1)
|
|
394
|
+
|
|
395
|
+
asyncio.run(
|
|
396
|
+
run_continuous_streaming(
|
|
397
|
+
audio_files=args.audio,
|
|
398
|
+
host=args.host,
|
|
399
|
+
port=args.port,
|
|
400
|
+
language=args.language,
|
|
401
|
+
silence_duration_ms=args.silence_ms,
|
|
402
|
+
concurrent=args.concurrent,
|
|
403
|
+
realtime_factor=args.realtime,
|
|
404
|
+
)
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
if __name__ == "__main__":
|
|
409
|
+
main()
|
proto/__init__.py
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Generated protocol buffer code for ASR Concurrent Stream
|
|
3
|
+
|
|
4
|
+
This module is generated from proto/asr.proto
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from proto.asr_pb2 import (
|
|
8
|
+
HealthCheckRequest,
|
|
9
|
+
HealthCheckResponse,
|
|
10
|
+
StreamStartRequest,
|
|
11
|
+
StreamStartResponse,
|
|
12
|
+
StreamFinalizeRequest,
|
|
13
|
+
StreamFinalizeResponse,
|
|
14
|
+
StreamStatusRequest,
|
|
15
|
+
StreamStatusResponse,
|
|
16
|
+
StreamConfig,
|
|
17
|
+
AudioChunkRequest,
|
|
18
|
+
TranscriptionResponse,
|
|
19
|
+
DESCRIPTOR,
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
from proto.asr_pb2_grpc import (
|
|
23
|
+
ASRServiceStub,
|
|
24
|
+
ASRServiceServicer,
|
|
25
|
+
ASRService,
|
|
26
|
+
add_ASRServiceServicer_to_server,
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
__all__ = [
|
|
30
|
+
"HealthCheckRequest",
|
|
31
|
+
"HealthCheckResponse",
|
|
32
|
+
"StreamStartRequest",
|
|
33
|
+
"StreamStartResponse",
|
|
34
|
+
"StreamFinalizeRequest",
|
|
35
|
+
"StreamFinalizeResponse",
|
|
36
|
+
"StreamStatusRequest",
|
|
37
|
+
"StreamStatusResponse",
|
|
38
|
+
"StreamConfig",
|
|
39
|
+
"AudioChunkRequest",
|
|
40
|
+
"TranscriptionResponse",
|
|
41
|
+
"ASRServiceStub",
|
|
42
|
+
"ASRServiceServicer",
|
|
43
|
+
"ASRService",
|
|
44
|
+
"add_ASRServiceServicer_to_server",
|
|
45
|
+
"DESCRIPTOR",
|
|
46
|
+
]
|