speechrevolutions 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- speechrevolutions-0.2.0/LICENSE +21 -0
- speechrevolutions-0.2.0/PKG-INFO +193 -0
- speechrevolutions-0.2.0/README.md +157 -0
- speechrevolutions-0.2.0/pyproject.toml +49 -0
- speechrevolutions-0.2.0/setup.cfg +4 -0
- speechrevolutions-0.2.0/src/speechrevolutions/__init__.py +72 -0
- speechrevolutions-0.2.0/src/speechrevolutions/_audio.py +63 -0
- speechrevolutions-0.2.0/src/speechrevolutions/_config.py +129 -0
- speechrevolutions-0.2.0/src/speechrevolutions/_progress.py +148 -0
- speechrevolutions-0.2.0/src/speechrevolutions/_sse.py +43 -0
- speechrevolutions-0.2.0/src/speechrevolutions/_upload.py +107 -0
- speechrevolutions-0.2.0/src/speechrevolutions/async_client.py +749 -0
- speechrevolutions-0.2.0/src/speechrevolutions/client.py +795 -0
- speechrevolutions-0.2.0/src/speechrevolutions/exceptions.py +74 -0
- speechrevolutions-0.2.0/src/speechrevolutions/models.py +165 -0
- speechrevolutions-0.2.0/src/speechrevolutions/py.typed +0 -0
- speechrevolutions-0.2.0/src/speechrevolutions/transcript.py +420 -0
- speechrevolutions-0.2.0/src/speechrevolutions.egg-info/PKG-INFO +193 -0
- speechrevolutions-0.2.0/src/speechrevolutions.egg-info/SOURCES.txt +30 -0
- speechrevolutions-0.2.0/src/speechrevolutions.egg-info/dependency_links.txt +1 -0
- speechrevolutions-0.2.0/src/speechrevolutions.egg-info/requires.txt +10 -0
- speechrevolutions-0.2.0/src/speechrevolutions.egg-info/top_level.txt +1 -0
- speechrevolutions-0.2.0/tests/test_async_client.py +199 -0
- speechrevolutions-0.2.0/tests/test_async_retry_policy.py +126 -0
- speechrevolutions-0.2.0/tests/test_jobs_api.py +204 -0
- speechrevolutions-0.2.0/tests/test_progress_and_polling.py +153 -0
- speechrevolutions-0.2.0/tests/test_retry_policy.py +168 -0
- speechrevolutions-0.2.0/tests/test_timing_behaviour.py +119 -0
- speechrevolutions-0.2.0/tests/test_transcript_and_config.py +206 -0
- speechrevolutions-0.2.0/tests/test_transcript_exports.py +195 -0
- speechrevolutions-0.2.0/tests/test_upload_flows.py +197 -0
- speechrevolutions-0.2.0/tests/test_webhooks.py +305 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Speech Revolutions
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: speechrevolutions
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Official Python SDK for the Speech Revolutions speech-to-text API
|
|
5
|
+
Author: Speech Revolutions
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://speechrevolutions.com
|
|
8
|
+
Project-URL: Documentation, https://docs.speechrevolutions.com
|
|
9
|
+
Project-URL: Repository, https://github.com/SpeechRevolutions/python-sdk
|
|
10
|
+
Project-URL: Issues, https://github.com/SpeechRevolutions/python-sdk/issues
|
|
11
|
+
Keywords: speech-to-text,stt,transcription,whisper,asr
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
22
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
23
|
+
Classifier: Typing :: Typed
|
|
24
|
+
Requires-Python: >=3.9
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
License-File: LICENSE
|
|
27
|
+
Requires-Dist: requests>=2.28
|
|
28
|
+
Requires-Dist: httpx>=0.27
|
|
29
|
+
Provides-Extra: progress
|
|
30
|
+
Requires-Dist: tqdm>=4.65; extra == "progress"
|
|
31
|
+
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
33
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == "dev"
|
|
34
|
+
Requires-Dist: tqdm>=4.65; extra == "dev"
|
|
35
|
+
Dynamic: license-file
|
|
36
|
+
|
|
37
|
+
# Speech Revolutions — Python SDK
|
|
38
|
+
|
|
39
|
+
Official Python client for the [Speech Revolutions](https://speechrevolutions.com) speech-to-text API.
|
|
40
|
+
|
|
41
|
+
## Install
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
pip install speechrevolutions
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
For console progress bars, install the optional `tqdm` extra:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
pip install "speechrevolutions[progress]"
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
## Quick start
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
from speechrevolutions import SpeechRevolutions
|
|
57
|
+
|
|
58
|
+
client = SpeechRevolutions() # reads SPEECHREVOLUTIONS_API_KEY or STT_API_KEY
|
|
59
|
+
result = client.transcribe("meeting.mp3", speaker_labels=True)
|
|
60
|
+
print(result.text)
|
|
61
|
+
|
|
62
|
+
for u in result.utterances:
|
|
63
|
+
print(f"Speaker {u.speaker}: {u.text}")
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
### Async
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from speechrevolutions import AsyncSpeechRevolutions
|
|
70
|
+
|
|
71
|
+
async with AsyncSpeechRevolutions() as client:
|
|
72
|
+
result = await client.transcribe("meeting.mp3", speaker_labels=True)
|
|
73
|
+
print(result.text)
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
### From a URL (Deepgram-style)
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
result = client.transcribe_url("https://example.com/audio.mp3")
|
|
80
|
+
# or
|
|
81
|
+
result = client.transcribe("https://example.com/audio.mp3")
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
### Options as kwargs or config object
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
# kwargs (ElevenLabs / Deepgram style). `diarize` is an alias for `speaker_labels`.
|
|
88
|
+
result = client.transcribe("a.mp3", diarize=True, output_type="json")
|
|
89
|
+
|
|
90
|
+
# config object (AssemblyAI style)
|
|
91
|
+
from speechrevolutions import TranscribeOptions
|
|
92
|
+
result = client.transcribe("a.mp3", options=TranscribeOptions(speaker_labels=True))
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Defaults: `output_type="json"`, `word_timestamps`, `speaker_labels`, `nltk` all
|
|
96
|
+
`True`, `tier="standard"`, `custom_vocabulary=None`.
|
|
97
|
+
|
|
98
|
+
### Live progress
|
|
99
|
+
|
|
100
|
+
Unlike AssemblyAI/Deepgram (which give no percentage for pre-recorded audio),
|
|
101
|
+
you get real-time progress — for **both** the file upload and the
|
|
102
|
+
transcription — as a console bar, a callback, or both.
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
# 1. Console bars (uses tqdm if installed: pip install "speechrevolutions[progress]")
|
|
106
|
+
# Shows an "Uploading" byte bar, then a "Transcribing" bar.
|
|
107
|
+
result = client.transcribe("meeting.mp3", progress=True)
|
|
108
|
+
|
|
109
|
+
# 2. Programmatic — read event.percent (0–100) to drive your own UI / API
|
|
110
|
+
def on_progress(event): # transcription
|
|
111
|
+
print(event.percent, event.step) # e.g. 42.0 "transcribe"
|
|
112
|
+
|
|
113
|
+
def on_upload(event): # upload (event.step == "upload")
|
|
114
|
+
print("upload", event.percent)
|
|
115
|
+
|
|
116
|
+
result = client.transcribe(
|
|
117
|
+
"meeting.mp3",
|
|
118
|
+
on_progress=on_progress,
|
|
119
|
+
on_upload_progress=on_upload,
|
|
120
|
+
)
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
`progress=True` and the callbacks compose — the bars render *and* your callbacks
|
|
124
|
+
still fire for every event.
|
|
125
|
+
|
|
126
|
+
## Result shape
|
|
127
|
+
|
|
128
|
+
Default `output_type` is `json`. The SDK parses it into a transcript-first object:
|
|
129
|
+
|
|
130
|
+
| Field | Like |
|
|
131
|
+
|-------|------|
|
|
132
|
+
| `result.text` | AssemblyAI / ElevenLabs |
|
|
133
|
+
| `result.transcript` | Deepgram alias |
|
|
134
|
+
| `result.words` | word + start/end/speaker |
|
|
135
|
+
| `result.utterances` | AssemblyAI speaker turns |
|
|
136
|
+
| `result.to_deepgram()` | Deepgram-shaped dict |
|
|
137
|
+
| `result.to_dict()` | normalized JSON |
|
|
138
|
+
| `result.content` / `result.save()` | raw bytes / file |
|
|
139
|
+
|
|
140
|
+
```python
|
|
141
|
+
dg = result.to_deepgram()
|
|
142
|
+
print(dg["results"]["channels"][0]["alternatives"][0]["transcript"])
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
## Webhooks & retrieving results later
|
|
146
|
+
|
|
147
|
+
`submit()` uploads and enqueues a job and returns its id **without waiting** —
|
|
148
|
+
ideal for batch/background work. Collect the result later via a webhook
|
|
149
|
+
(`callback_url`, a signed POST — verify `X-SR-Signature: sha256=…` against the
|
|
150
|
+
raw bytes) or by polling:
|
|
151
|
+
|
|
152
|
+
```python
|
|
153
|
+
job_id = client.submit("meeting.mp3") # returns immediately, no waiting
|
|
154
|
+
# ...or notify a webhook instead of polling:
|
|
155
|
+
client.transcribe("meeting.mp3", callback_url="https://you.example.com/hook")
|
|
156
|
+
|
|
157
|
+
status = client.get_job_status(job_id) # .status: processing|completed|failed
|
|
158
|
+
if status.is_completed:
|
|
159
|
+
result = client.get_transcript(job_id) # downloads + parses
|
|
160
|
+
page = client.list_jobs(limit=50) # {"jobs": [...], "next_before": ...}
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
See `examples/` for a full submit/poll/webhook walkthrough.
|
|
164
|
+
|
|
165
|
+
## Robustness
|
|
166
|
+
|
|
167
|
+
`SpeechRevolutions(max_retries=3, retry_backoff=0.5, proxies={"https": "..."})`.
|
|
168
|
+
Transient 429/5xx/network errors are retried (honoring `Retry-After`). Errors are
|
|
169
|
+
typed (`RateLimitError`, `AuthenticationError`, …) and carry `.status_code` and
|
|
170
|
+
`.request_id` for correlating with support.
|
|
171
|
+
|
|
172
|
+
## Auth
|
|
173
|
+
|
|
174
|
+
```bash
|
|
175
|
+
export SPEECHREVOLUTIONS_API_KEY=stt_...
|
|
176
|
+
# or
|
|
177
|
+
export STT_API_KEY=stt_...
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
Or `SpeechRevolutions(api_key="stt_...")`.
|
|
181
|
+
|
|
182
|
+
## Other languages
|
|
183
|
+
|
|
184
|
+
Speech Revolutions also publishes SDKs for
|
|
185
|
+
[JavaScript/TypeScript](https://github.com/SpeechRevolutions/node-sdk),
|
|
186
|
+
[Go](https://github.com/SpeechRevolutions/speechrevolutions-go), and
|
|
187
|
+
[C#/.NET](https://github.com/SpeechRevolutions/csharp-sdk) — see
|
|
188
|
+
[docs.speechrevolutions.com](https://docs.speechrevolutions.com) for a
|
|
189
|
+
cross-language feature comparison.
|
|
190
|
+
|
|
191
|
+
## License
|
|
192
|
+
|
|
193
|
+
MIT
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
# Speech Revolutions — Python SDK
|
|
2
|
+
|
|
3
|
+
Official Python client for the [Speech Revolutions](https://speechrevolutions.com) speech-to-text API.
|
|
4
|
+
|
|
5
|
+
## Install
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install speechrevolutions
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
For console progress bars, install the optional `tqdm` extra:
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
pip install "speechrevolutions[progress]"
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
## Quick start
|
|
18
|
+
|
|
19
|
+
```python
|
|
20
|
+
from speechrevolutions import SpeechRevolutions
|
|
21
|
+
|
|
22
|
+
client = SpeechRevolutions() # reads SPEECHREVOLUTIONS_API_KEY or STT_API_KEY
|
|
23
|
+
result = client.transcribe("meeting.mp3", speaker_labels=True)
|
|
24
|
+
print(result.text)
|
|
25
|
+
|
|
26
|
+
for u in result.utterances:
|
|
27
|
+
print(f"Speaker {u.speaker}: {u.text}")
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
### Async
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from speechrevolutions import AsyncSpeechRevolutions
|
|
34
|
+
|
|
35
|
+
async with AsyncSpeechRevolutions() as client:
|
|
36
|
+
result = await client.transcribe("meeting.mp3", speaker_labels=True)
|
|
37
|
+
print(result.text)
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
### From a URL (Deepgram-style)
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
result = client.transcribe_url("https://example.com/audio.mp3")
|
|
44
|
+
# or
|
|
45
|
+
result = client.transcribe("https://example.com/audio.mp3")
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
### Options as kwargs or config object
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
# kwargs (ElevenLabs / Deepgram style). `diarize` is an alias for `speaker_labels`.
|
|
52
|
+
result = client.transcribe("a.mp3", diarize=True, output_type="json")
|
|
53
|
+
|
|
54
|
+
# config object (AssemblyAI style)
|
|
55
|
+
from speechrevolutions import TranscribeOptions
|
|
56
|
+
result = client.transcribe("a.mp3", options=TranscribeOptions(speaker_labels=True))
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Defaults: `output_type="json"`, `word_timestamps`, `speaker_labels`, `nltk` all
|
|
60
|
+
`True`, `tier="standard"`, `custom_vocabulary=None`.
|
|
61
|
+
|
|
62
|
+
### Live progress
|
|
63
|
+
|
|
64
|
+
Unlike AssemblyAI/Deepgram (which give no percentage for pre-recorded audio),
|
|
65
|
+
you get real-time progress — for **both** the file upload and the
|
|
66
|
+
transcription — as a console bar, a callback, or both.
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
# 1. Console bars (uses tqdm if installed: pip install "speechrevolutions[progress]")
|
|
70
|
+
# Shows an "Uploading" byte bar, then a "Transcribing" bar.
|
|
71
|
+
result = client.transcribe("meeting.mp3", progress=True)
|
|
72
|
+
|
|
73
|
+
# 2. Programmatic — read event.percent (0–100) to drive your own UI / API
|
|
74
|
+
def on_progress(event): # transcription
|
|
75
|
+
print(event.percent, event.step) # e.g. 42.0 "transcribe"
|
|
76
|
+
|
|
77
|
+
def on_upload(event): # upload (event.step == "upload")
|
|
78
|
+
print("upload", event.percent)
|
|
79
|
+
|
|
80
|
+
result = client.transcribe(
|
|
81
|
+
"meeting.mp3",
|
|
82
|
+
on_progress=on_progress,
|
|
83
|
+
on_upload_progress=on_upload,
|
|
84
|
+
)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
`progress=True` and the callbacks compose — the bars render *and* your callbacks
|
|
88
|
+
still fire for every event.
|
|
89
|
+
|
|
90
|
+
## Result shape
|
|
91
|
+
|
|
92
|
+
Default `output_type` is `json`. The SDK parses it into a transcript-first object:
|
|
93
|
+
|
|
94
|
+
| Field | Like |
|
|
95
|
+
|-------|------|
|
|
96
|
+
| `result.text` | AssemblyAI / ElevenLabs |
|
|
97
|
+
| `result.transcript` | Deepgram alias |
|
|
98
|
+
| `result.words` | word + start/end/speaker |
|
|
99
|
+
| `result.utterances` | AssemblyAI speaker turns |
|
|
100
|
+
| `result.to_deepgram()` | Deepgram-shaped dict |
|
|
101
|
+
| `result.to_dict()` | normalized JSON |
|
|
102
|
+
| `result.content` / `result.save()` | raw bytes / file |
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
dg = result.to_deepgram()
|
|
106
|
+
print(dg["results"]["channels"][0]["alternatives"][0]["transcript"])
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
## Webhooks & retrieving results later
|
|
110
|
+
|
|
111
|
+
`submit()` uploads and enqueues a job and returns its id **without waiting** —
|
|
112
|
+
ideal for batch/background work. Collect the result later via a webhook
|
|
113
|
+
(`callback_url`, a signed POST — verify `X-SR-Signature: sha256=…` against the
|
|
114
|
+
raw bytes) or by polling:
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
job_id = client.submit("meeting.mp3") # returns immediately, no waiting
|
|
118
|
+
# ...or notify a webhook instead of polling:
|
|
119
|
+
client.transcribe("meeting.mp3", callback_url="https://you.example.com/hook")
|
|
120
|
+
|
|
121
|
+
status = client.get_job_status(job_id) # .status: processing|completed|failed
|
|
122
|
+
if status.is_completed:
|
|
123
|
+
result = client.get_transcript(job_id) # downloads + parses
|
|
124
|
+
page = client.list_jobs(limit=50) # {"jobs": [...], "next_before": ...}
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
See `examples/` for a full submit/poll/webhook walkthrough.
|
|
128
|
+
|
|
129
|
+
## Robustness
|
|
130
|
+
|
|
131
|
+
`SpeechRevolutions(max_retries=3, retry_backoff=0.5, proxies={"https": "..."})`.
|
|
132
|
+
Transient 429/5xx/network errors are retried (honoring `Retry-After`). Errors are
|
|
133
|
+
typed (`RateLimitError`, `AuthenticationError`, …) and carry `.status_code` and
|
|
134
|
+
`.request_id` for correlating with support.
|
|
135
|
+
|
|
136
|
+
## Auth
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
export SPEECHREVOLUTIONS_API_KEY=stt_...
|
|
140
|
+
# or
|
|
141
|
+
export STT_API_KEY=stt_...
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
Or `SpeechRevolutions(api_key="stt_...")`.
|
|
145
|
+
|
|
146
|
+
## Other languages
|
|
147
|
+
|
|
148
|
+
Speech Revolutions also publishes SDKs for
|
|
149
|
+
[JavaScript/TypeScript](https://github.com/SpeechRevolutions/node-sdk),
|
|
150
|
+
[Go](https://github.com/SpeechRevolutions/speechrevolutions-go), and
|
|
151
|
+
[C#/.NET](https://github.com/SpeechRevolutions/csharp-sdk) — see
|
|
152
|
+
[docs.speechrevolutions.com](https://docs.speechrevolutions.com) for a
|
|
153
|
+
cross-language feature comparison.
|
|
154
|
+
|
|
155
|
+
## License
|
|
156
|
+
|
|
157
|
+
MIT
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "speechrevolutions"
|
|
7
|
+
version = "0.2.0"
|
|
8
|
+
description = "Official Python SDK for the Speech Revolutions speech-to-text API"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Speech Revolutions" }]
|
|
13
|
+
keywords = ["speech-to-text", "stt", "transcription", "whisper", "asr"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.9",
|
|
20
|
+
"Programming Language :: Python :: 3.10",
|
|
21
|
+
"Programming Language :: Python :: 3.11",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Programming Language :: Python :: 3.13",
|
|
24
|
+
"Topic :: Multimedia :: Sound/Audio :: Speech",
|
|
25
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
26
|
+
"Typing :: Typed",
|
|
27
|
+
]
|
|
28
|
+
dependencies = [
|
|
29
|
+
"requests>=2.28",
|
|
30
|
+
"httpx>=0.27",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
[project.optional-dependencies]
|
|
34
|
+
progress = ["tqdm>=4.65"]
|
|
35
|
+
dev = ["pytest>=7.0", "pytest-asyncio>=0.23", "tqdm>=4.65"]
|
|
36
|
+
|
|
37
|
+
[project.urls]
|
|
38
|
+
Homepage = "https://speechrevolutions.com"
|
|
39
|
+
Documentation = "https://docs.speechrevolutions.com"
|
|
40
|
+
Repository = "https://github.com/SpeechRevolutions/python-sdk"
|
|
41
|
+
Issues = "https://github.com/SpeechRevolutions/python-sdk/issues"
|
|
42
|
+
|
|
43
|
+
[tool.setuptools.packages.find]
|
|
44
|
+
where = ["src"]
|
|
45
|
+
|
|
46
|
+
# Without this marker the wheel ships no types, and every consumer of a fully
|
|
47
|
+
# annotated SDK gets `Any` back from their type checker. PEP 561.
|
|
48
|
+
[tool.setuptools.package-data]
|
|
49
|
+
speechrevolutions = ["py.typed"]
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
"""Speech Revolutions STT Python SDK."""
|
|
2
|
+
|
|
3
|
+
from speechrevolutions.client import SpeechRevolutions, SpeechRevolutionsClient, STTClient
|
|
4
|
+
from speechrevolutions.exceptions import (
|
|
5
|
+
APIError,
|
|
6
|
+
AuthenticationError,
|
|
7
|
+
JobFailedError,
|
|
8
|
+
JobNotFoundError,
|
|
9
|
+
RateLimitError,
|
|
10
|
+
STTError,
|
|
11
|
+
TimeoutError,
|
|
12
|
+
UploadError,
|
|
13
|
+
)
|
|
14
|
+
from speechrevolutions.models import (
|
|
15
|
+
JobStatus,
|
|
16
|
+
OutputType,
|
|
17
|
+
ProcessingTier,
|
|
18
|
+
ProgressEvent,
|
|
19
|
+
TranscribeOptions,
|
|
20
|
+
UploadJob,
|
|
21
|
+
)
|
|
22
|
+
from speechrevolutions.transcript import LanguageSegment, Transcript, Utterance, Word
|
|
23
|
+
|
|
24
|
+
__all__ = [
|
|
25
|
+
# Clients
|
|
26
|
+
"STTClient",
|
|
27
|
+
"SpeechRevolutions",
|
|
28
|
+
"SpeechRevolutionsClient",
|
|
29
|
+
"AsyncSTTClient",
|
|
30
|
+
"AsyncSpeechRevolutions",
|
|
31
|
+
"AsyncSpeechRevolutionsClient",
|
|
32
|
+
# Models
|
|
33
|
+
"OutputType",
|
|
34
|
+
"ProcessingTier",
|
|
35
|
+
"TranscribeOptions",
|
|
36
|
+
"ProgressEvent",
|
|
37
|
+
"UploadJob",
|
|
38
|
+
"JobStatus",
|
|
39
|
+
"Transcript",
|
|
40
|
+
"Word",
|
|
41
|
+
"Utterance",
|
|
42
|
+
"LanguageSegment",
|
|
43
|
+
# Errors
|
|
44
|
+
"STTError",
|
|
45
|
+
"AuthenticationError",
|
|
46
|
+
"RateLimitError",
|
|
47
|
+
"JobNotFoundError",
|
|
48
|
+
"JobFailedError",
|
|
49
|
+
"UploadError",
|
|
50
|
+
"TimeoutError",
|
|
51
|
+
"APIError",
|
|
52
|
+
]
|
|
53
|
+
|
|
54
|
+
__version__ = "0.2.0"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def __getattr__(name: str):
|
|
58
|
+
"""Lazy-load async client so sync users don't need httpx until they import it."""
|
|
59
|
+
if name in {"AsyncSTTClient", "AsyncSpeechRevolutions", "AsyncSpeechRevolutionsClient"}:
|
|
60
|
+
from speechrevolutions.async_client import (
|
|
61
|
+
AsyncSpeechRevolutions,
|
|
62
|
+
AsyncSpeechRevolutionsClient,
|
|
63
|
+
AsyncSTTClient,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
mapping = {
|
|
67
|
+
"AsyncSTTClient": AsyncSTTClient,
|
|
68
|
+
"AsyncSpeechRevolutions": AsyncSpeechRevolutions,
|
|
69
|
+
"AsyncSpeechRevolutionsClient": AsyncSpeechRevolutionsClient,
|
|
70
|
+
}
|
|
71
|
+
return mapping[name]
|
|
72
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Audio input helpers: local path, bytes, file objects, or remote URL."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import BinaryIO
|
|
7
|
+
from urllib.parse import urlparse
|
|
8
|
+
|
|
9
|
+
import requests
|
|
10
|
+
|
|
11
|
+
from speechrevolutions.exceptions import APIError
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def is_url(value: str) -> bool:
|
|
15
|
+
try:
|
|
16
|
+
parsed = urlparse(value)
|
|
17
|
+
return parsed.scheme in ("http", "https") and bool(parsed.netloc)
|
|
18
|
+
except Exception:
|
|
19
|
+
return False
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def read_audio(
|
|
23
|
+
audio: str | Path | bytes | BinaryIO,
|
|
24
|
+
*,
|
|
25
|
+
session: requests.Session | None = None,
|
|
26
|
+
timeout: float = 120.0,
|
|
27
|
+
) -> tuple[bytes, int]:
|
|
28
|
+
"""
|
|
29
|
+
Normalize audio input to bytes.
|
|
30
|
+
|
|
31
|
+
Accepts a local filesystem path, http(s) URL, raw bytes, or binary file object.
|
|
32
|
+
"""
|
|
33
|
+
if isinstance(audio, (str, Path)):
|
|
34
|
+
value = str(audio)
|
|
35
|
+
if is_url(value):
|
|
36
|
+
sess = session or requests.Session()
|
|
37
|
+
try:
|
|
38
|
+
resp = sess.get(value, timeout=timeout)
|
|
39
|
+
except requests.exceptions.RequestException as exc:
|
|
40
|
+
raise APIError(f"Failed to download audio URL: {exc}") from exc
|
|
41
|
+
if resp.status_code != 200:
|
|
42
|
+
raise APIError(
|
|
43
|
+
f"Failed to download audio URL (HTTP {resp.status_code})",
|
|
44
|
+
status_code=resp.status_code,
|
|
45
|
+
body=resp.text[:300],
|
|
46
|
+
)
|
|
47
|
+
data = resp.content
|
|
48
|
+
else:
|
|
49
|
+
path = Path(value)
|
|
50
|
+
if not path.exists():
|
|
51
|
+
raise FileNotFoundError(f"Audio file not found: {path}")
|
|
52
|
+
data = path.read_bytes()
|
|
53
|
+
elif isinstance(audio, bytes):
|
|
54
|
+
data = audio
|
|
55
|
+
elif hasattr(audio, "read"):
|
|
56
|
+
raw = audio.read()
|
|
57
|
+
data = raw if isinstance(raw, bytes) else raw.encode("utf-8")
|
|
58
|
+
else:
|
|
59
|
+
raise TypeError(f"Unsupported audio type: {type(audio)!r}")
|
|
60
|
+
|
|
61
|
+
if not data:
|
|
62
|
+
raise ValueError("Audio is empty")
|
|
63
|
+
return data, len(data)
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
"""Shared SDK constants and API-key resolution."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
|
|
7
|
+
from speechrevolutions.exceptions import AuthenticationError
|
|
8
|
+
|
|
9
|
+
DEFAULT_BASE_URL = "https://api.speechrevolutions.com"
|
|
10
|
+
ENV_API_KEY_NAMES = ("SPEECHREVOLUTIONS_API_KEY", "STT_API_KEY")
|
|
11
|
+
|
|
12
|
+
#: Overrides the API host. Symmetric with the key: if a caller can supply an
|
|
13
|
+
#: API key from the environment, they can point it at an environment too.
|
|
14
|
+
#: Needed for staging, for an egress proxy or gateway, and for running any
|
|
15
|
+
#: published example (the cookbook) against something that is not production.
|
|
16
|
+
ENV_BASE_URL_NAMES = ("SPEECHREVOLUTIONS_BASE_URL", "STT_BASE_URL")
|
|
17
|
+
|
|
18
|
+
UPLOAD_PROGRESS_INTERVAL = 10
|
|
19
|
+
UPLOAD_MAX_ATTEMPTS = 4
|
|
20
|
+
UPLOAD_BASE_DELAY = 1.0
|
|
21
|
+
|
|
22
|
+
SSE_MAX_RECONNECTS = 10
|
|
23
|
+
SSE_RECONNECT_DELAY = 3.0
|
|
24
|
+
|
|
25
|
+
#: How many times to retry a stream endpoint that answered with a NON-2xx
|
|
26
|
+
#: status, as opposed to one whose connection dropped.
|
|
27
|
+
#:
|
|
28
|
+
#: The two failures look the same to the reconnect loop and are not the same
|
|
29
|
+
#: thing. A dropped connection is transient — the server was streaming a moment
|
|
30
|
+
#: ago and will be again — so ten attempts on a 3s timer is right. A non-2xx
|
|
31
|
+
#: status is a refusal: a proxy, load balancer or corporate egress that does not
|
|
32
|
+
#: pass `text/event-stream` answers every attempt identically, forever. Retrying
|
|
33
|
+
#: that ten times costs 30 seconds before the client falls back to polling, on
|
|
34
|
+
#: EVERY job, which is longer than the median job takes to transcribe.
|
|
35
|
+
#:
|
|
36
|
+
#: Two attempts, so a genuinely transient 502/503 still gets a second chance,
|
|
37
|
+
#: then fall back to polling — which works and is only marginally slower.
|
|
38
|
+
SSE_MAX_STATUS_REFUSALS = 2
|
|
39
|
+
|
|
40
|
+
POLL_INTERVAL = 5.0
|
|
41
|
+
|
|
42
|
+
# Transient-failure retry policy for JSON API requests (not uploads/SSE, which
|
|
43
|
+
# have their own retry loops). Overridable per-client via STTClient(...).
|
|
44
|
+
DEFAULT_MAX_RETRIES = 3
|
|
45
|
+
DEFAULT_RETRY_BACKOFF = 0.5 # seconds; exponential (0.5, 1.0, 2.0, …), capped
|
|
46
|
+
RETRY_BACKOFF_MAX = 30.0
|
|
47
|
+
RETRY_STATUS_CODES = frozenset({429, 500, 502, 503, 504})
|
|
48
|
+
|
|
49
|
+
# Endpoints that CREATE a job, and so are not safe to blindly retry.
|
|
50
|
+
#
|
|
51
|
+
# A job is created the moment the server handles one of these; the response
|
|
52
|
+
# carrying the job_id back is what can be lost. Retrying after the request may
|
|
53
|
+
# have arrived creates a SECOND job for the same audio — two transcripts, two
|
|
54
|
+
# charges — and the caller never learns about the orphan. The API has no
|
|
55
|
+
# idempotency key, so the only safe rule is to retry these solely when the
|
|
56
|
+
# request provably never reached the server: a connect timeout (no connection
|
|
57
|
+
# was ever established) or a 429 (explicitly refused before any work).
|
|
58
|
+
#
|
|
59
|
+
# Every other endpoint either reads, or acts on a job_id the caller already
|
|
60
|
+
# holds, and stays fully retryable.
|
|
61
|
+
JOB_CREATING_PATHS = frozenset(
|
|
62
|
+
{
|
|
63
|
+
"/api/v1/upload",
|
|
64
|
+
"/api/v1/upload/multipart/create",
|
|
65
|
+
}
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def creates_job(path: str) -> bool:
|
|
70
|
+
"""True if `path` creates a job, and so must not be blindly retried."""
|
|
71
|
+
return path.split("?", 1)[0].rstrip("/") in JOB_CREATING_PATHS
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
# Response headers checked (case-insensitively) for a correlation id.
|
|
75
|
+
REQUEST_ID_HEADERS = ("x-request-id", "x-amzn-requestid", "cf-ray")
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def extract_request_id(headers: object) -> str | None:
|
|
79
|
+
"""Return the first present request-id header value, or None."""
|
|
80
|
+
get = getattr(headers, "get", None)
|
|
81
|
+
if get is None:
|
|
82
|
+
return None
|
|
83
|
+
for name in REQUEST_ID_HEADERS:
|
|
84
|
+
value = get(name)
|
|
85
|
+
if value:
|
|
86
|
+
return str(value)
|
|
87
|
+
return None
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def parse_retry_after(value: str | None) -> float | None:
|
|
91
|
+
"""Parse a Retry-After header (delta-seconds form) into seconds."""
|
|
92
|
+
if not value:
|
|
93
|
+
return None
|
|
94
|
+
try:
|
|
95
|
+
return max(0.0, float(value))
|
|
96
|
+
except (TypeError, ValueError):
|
|
97
|
+
return None # HTTP-date form is not honored; caller falls back to backoff
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def resolve_base_url(base_url: str | None) -> str:
|
|
101
|
+
"""An explicit argument wins, then the environment, then production."""
|
|
102
|
+
if base_url:
|
|
103
|
+
return base_url.rstrip("/")
|
|
104
|
+
for name in ENV_BASE_URL_NAMES:
|
|
105
|
+
value = os.environ.get(name)
|
|
106
|
+
if value:
|
|
107
|
+
return value.rstrip("/")
|
|
108
|
+
return DEFAULT_BASE_URL
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def resolve_api_key(api_key: str | None) -> str:
|
|
112
|
+
if api_key:
|
|
113
|
+
return api_key
|
|
114
|
+
for name in ENV_API_KEY_NAMES:
|
|
115
|
+
value = os.environ.get(name)
|
|
116
|
+
if value:
|
|
117
|
+
return value
|
|
118
|
+
raise AuthenticationError(
|
|
119
|
+
"api_key is required (pass api_key=... or set "
|
|
120
|
+
"SPEECHREVOLUTIONS_API_KEY / STT_API_KEY)"
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
# Identifies the SDK to the platform, which makes a client-side problem findable
|
|
125
|
+
# in our edge logs without the caller reproducing it. It is also insurance: the
|
|
126
|
+
# edge answers a request with NO User-Agent with a bare 403, which is how the C#
|
|
127
|
+
# client turned out to be unable to reach production at all while passing every
|
|
128
|
+
# test that pointed at a local mock.
|
|
129
|
+
USER_AGENT = "speechrevolutions-python/0.2.0"
|