speechrevolutions 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. speechrevolutions-0.2.0/LICENSE +21 -0
  2. speechrevolutions-0.2.0/PKG-INFO +193 -0
  3. speechrevolutions-0.2.0/README.md +157 -0
  4. speechrevolutions-0.2.0/pyproject.toml +49 -0
  5. speechrevolutions-0.2.0/setup.cfg +4 -0
  6. speechrevolutions-0.2.0/src/speechrevolutions/__init__.py +72 -0
  7. speechrevolutions-0.2.0/src/speechrevolutions/_audio.py +63 -0
  8. speechrevolutions-0.2.0/src/speechrevolutions/_config.py +129 -0
  9. speechrevolutions-0.2.0/src/speechrevolutions/_progress.py +148 -0
  10. speechrevolutions-0.2.0/src/speechrevolutions/_sse.py +43 -0
  11. speechrevolutions-0.2.0/src/speechrevolutions/_upload.py +107 -0
  12. speechrevolutions-0.2.0/src/speechrevolutions/async_client.py +749 -0
  13. speechrevolutions-0.2.0/src/speechrevolutions/client.py +795 -0
  14. speechrevolutions-0.2.0/src/speechrevolutions/exceptions.py +74 -0
  15. speechrevolutions-0.2.0/src/speechrevolutions/models.py +165 -0
  16. speechrevolutions-0.2.0/src/speechrevolutions/py.typed +0 -0
  17. speechrevolutions-0.2.0/src/speechrevolutions/transcript.py +420 -0
  18. speechrevolutions-0.2.0/src/speechrevolutions.egg-info/PKG-INFO +193 -0
  19. speechrevolutions-0.2.0/src/speechrevolutions.egg-info/SOURCES.txt +30 -0
  20. speechrevolutions-0.2.0/src/speechrevolutions.egg-info/dependency_links.txt +1 -0
  21. speechrevolutions-0.2.0/src/speechrevolutions.egg-info/requires.txt +10 -0
  22. speechrevolutions-0.2.0/src/speechrevolutions.egg-info/top_level.txt +1 -0
  23. speechrevolutions-0.2.0/tests/test_async_client.py +199 -0
  24. speechrevolutions-0.2.0/tests/test_async_retry_policy.py +126 -0
  25. speechrevolutions-0.2.0/tests/test_jobs_api.py +204 -0
  26. speechrevolutions-0.2.0/tests/test_progress_and_polling.py +153 -0
  27. speechrevolutions-0.2.0/tests/test_retry_policy.py +168 -0
  28. speechrevolutions-0.2.0/tests/test_timing_behaviour.py +119 -0
  29. speechrevolutions-0.2.0/tests/test_transcript_and_config.py +206 -0
  30. speechrevolutions-0.2.0/tests/test_transcript_exports.py +195 -0
  31. speechrevolutions-0.2.0/tests/test_upload_flows.py +197 -0
  32. speechrevolutions-0.2.0/tests/test_webhooks.py +305 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Speech Revolutions
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,193 @@
1
+ Metadata-Version: 2.4
2
+ Name: speechrevolutions
3
+ Version: 0.2.0
4
+ Summary: Official Python SDK for the Speech Revolutions speech-to-text API
5
+ Author: Speech Revolutions
6
+ License: MIT
7
+ Project-URL: Homepage, https://speechrevolutions.com
8
+ Project-URL: Documentation, https://docs.speechrevolutions.com
9
+ Project-URL: Repository, https://github.com/SpeechRevolutions/python-sdk
10
+ Project-URL: Issues, https://github.com/SpeechRevolutions/python-sdk/issues
11
+ Keywords: speech-to-text,stt,transcription,whisper,asr
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
22
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
23
+ Classifier: Typing :: Typed
24
+ Requires-Python: >=3.9
25
+ Description-Content-Type: text/markdown
26
+ License-File: LICENSE
27
+ Requires-Dist: requests>=2.28
28
+ Requires-Dist: httpx>=0.27
29
+ Provides-Extra: progress
30
+ Requires-Dist: tqdm>=4.65; extra == "progress"
31
+ Provides-Extra: dev
32
+ Requires-Dist: pytest>=7.0; extra == "dev"
33
+ Requires-Dist: pytest-asyncio>=0.23; extra == "dev"
34
+ Requires-Dist: tqdm>=4.65; extra == "dev"
35
+ Dynamic: license-file
36
+
37
+ # Speech Revolutions — Python SDK
38
+
39
+ Official Python client for the [Speech Revolutions](https://speechrevolutions.com) speech-to-text API.
40
+
41
+ ## Install
42
+
43
+ ```bash
44
+ pip install speechrevolutions
45
+ ```
46
+
47
+ For console progress bars, install the optional `tqdm` extra:
48
+
49
+ ```bash
50
+ pip install "speechrevolutions[progress]"
51
+ ```
52
+
53
+ ## Quick start
54
+
55
+ ```python
56
+ from speechrevolutions import SpeechRevolutions
57
+
58
+ client = SpeechRevolutions() # reads SPEECHREVOLUTIONS_API_KEY or STT_API_KEY
59
+ result = client.transcribe("meeting.mp3", speaker_labels=True)
60
+ print(result.text)
61
+
62
+ for u in result.utterances:
63
+ print(f"Speaker {u.speaker}: {u.text}")
64
+ ```
65
+
66
+ ### Async
67
+
68
+ ```python
69
+ from speechrevolutions import AsyncSpeechRevolutions
70
+
71
+ async with AsyncSpeechRevolutions() as client:
72
+ result = await client.transcribe("meeting.mp3", speaker_labels=True)
73
+ print(result.text)
74
+ ```
75
+
76
+ ### From a URL (Deepgram-style)
77
+
78
+ ```python
79
+ result = client.transcribe_url("https://example.com/audio.mp3")
80
+ # or
81
+ result = client.transcribe("https://example.com/audio.mp3")
82
+ ```
83
+
84
+ ### Options as kwargs or config object
85
+
86
+ ```python
87
+ # kwargs (ElevenLabs / Deepgram style). `diarize` is an alias for `speaker_labels`.
88
+ result = client.transcribe("a.mp3", diarize=True, output_type="json")
89
+
90
+ # config object (AssemblyAI style)
91
+ from speechrevolutions import TranscribeOptions
92
+ result = client.transcribe("a.mp3", options=TranscribeOptions(speaker_labels=True))
93
+ ```
94
+
95
+ Defaults: `output_type="json"`, `word_timestamps`, `speaker_labels`, `nltk` all
96
+ `True`, `tier="standard"`, `custom_vocabulary=None`.
97
+
98
+ ### Live progress
99
+
100
+ Unlike AssemblyAI/Deepgram (which give no percentage for pre-recorded audio),
101
+ you get real-time progress — for **both** the file upload and the
102
+ transcription — as a console bar, a callback, or both.
103
+
104
+ ```python
105
+ # 1. Console bars (uses tqdm if installed: pip install "speechrevolutions[progress]")
106
+ # Shows an "Uploading" byte bar, then a "Transcribing" bar.
107
+ result = client.transcribe("meeting.mp3", progress=True)
108
+
109
+ # 2. Programmatic — read event.percent (0–100) to drive your own UI / API
110
+ def on_progress(event): # transcription
111
+ print(event.percent, event.step) # e.g. 42.0 "transcribe"
112
+
113
+ def on_upload(event): # upload (event.step == "upload")
114
+ print("upload", event.percent)
115
+
116
+ result = client.transcribe(
117
+ "meeting.mp3",
118
+ on_progress=on_progress,
119
+ on_upload_progress=on_upload,
120
+ )
121
+ ```
122
+
123
+ `progress=True` and the callbacks compose — the bars render *and* your callbacks
124
+ still fire for every event.
125
+
126
+ ## Result shape
127
+
128
+ Default `output_type` is `json`. The SDK parses it into a transcript-first object:
129
+
130
+ | Field | Like |
131
+ |-------|------|
132
+ | `result.text` | AssemblyAI / ElevenLabs |
133
+ | `result.transcript` | Deepgram alias |
134
+ | `result.words` | word + start/end/speaker |
135
+ | `result.utterances` | AssemblyAI speaker turns |
136
+ | `result.to_deepgram()` | Deepgram-shaped dict |
137
+ | `result.to_dict()` | normalized JSON |
138
+ | `result.content` / `result.save()` | raw bytes / file |
139
+
140
+ ```python
141
+ dg = result.to_deepgram()
142
+ print(dg["results"]["channels"][0]["alternatives"][0]["transcript"])
143
+ ```
144
+
145
+ ## Webhooks & retrieving results later
146
+
147
+ `submit()` uploads and enqueues a job and returns its id **without waiting** —
148
+ ideal for batch/background work. Collect the result later via a webhook
149
+ (`callback_url`, a signed POST — verify `X-SR-Signature: sha256=…` against the
150
+ raw bytes) or by polling:
151
+
152
+ ```python
153
+ job_id = client.submit("meeting.mp3") # returns immediately, no waiting
154
+ # ...or notify a webhook instead of polling:
155
+ client.transcribe("meeting.mp3", callback_url="https://you.example.com/hook")
156
+
157
+ status = client.get_job_status(job_id) # .status: processing|completed|failed
158
+ if status.is_completed:
159
+ result = client.get_transcript(job_id) # downloads + parses
160
+ page = client.list_jobs(limit=50) # {"jobs": [...], "next_before": ...}
161
+ ```
162
+
163
+ See `examples/` for a full submit/poll/webhook walkthrough.
164
+
165
+ ## Robustness
166
+
167
+ `SpeechRevolutions(max_retries=3, retry_backoff=0.5, proxies={"https": "..."})`.
168
+ Transient 429/5xx/network errors are retried (honoring `Retry-After`). Errors are
169
+ typed (`RateLimitError`, `AuthenticationError`, …) and carry `.status_code` and
170
+ `.request_id` for correlating with support.
171
+
172
+ ## Auth
173
+
174
+ ```bash
175
+ export SPEECHREVOLUTIONS_API_KEY=stt_...
176
+ # or
177
+ export STT_API_KEY=stt_...
178
+ ```
179
+
180
+ Or `SpeechRevolutions(api_key="stt_...")`.
181
+
182
+ ## Other languages
183
+
184
+ Speech Revolutions also publishes SDKs for
185
+ [JavaScript/TypeScript](https://github.com/SpeechRevolutions/node-sdk),
186
+ [Go](https://github.com/SpeechRevolutions/speechrevolutions-go), and
187
+ [C#/.NET](https://github.com/SpeechRevolutions/csharp-sdk) — see
188
+ [docs.speechrevolutions.com](https://docs.speechrevolutions.com) for a
189
+ cross-language feature comparison.
190
+
191
+ ## License
192
+
193
+ MIT
@@ -0,0 +1,157 @@
1
+ # Speech Revolutions — Python SDK
2
+
3
+ Official Python client for the [Speech Revolutions](https://speechrevolutions.com) speech-to-text API.
4
+
5
+ ## Install
6
+
7
+ ```bash
8
+ pip install speechrevolutions
9
+ ```
10
+
11
+ For console progress bars, install the optional `tqdm` extra:
12
+
13
+ ```bash
14
+ pip install "speechrevolutions[progress]"
15
+ ```
16
+
17
+ ## Quick start
18
+
19
+ ```python
20
+ from speechrevolutions import SpeechRevolutions
21
+
22
+ client = SpeechRevolutions() # reads SPEECHREVOLUTIONS_API_KEY or STT_API_KEY
23
+ result = client.transcribe("meeting.mp3", speaker_labels=True)
24
+ print(result.text)
25
+
26
+ for u in result.utterances:
27
+ print(f"Speaker {u.speaker}: {u.text}")
28
+ ```
29
+
30
+ ### Async
31
+
32
+ ```python
33
+ from speechrevolutions import AsyncSpeechRevolutions
34
+
35
+ async with AsyncSpeechRevolutions() as client:
36
+ result = await client.transcribe("meeting.mp3", speaker_labels=True)
37
+ print(result.text)
38
+ ```
39
+
40
+ ### From a URL (Deepgram-style)
41
+
42
+ ```python
43
+ result = client.transcribe_url("https://example.com/audio.mp3")
44
+ # or
45
+ result = client.transcribe("https://example.com/audio.mp3")
46
+ ```
47
+
48
+ ### Options as kwargs or config object
49
+
50
+ ```python
51
+ # kwargs (ElevenLabs / Deepgram style). `diarize` is an alias for `speaker_labels`.
52
+ result = client.transcribe("a.mp3", diarize=True, output_type="json")
53
+
54
+ # config object (AssemblyAI style)
55
+ from speechrevolutions import TranscribeOptions
56
+ result = client.transcribe("a.mp3", options=TranscribeOptions(speaker_labels=True))
57
+ ```
58
+
59
+ Defaults: `output_type="json"`, `word_timestamps`, `speaker_labels`, `nltk` all
60
+ `True`, `tier="standard"`, `custom_vocabulary=None`.
61
+
62
+ ### Live progress
63
+
64
+ Unlike AssemblyAI/Deepgram (which give no percentage for pre-recorded audio),
65
+ you get real-time progress — for **both** the file upload and the
66
+ transcription — as a console bar, a callback, or both.
67
+
68
+ ```python
69
+ # 1. Console bars (uses tqdm if installed: pip install "speechrevolutions[progress]")
70
+ # Shows an "Uploading" byte bar, then a "Transcribing" bar.
71
+ result = client.transcribe("meeting.mp3", progress=True)
72
+
73
+ # 2. Programmatic — read event.percent (0–100) to drive your own UI / API
74
+ def on_progress(event): # transcription
75
+ print(event.percent, event.step) # e.g. 42.0 "transcribe"
76
+
77
+ def on_upload(event): # upload (event.step == "upload")
78
+ print("upload", event.percent)
79
+
80
+ result = client.transcribe(
81
+ "meeting.mp3",
82
+ on_progress=on_progress,
83
+ on_upload_progress=on_upload,
84
+ )
85
+ ```
86
+
87
+ `progress=True` and the callbacks compose — the bars render *and* your callbacks
88
+ still fire for every event.
89
+
90
+ ## Result shape
91
+
92
+ Default `output_type` is `json`. The SDK parses it into a transcript-first object:
93
+
94
+ | Field | Like |
95
+ |-------|------|
96
+ | `result.text` | AssemblyAI / ElevenLabs |
97
+ | `result.transcript` | Deepgram alias |
98
+ | `result.words` | word + start/end/speaker |
99
+ | `result.utterances` | AssemblyAI speaker turns |
100
+ | `result.to_deepgram()` | Deepgram-shaped dict |
101
+ | `result.to_dict()` | normalized JSON |
102
+ | `result.content` / `result.save()` | raw bytes / file |
103
+
104
+ ```python
105
+ dg = result.to_deepgram()
106
+ print(dg["results"]["channels"][0]["alternatives"][0]["transcript"])
107
+ ```
108
+
109
+ ## Webhooks & retrieving results later
110
+
111
+ `submit()` uploads and enqueues a job and returns its id **without waiting** —
112
+ ideal for batch/background work. Collect the result later via a webhook
113
+ (`callback_url`, a signed POST — verify `X-SR-Signature: sha256=…` against the
114
+ raw bytes) or by polling:
115
+
116
+ ```python
117
+ job_id = client.submit("meeting.mp3") # returns immediately, no waiting
118
+ # ...or notify a webhook instead of polling:
119
+ client.transcribe("meeting.mp3", callback_url="https://you.example.com/hook")
120
+
121
+ status = client.get_job_status(job_id) # .status: processing|completed|failed
122
+ if status.is_completed:
123
+ result = client.get_transcript(job_id) # downloads + parses
124
+ page = client.list_jobs(limit=50) # {"jobs": [...], "next_before": ...}
125
+ ```
126
+
127
+ See `examples/` for a full submit/poll/webhook walkthrough.
128
+
129
+ ## Robustness
130
+
131
+ `SpeechRevolutions(max_retries=3, retry_backoff=0.5, proxies={"https": "..."})`.
132
+ Transient 429/5xx/network errors are retried (honoring `Retry-After`). Errors are
133
+ typed (`RateLimitError`, `AuthenticationError`, …) and carry `.status_code` and
134
+ `.request_id` for correlating with support.
135
+
136
+ ## Auth
137
+
138
+ ```bash
139
+ export SPEECHREVOLUTIONS_API_KEY=stt_...
140
+ # or
141
+ export STT_API_KEY=stt_...
142
+ ```
143
+
144
+ Or `SpeechRevolutions(api_key="stt_...")`.
145
+
146
+ ## Other languages
147
+
148
+ Speech Revolutions also publishes SDKs for
149
+ [JavaScript/TypeScript](https://github.com/SpeechRevolutions/node-sdk),
150
+ [Go](https://github.com/SpeechRevolutions/speechrevolutions-go), and
151
+ [C#/.NET](https://github.com/SpeechRevolutions/csharp-sdk) — see
152
+ [docs.speechrevolutions.com](https://docs.speechrevolutions.com) for a
153
+ cross-language feature comparison.
154
+
155
+ ## License
156
+
157
+ MIT
@@ -0,0 +1,49 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "speechrevolutions"
7
+ version = "0.2.0"
8
+ description = "Official Python SDK for the Speech Revolutions speech-to-text API"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "Speech Revolutions" }]
13
+ keywords = ["speech-to-text", "stt", "transcription", "whisper", "asr"]
14
+ classifiers = [
15
+ "Development Status :: 4 - Beta",
16
+ "Intended Audience :: Developers",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.9",
20
+ "Programming Language :: Python :: 3.10",
21
+ "Programming Language :: Python :: 3.11",
22
+ "Programming Language :: Python :: 3.12",
23
+ "Programming Language :: Python :: 3.13",
24
+ "Topic :: Multimedia :: Sound/Audio :: Speech",
25
+ "Topic :: Software Development :: Libraries :: Python Modules",
26
+ "Typing :: Typed",
27
+ ]
28
+ dependencies = [
29
+ "requests>=2.28",
30
+ "httpx>=0.27",
31
+ ]
32
+
33
+ [project.optional-dependencies]
34
+ progress = ["tqdm>=4.65"]
35
+ dev = ["pytest>=7.0", "pytest-asyncio>=0.23", "tqdm>=4.65"]
36
+
37
+ [project.urls]
38
+ Homepage = "https://speechrevolutions.com"
39
+ Documentation = "https://docs.speechrevolutions.com"
40
+ Repository = "https://github.com/SpeechRevolutions/python-sdk"
41
+ Issues = "https://github.com/SpeechRevolutions/python-sdk/issues"
42
+
43
+ [tool.setuptools.packages.find]
44
+ where = ["src"]
45
+
46
+ # Without this marker the wheel ships no types, and every consumer of a fully
47
+ # annotated SDK gets `Any` back from their type checker. PEP 561.
48
+ [tool.setuptools.package-data]
49
+ speechrevolutions = ["py.typed"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,72 @@
1
+ """Speech Revolutions STT Python SDK."""
2
+
3
+ from speechrevolutions.client import SpeechRevolutions, SpeechRevolutionsClient, STTClient
4
+ from speechrevolutions.exceptions import (
5
+ APIError,
6
+ AuthenticationError,
7
+ JobFailedError,
8
+ JobNotFoundError,
9
+ RateLimitError,
10
+ STTError,
11
+ TimeoutError,
12
+ UploadError,
13
+ )
14
+ from speechrevolutions.models import (
15
+ JobStatus,
16
+ OutputType,
17
+ ProcessingTier,
18
+ ProgressEvent,
19
+ TranscribeOptions,
20
+ UploadJob,
21
+ )
22
+ from speechrevolutions.transcript import LanguageSegment, Transcript, Utterance, Word
23
+
24
+ __all__ = [
25
+ # Clients
26
+ "STTClient",
27
+ "SpeechRevolutions",
28
+ "SpeechRevolutionsClient",
29
+ "AsyncSTTClient",
30
+ "AsyncSpeechRevolutions",
31
+ "AsyncSpeechRevolutionsClient",
32
+ # Models
33
+ "OutputType",
34
+ "ProcessingTier",
35
+ "TranscribeOptions",
36
+ "ProgressEvent",
37
+ "UploadJob",
38
+ "JobStatus",
39
+ "Transcript",
40
+ "Word",
41
+ "Utterance",
42
+ "LanguageSegment",
43
+ # Errors
44
+ "STTError",
45
+ "AuthenticationError",
46
+ "RateLimitError",
47
+ "JobNotFoundError",
48
+ "JobFailedError",
49
+ "UploadError",
50
+ "TimeoutError",
51
+ "APIError",
52
+ ]
53
+
54
+ __version__ = "0.2.0"
55
+
56
+
57
+ def __getattr__(name: str):
58
+ """Lazy-load async client so sync users don't need httpx until they import it."""
59
+ if name in {"AsyncSTTClient", "AsyncSpeechRevolutions", "AsyncSpeechRevolutionsClient"}:
60
+ from speechrevolutions.async_client import (
61
+ AsyncSpeechRevolutions,
62
+ AsyncSpeechRevolutionsClient,
63
+ AsyncSTTClient,
64
+ )
65
+
66
+ mapping = {
67
+ "AsyncSTTClient": AsyncSTTClient,
68
+ "AsyncSpeechRevolutions": AsyncSpeechRevolutions,
69
+ "AsyncSpeechRevolutionsClient": AsyncSpeechRevolutionsClient,
70
+ }
71
+ return mapping[name]
72
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
@@ -0,0 +1,63 @@
1
+ """Audio input helpers: local path, bytes, file objects, or remote URL."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import BinaryIO
7
+ from urllib.parse import urlparse
8
+
9
+ import requests
10
+
11
+ from speechrevolutions.exceptions import APIError
12
+
13
+
14
+ def is_url(value: str) -> bool:
15
+ try:
16
+ parsed = urlparse(value)
17
+ return parsed.scheme in ("http", "https") and bool(parsed.netloc)
18
+ except Exception:
19
+ return False
20
+
21
+
22
+ def read_audio(
23
+ audio: str | Path | bytes | BinaryIO,
24
+ *,
25
+ session: requests.Session | None = None,
26
+ timeout: float = 120.0,
27
+ ) -> tuple[bytes, int]:
28
+ """
29
+ Normalize audio input to bytes.
30
+
31
+ Accepts a local filesystem path, http(s) URL, raw bytes, or binary file object.
32
+ """
33
+ if isinstance(audio, (str, Path)):
34
+ value = str(audio)
35
+ if is_url(value):
36
+ sess = session or requests.Session()
37
+ try:
38
+ resp = sess.get(value, timeout=timeout)
39
+ except requests.exceptions.RequestException as exc:
40
+ raise APIError(f"Failed to download audio URL: {exc}") from exc
41
+ if resp.status_code != 200:
42
+ raise APIError(
43
+ f"Failed to download audio URL (HTTP {resp.status_code})",
44
+ status_code=resp.status_code,
45
+ body=resp.text[:300],
46
+ )
47
+ data = resp.content
48
+ else:
49
+ path = Path(value)
50
+ if not path.exists():
51
+ raise FileNotFoundError(f"Audio file not found: {path}")
52
+ data = path.read_bytes()
53
+ elif isinstance(audio, bytes):
54
+ data = audio
55
+ elif hasattr(audio, "read"):
56
+ raw = audio.read()
57
+ data = raw if isinstance(raw, bytes) else raw.encode("utf-8")
58
+ else:
59
+ raise TypeError(f"Unsupported audio type: {type(audio)!r}")
60
+
61
+ if not data:
62
+ raise ValueError("Audio is empty")
63
+ return data, len(data)
@@ -0,0 +1,129 @@
1
+ """Shared SDK constants and API-key resolution."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+
7
+ from speechrevolutions.exceptions import AuthenticationError
8
+
9
+ DEFAULT_BASE_URL = "https://api.speechrevolutions.com"
10
+ ENV_API_KEY_NAMES = ("SPEECHREVOLUTIONS_API_KEY", "STT_API_KEY")
11
+
12
+ #: Overrides the API host. Symmetric with the key: if a caller can supply an
13
+ #: API key from the environment, they can point it at an environment too.
14
+ #: Needed for staging, for an egress proxy or gateway, and for running any
15
+ #: published example (the cookbook) against something that is not production.
16
+ ENV_BASE_URL_NAMES = ("SPEECHREVOLUTIONS_BASE_URL", "STT_BASE_URL")
17
+
18
+ UPLOAD_PROGRESS_INTERVAL = 10
19
+ UPLOAD_MAX_ATTEMPTS = 4
20
+ UPLOAD_BASE_DELAY = 1.0
21
+
22
+ SSE_MAX_RECONNECTS = 10
23
+ SSE_RECONNECT_DELAY = 3.0
24
+
25
+ #: How many times to retry a stream endpoint that answered with a NON-2xx
26
+ #: status, as opposed to one whose connection dropped.
27
+ #:
28
+ #: The two failures look the same to the reconnect loop and are not the same
29
+ #: thing. A dropped connection is transient — the server was streaming a moment
30
+ #: ago and will be again — so ten attempts on a 3s timer is right. A non-2xx
31
+ #: status is a refusal: a proxy, load balancer or corporate egress that does not
32
+ #: pass `text/event-stream` answers every attempt identically, forever. Retrying
33
+ #: that ten times costs 30 seconds before the client falls back to polling, on
34
+ #: EVERY job, which is longer than the median job takes to transcribe.
35
+ #:
36
+ #: Two attempts, so a genuinely transient 502/503 still gets a second chance,
37
+ #: then fall back to polling — which works and is only marginally slower.
38
+ SSE_MAX_STATUS_REFUSALS = 2
39
+
40
+ POLL_INTERVAL = 5.0
41
+
42
+ # Transient-failure retry policy for JSON API requests (not uploads/SSE, which
43
+ # have their own retry loops). Overridable per-client via STTClient(...).
44
+ DEFAULT_MAX_RETRIES = 3
45
+ DEFAULT_RETRY_BACKOFF = 0.5 # seconds; exponential (0.5, 1.0, 2.0, …), capped
46
+ RETRY_BACKOFF_MAX = 30.0
47
+ RETRY_STATUS_CODES = frozenset({429, 500, 502, 503, 504})
48
+
49
+ # Endpoints that CREATE a job, and so are not safe to blindly retry.
50
+ #
51
+ # A job is created the moment the server handles one of these; the response
52
+ # carrying the job_id back is what can be lost. Retrying after the request may
53
+ # have arrived creates a SECOND job for the same audio — two transcripts, two
54
+ # charges — and the caller never learns about the orphan. The API has no
55
+ # idempotency key, so the only safe rule is to retry these solely when the
56
+ # request provably never reached the server: a connect timeout (no connection
57
+ # was ever established) or a 429 (explicitly refused before any work).
58
+ #
59
+ # Every other endpoint either reads, or acts on a job_id the caller already
60
+ # holds, and stays fully retryable.
61
+ JOB_CREATING_PATHS = frozenset(
62
+ {
63
+ "/api/v1/upload",
64
+ "/api/v1/upload/multipart/create",
65
+ }
66
+ )
67
+
68
+
69
+ def creates_job(path: str) -> bool:
70
+ """True if `path` creates a job, and so must not be blindly retried."""
71
+ return path.split("?", 1)[0].rstrip("/") in JOB_CREATING_PATHS
72
+
73
+
74
+ # Response headers checked (case-insensitively) for a correlation id.
75
+ REQUEST_ID_HEADERS = ("x-request-id", "x-amzn-requestid", "cf-ray")
76
+
77
+
78
+ def extract_request_id(headers: object) -> str | None:
79
+ """Return the first present request-id header value, or None."""
80
+ get = getattr(headers, "get", None)
81
+ if get is None:
82
+ return None
83
+ for name in REQUEST_ID_HEADERS:
84
+ value = get(name)
85
+ if value:
86
+ return str(value)
87
+ return None
88
+
89
+
90
+ def parse_retry_after(value: str | None) -> float | None:
91
+ """Parse a Retry-After header (delta-seconds form) into seconds."""
92
+ if not value:
93
+ return None
94
+ try:
95
+ return max(0.0, float(value))
96
+ except (TypeError, ValueError):
97
+ return None # HTTP-date form is not honored; caller falls back to backoff
98
+
99
+
100
+ def resolve_base_url(base_url: str | None) -> str:
101
+ """An explicit argument wins, then the environment, then production."""
102
+ if base_url:
103
+ return base_url.rstrip("/")
104
+ for name in ENV_BASE_URL_NAMES:
105
+ value = os.environ.get(name)
106
+ if value:
107
+ return value.rstrip("/")
108
+ return DEFAULT_BASE_URL
109
+
110
+
111
+ def resolve_api_key(api_key: str | None) -> str:
112
+ if api_key:
113
+ return api_key
114
+ for name in ENV_API_KEY_NAMES:
115
+ value = os.environ.get(name)
116
+ if value:
117
+ return value
118
+ raise AuthenticationError(
119
+ "api_key is required (pass api_key=... or set "
120
+ "SPEECHREVOLUTIONS_API_KEY / STT_API_KEY)"
121
+ )
122
+
123
+
124
+ # Identifies the SDK to the platform, which makes a client-side problem findable
125
+ # in our edge logs without the caller reproducing it. It is also insurance: the
126
+ # edge answers a request with NO User-Agent with a bare 403, which is how the C#
127
+ # client turned out to be unable to reach production at all while passing every
128
+ # test that pointed at a local mock.
129
+ USER_AGENT = "speechrevolutions-python/0.2.0"