vodpipe 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vodpipe/__init__.py +7 -0
- vodpipe/__main__.py +3 -0
- vodpipe/app.py +166 -0
- vodpipe/asr.py +415 -0
- vodpipe/channels.py +146 -0
- vodpipe/chat.py +1041 -0
- vodpipe/cli.py +713 -0
- vodpipe/config.py +527 -0
- vodpipe/disk.py +136 -0
- vodpipe/exports.py +597 -0
- vodpipe/jobs.py +288 -0
- vodpipe/locks.py +258 -0
- vodpipe/media.py +1514 -0
- vodpipe/models.py +358 -0
- vodpipe/moments.py +391 -0
- vodpipe/net.py +186 -0
- vodpipe/pipeline.py +4744 -0
- vodpipe/quality.py +202 -0
- vodpipe/recorder.py +1178 -0
- vodpipe/schema.py +531 -0
- vodpipe/server.py +907 -0
- vodpipe/snapshot.py +343 -0
- vodpipe/state.py +1254 -0
- vodpipe/static/app.js +1072 -0
- vodpipe/static/favicon.ico +0 -0
- vodpipe/static/icon.png +0 -0
- vodpipe/static/icon.svg +7 -0
- vodpipe/static/index.html +113 -0
- vodpipe/static/manifest.webmanifest +14 -0
- vodpipe/static/style.css +733 -0
- vodpipe/summarize.py +272 -0
- vodpipe/transcribe.py +1184 -0
- vodpipe/transcript.py +1287 -0
- vodpipe/util.py +482 -0
- vodpipe/winapp.py +159 -0
- vodpipe-1.0.1.dist-info/METADATA +874 -0
- vodpipe-1.0.1.dist-info/RECORD +41 -0
- vodpipe-1.0.1.dist-info/WHEEL +5 -0
- vodpipe-1.0.1.dist-info/entry_points.txt +2 -0
- vodpipe-1.0.1.dist-info/licenses/LICENSE +21 -0
- vodpipe-1.0.1.dist-info/top_level.txt +1 -0
vodpipe/__init__.py
ADDED
vodpipe/__main__.py
ADDED
vodpipe/app.py
ADDED
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""Desktop host: the same local server, in its own Chromium window.
|
|
2
|
+
|
|
3
|
+
The pipeline is a loopback HTTP service because that is the control surface
|
|
4
|
+
that already exists. This module does not replace it with a second UI; it
|
|
5
|
+
opens a dedicated Chromium (or Chrome) window pointed at that server.
|
|
6
|
+
|
|
7
|
+
Edge is never used. Chromium first, then Google Chrome, then a copy dropped
|
|
8
|
+
in `vendor/chromium`. Window close is process-exit: we use a private
|
|
9
|
+
`--user-data-dir`, so the browser process is ours, and when it dies we shut
|
|
10
|
+
the pipeline down. If no Chromium-family browser is installed we fall back
|
|
11
|
+
to the ordinary dashboard (system browser + serve_forever).
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import shutil
|
|
18
|
+
import subprocess
|
|
19
|
+
import sys
|
|
20
|
+
import threading
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
from .config import APP_ROOT, Config
|
|
24
|
+
from .pipeline import Pipeline
|
|
25
|
+
from .server import serve
|
|
26
|
+
from .util import LOG, setup_logging
|
|
27
|
+
from .winapp import AUMID, set_app_user_model_id
|
|
28
|
+
|
|
29
|
+
# Chromium and Chrome only. Edge is a Chromium fork but it is the Edge
|
|
30
|
+
# browser, which is not what this window is supposed to be.
|
|
31
|
+
_CHROMIUM_HINTS = (
|
|
32
|
+
str(APP_ROOT / "vendor" / "chromium" / "chrome.exe"),
|
|
33
|
+
str(APP_ROOT / "vendor" / "chromium" / "chromium.exe"),
|
|
34
|
+
r"%LOCALAPPDATA%\VOD Pipeline\chromium\chrome.exe",
|
|
35
|
+
r"%LOCALAPPDATA%\Chromium\Application\chrome.exe",
|
|
36
|
+
r"%LOCALAPPDATA%\Chromium\Application\chromium.exe",
|
|
37
|
+
r"%ProgramFiles%\Chromium\Application\chrome.exe",
|
|
38
|
+
r"%ProgramFiles%\Chromium\Application\chromium.exe",
|
|
39
|
+
r"%ProgramFiles(x86)%\Chromium\Application\chrome.exe",
|
|
40
|
+
r"%ProgramFiles(x86)%\Chromium\Application\chromium.exe",
|
|
41
|
+
)
|
|
42
|
+
_CHROME_HINTS = (
|
|
43
|
+
r"%ProgramFiles%\Google\Chrome\Application\chrome.exe",
|
|
44
|
+
r"%ProgramFiles(x86)%\Google\Chrome\Application\chrome.exe",
|
|
45
|
+
r"%LocalAppData%\Google\Chrome\Application\chrome.exe",
|
|
46
|
+
r"%ProgramFiles%\Google\Chrome Beta\Application\chrome.exe",
|
|
47
|
+
r"%LocalAppData%\Google\Chrome SxS\Application\chrome.exe",
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def browser_candidates() -> list[str]:
|
|
52
|
+
"""Ordered Chromium-family paths. Never includes msedge."""
|
|
53
|
+
out: list[str] = []
|
|
54
|
+
seen: set[str] = set()
|
|
55
|
+
for hint in (*_CHROMIUM_HINTS, *_CHROME_HINTS):
|
|
56
|
+
expanded = os.path.expandvars(hint)
|
|
57
|
+
key = os.path.normcase(expanded)
|
|
58
|
+
if key not in seen:
|
|
59
|
+
seen.add(key)
|
|
60
|
+
out.append(expanded)
|
|
61
|
+
return out
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def find_app_browser() -> str | None:
|
|
65
|
+
for candidate in browser_candidates():
|
|
66
|
+
if Path(candidate).is_file():
|
|
67
|
+
return candidate
|
|
68
|
+
for name in ("chromium", "chromium.exe", "chrome", "chrome.exe"):
|
|
69
|
+
found = shutil.which(name)
|
|
70
|
+
if found and "msedge" not in os.path.normcase(found):
|
|
71
|
+
return found
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def run_app(config: Config, *, port: int | None = None,
|
|
76
|
+
open_window: bool = True) -> int:
|
|
77
|
+
set_app_user_model_id(AUMID)
|
|
78
|
+
log_file = APP_ROOT / "logs" / "vodpipe-app.log"
|
|
79
|
+
setup_logging(log_file=log_file)
|
|
80
|
+
pipeline = Pipeline(config)
|
|
81
|
+
httpd = None
|
|
82
|
+
browser: subprocess.Popen | None = None
|
|
83
|
+
try:
|
|
84
|
+
pipeline.start()
|
|
85
|
+
httpd = serve(pipeline, config, port=port, open_browser=False)
|
|
86
|
+
host, bound = httpd.server_address[:2]
|
|
87
|
+
url = f"http://{host}:{bound}/"
|
|
88
|
+
LOG.info("app listening on %s", url)
|
|
89
|
+
|
|
90
|
+
if open_window:
|
|
91
|
+
browser = _open_window(url)
|
|
92
|
+
if browser is None:
|
|
93
|
+
LOG.info("no Chromium/Chrome window available; opening the system browser")
|
|
94
|
+
import webbrowser
|
|
95
|
+
webbrowser.open(url)
|
|
96
|
+
httpd.serve_forever()
|
|
97
|
+
return 0
|
|
98
|
+
_wait_for(browser, httpd)
|
|
99
|
+
else:
|
|
100
|
+
httpd.serve_forever()
|
|
101
|
+
return 0
|
|
102
|
+
except KeyboardInterrupt:
|
|
103
|
+
print("\nstopping...", file=sys.stderr)
|
|
104
|
+
return 0
|
|
105
|
+
finally:
|
|
106
|
+
if browser is not None and browser.poll() is None:
|
|
107
|
+
try:
|
|
108
|
+
browser.terminate()
|
|
109
|
+
except OSError:
|
|
110
|
+
pass
|
|
111
|
+
try:
|
|
112
|
+
if httpd is not None:
|
|
113
|
+
try:
|
|
114
|
+
httpd.shutdown()
|
|
115
|
+
except Exception:
|
|
116
|
+
pass
|
|
117
|
+
httpd.server_close()
|
|
118
|
+
finally:
|
|
119
|
+
try:
|
|
120
|
+
LOG.info("finishing queued work before exit...")
|
|
121
|
+
finally:
|
|
122
|
+
pipeline.shutdown_until_stopped()
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _open_window(url: str) -> subprocess.Popen | None:
|
|
126
|
+
executable = find_app_browser()
|
|
127
|
+
if not executable:
|
|
128
|
+
return None
|
|
129
|
+
profile = APP_ROOT / ".app-profile"
|
|
130
|
+
profile.mkdir(parents=True, exist_ok=True)
|
|
131
|
+
argv = [
|
|
132
|
+
executable,
|
|
133
|
+
f"--app={url}",
|
|
134
|
+
f"--user-data-dir={profile}",
|
|
135
|
+
"--no-first-run",
|
|
136
|
+
"--no-default-browser-check",
|
|
137
|
+
"--disable-extensions",
|
|
138
|
+
"--disable-sync",
|
|
139
|
+
"--disable-features=TranslateUI,MediaRouter",
|
|
140
|
+
"--window-size=1440,900",
|
|
141
|
+
"--class=VODPipeline",
|
|
142
|
+
]
|
|
143
|
+
LOG.info("opening Chromium window via %s", executable)
|
|
144
|
+
try:
|
|
145
|
+
return subprocess.Popen(argv, stdout=subprocess.DEVNULL,
|
|
146
|
+
stderr=subprocess.DEVNULL)
|
|
147
|
+
except OSError as exc:
|
|
148
|
+
LOG.warning("could not launch %s: %s", executable, exc)
|
|
149
|
+
return None
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _wait_for(browser: subprocess.Popen, httpd) -> None:
|
|
153
|
+
"""Serve HTTP until the window process exits, then stop the server."""
|
|
154
|
+
|
|
155
|
+
def wait_browser() -> None:
|
|
156
|
+
try:
|
|
157
|
+
browser.wait()
|
|
158
|
+
finally:
|
|
159
|
+
threading.Thread(target=httpd.shutdown, daemon=True,
|
|
160
|
+
name="app-shutdown").start()
|
|
161
|
+
|
|
162
|
+
watcher = threading.Thread(target=wait_browser, name="app-window",
|
|
163
|
+
daemon=True)
|
|
164
|
+
watcher.start()
|
|
165
|
+
httpd.serve_forever()
|
|
166
|
+
watcher.join(timeout=5.0)
|
vodpipe/asr.py
ADDED
|
@@ -0,0 +1,415 @@
|
|
|
1
|
+
"""Speech recognition providers.
|
|
2
|
+
|
|
3
|
+
Deepgram is the configured engine: it returns native word-level start/end times,
|
|
4
|
+
which is the single reason this pipeline needs no forced-alignment stage and can
|
|
5
|
+
publish a chunk's transcript about a minute after the chunk closes rather than six.
|
|
6
|
+
|
|
7
|
+
The provider interface is deliberately thin so a different engine can be dropped in
|
|
8
|
+
without anything downstream noticing.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
import inspect
|
|
15
|
+
import math
|
|
16
|
+
import time
|
|
17
|
+
from email.utils import parsedate_to_datetime
|
|
18
|
+
import urllib.error
|
|
19
|
+
import urllib.parse
|
|
20
|
+
import urllib.request
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Any, Callable, Protocol
|
|
23
|
+
|
|
24
|
+
from .transcript import Word, normalise
|
|
25
|
+
from .util import LOG
|
|
26
|
+
|
|
27
|
+
DEEPGRAM_ENDPOINT = "https://api.deepgram.com/v1/listen"
|
|
28
|
+
|
|
29
|
+
# How far past the end of the submitted audio a word may *begin* before the
|
|
30
|
+
# response is treated as describing something other than what we sent.
|
|
31
|
+
#
|
|
32
|
+
# CORRECTED 2026-08-16 -- do not reinstate a hard end-time bound. An 8-hour
|
|
33
|
+
# recording of a live channel failed roughly half of all rolling passes on
|
|
34
|
+
# "deepgram word 'x' ends beyond the response audio duration", and both halves
|
|
35
|
+
# of that check were wrong:
|
|
36
|
+
#
|
|
37
|
+
# * `metadata.duration` is coarse. nova-3 reported it as a whole number of
|
|
38
|
+
# seconds for most slices (49.0, 63.0, 77.0, 93.0 ...) while the slice itself
|
|
39
|
+
# was 63.9s, so words landed "beyond" audio that was really there.
|
|
40
|
+
# * A word's *end* is an estimate, not a measurement. The last word of a passage
|
|
41
|
+
# routinely ends after the audio does, by a few milliseconds to about a second.
|
|
42
|
+
#
|
|
43
|
+
# Neither is a corruption signal. What is one is a word that *starts* after the
|
|
44
|
+
# audio ended: that describes a different, longer recording -- a mismatched or
|
|
45
|
+
# replayed response -- and no amount of end-time estimation produces it. Starts
|
|
46
|
+
# are therefore bounded and ends are clamped.
|
|
47
|
+
WORD_START_TOLERANCE = 2.0
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class TranscriptionError(RuntimeError):
|
|
51
|
+
pass
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class AuthError(TranscriptionError):
|
|
55
|
+
"""A bad or missing key. Retrying will not help."""
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class _RetryableHTTPError(TranscriptionError):
|
|
59
|
+
def __init__(self, message: str, retry_after: float | None = None) -> None:
|
|
60
|
+
super().__init__(message)
|
|
61
|
+
self.retry_after = retry_after
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class ASRProvider(Protocol):
|
|
65
|
+
def transcribe(self, audio: Path, *,
|
|
66
|
+
expected_duration: float | None = None) -> list[Word]:
|
|
67
|
+
...
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def transcribe_audio(provider: ASRProvider, audio: Path,
|
|
71
|
+
expected_duration: float) -> list[Word]:
|
|
72
|
+
"""Pass measured duration when supported, preserving simple test providers."""
|
|
73
|
+
method = provider.transcribe
|
|
74
|
+
try:
|
|
75
|
+
inspect.signature(method).bind(
|
|
76
|
+
audio, expected_duration=expected_duration)
|
|
77
|
+
except (TypeError, ValueError):
|
|
78
|
+
return method(audio)
|
|
79
|
+
return method(audio, expected_duration=expected_duration)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class DeepgramProvider:
|
|
83
|
+
def __init__(
|
|
84
|
+
self,
|
|
85
|
+
api_key: str,
|
|
86
|
+
*,
|
|
87
|
+
model: str = "nova-3",
|
|
88
|
+
language: str = "en",
|
|
89
|
+
filler_words: bool = True,
|
|
90
|
+
max_retries: int = 4,
|
|
91
|
+
timeout: float = 600.0,
|
|
92
|
+
on_response: Callable[[dict[str, Any]], None] | None = None,
|
|
93
|
+
) -> None:
|
|
94
|
+
if not api_key:
|
|
95
|
+
raise AuthError("no Deepgram API key configured")
|
|
96
|
+
self.api_key = api_key
|
|
97
|
+
self.model = model
|
|
98
|
+
self.language = language
|
|
99
|
+
self.filler_words = filler_words
|
|
100
|
+
self.max_retries = max(1, max_retries)
|
|
101
|
+
self.timeout = timeout
|
|
102
|
+
# Called with the provider's own answer, before we make anything of it.
|
|
103
|
+
# Everything else in this pipeline is derived from that answer, so
|
|
104
|
+
# keeping it is the only way to check a derivation afterwards -- or to
|
|
105
|
+
# rebuild against a future reading of the same response without paying
|
|
106
|
+
# for the audio again. Set by the transcriber when
|
|
107
|
+
# `transcription.keep_raw_responses` is on.
|
|
108
|
+
self.on_response = on_response
|
|
109
|
+
|
|
110
|
+
def _url(self) -> str:
|
|
111
|
+
params = {
|
|
112
|
+
"model": self.model,
|
|
113
|
+
"language": self.language,
|
|
114
|
+
"punctuate": "true",
|
|
115
|
+
"smart_format": "true",
|
|
116
|
+
# Fillers are transcribed by default so the transcript is verbatim:
|
|
117
|
+
# a cut made from the text then lands where the editor expects, and
|
|
118
|
+
# an "uh" can be selected and deleted on its own.
|
|
119
|
+
"filler_words": "true" if self.filler_words else "false",
|
|
120
|
+
"diarize": "false",
|
|
121
|
+
}
|
|
122
|
+
return f"{DEEPGRAM_ENDPOINT}?{urllib.parse.urlencode(params)}"
|
|
123
|
+
|
|
124
|
+
def transcribe(self, audio: Path, *,
|
|
125
|
+
expected_duration: float | None = None) -> list[Word]:
|
|
126
|
+
payload = audio.read_bytes()
|
|
127
|
+
if not payload:
|
|
128
|
+
return []
|
|
129
|
+
|
|
130
|
+
deadline = time.monotonic() + max(0.0, self.timeout)
|
|
131
|
+
last_error: Exception | None = None
|
|
132
|
+
attempts = 0
|
|
133
|
+
for attempt in range(self.max_retries):
|
|
134
|
+
if time.monotonic() >= deadline:
|
|
135
|
+
last_error = TimeoutError("Deepgram total deadline expired")
|
|
136
|
+
break
|
|
137
|
+
attempts = attempt + 1
|
|
138
|
+
try:
|
|
139
|
+
response = self._post(payload, deadline=deadline)
|
|
140
|
+
if self.on_response is not None:
|
|
141
|
+
# Before parsing, and never allowed to fail the request: an
|
|
142
|
+
# archive is a convenience and a transcription that
|
|
143
|
+
# succeeded must not be lost to a full disk.
|
|
144
|
+
try:
|
|
145
|
+
self.on_response(response)
|
|
146
|
+
except Exception: # noqa: BLE001 - archiving is optional
|
|
147
|
+
LOG.exception("could not archive the Deepgram response")
|
|
148
|
+
return parse_deepgram(
|
|
149
|
+
response, expected_duration=expected_duration)
|
|
150
|
+
except (_RetryableHTTPError, urllib.error.URLError, TimeoutError) as exc:
|
|
151
|
+
last_error = exc
|
|
152
|
+
if attempt == self.max_retries - 1:
|
|
153
|
+
break
|
|
154
|
+
delay = (exc.retry_after
|
|
155
|
+
if isinstance(exc, _RetryableHTTPError)
|
|
156
|
+
and exc.retry_after is not None
|
|
157
|
+
else min(60.0, 2.0 ** attempt * 2.0))
|
|
158
|
+
remaining = deadline - time.monotonic()
|
|
159
|
+
if remaining <= 0.0 or delay >= remaining:
|
|
160
|
+
last_error = TimeoutError(
|
|
161
|
+
"Deepgram total deadline expired before retry")
|
|
162
|
+
break
|
|
163
|
+
LOG.warning("deepgram attempt %d/%d failed (%s); retrying in %.0fs",
|
|
164
|
+
attempt + 1, self.max_retries, exc, delay)
|
|
165
|
+
time.sleep(delay)
|
|
166
|
+
|
|
167
|
+
raise TranscriptionError(f"deepgram failed after {attempts} attempt(s): "
|
|
168
|
+
f"{last_error}")
|
|
169
|
+
|
|
170
|
+
def _post(self, payload: bytes, *, deadline: float) -> dict[str, Any]:
|
|
171
|
+
request = urllib.request.Request(
|
|
172
|
+
self._url(),
|
|
173
|
+
data=payload,
|
|
174
|
+
method="POST",
|
|
175
|
+
headers={
|
|
176
|
+
"Authorization": f"Token {self.api_key}",
|
|
177
|
+
"Content-Type": "audio/flac",
|
|
178
|
+
},
|
|
179
|
+
)
|
|
180
|
+
try:
|
|
181
|
+
remaining = _remaining(deadline)
|
|
182
|
+
with urllib.request.urlopen(request, timeout=remaining) as response:
|
|
183
|
+
body = _read_response(response, deadline)
|
|
184
|
+
except urllib.error.HTTPError as exc:
|
|
185
|
+
retry_after = _retry_after(exc.headers)
|
|
186
|
+
try:
|
|
187
|
+
try:
|
|
188
|
+
detail = _read_response(exc, deadline).decode(
|
|
189
|
+
"utf-8", "replace")[:500]
|
|
190
|
+
except TimeoutError:
|
|
191
|
+
detail = "response body exceeded the total deadline"
|
|
192
|
+
finally:
|
|
193
|
+
exc.close()
|
|
194
|
+
if exc.code in (401, 403):
|
|
195
|
+
raise AuthError(f"deepgram rejected the API key ({exc.code}): {detail}")
|
|
196
|
+
if exc.code in (408, 429) or 500 <= exc.code <= 599:
|
|
197
|
+
raise _RetryableHTTPError(
|
|
198
|
+
f"deepgram HTTP {exc.code}: {detail}", retry_after)
|
|
199
|
+
raise TranscriptionError(f"deepgram HTTP {exc.code}: {detail}")
|
|
200
|
+
|
|
201
|
+
try:
|
|
202
|
+
return json.loads(body)
|
|
203
|
+
except json.JSONDecodeError as exc:
|
|
204
|
+
raise TranscriptionError(f"deepgram returned unparsable JSON: {exc}")
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _remaining(deadline: float) -> float:
|
|
208
|
+
remaining = deadline - time.monotonic()
|
|
209
|
+
if remaining <= 0.0:
|
|
210
|
+
raise TimeoutError("Deepgram total deadline expired")
|
|
211
|
+
return remaining
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _set_response_timeout(response: Any, timeout: float) -> None:
|
|
215
|
+
"""Keep each socket read bounded by the remaining total deadline."""
|
|
216
|
+
raw = getattr(getattr(response, "fp", None), "raw", None)
|
|
217
|
+
sock = getattr(raw, "_sock", None)
|
|
218
|
+
if sock is not None and hasattr(sock, "settimeout"):
|
|
219
|
+
sock.settimeout(max(0.001, timeout))
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _read_response(response: Any, deadline: float) -> bytes:
|
|
223
|
+
chunks: list[bytes] = []
|
|
224
|
+
reader = getattr(response, "read1", None)
|
|
225
|
+
if not callable(reader):
|
|
226
|
+
reader = response.read
|
|
227
|
+
while True:
|
|
228
|
+
_set_response_timeout(response, _remaining(deadline))
|
|
229
|
+
chunk = reader(64 * 1024)
|
|
230
|
+
if time.monotonic() > deadline:
|
|
231
|
+
raise TimeoutError("Deepgram total deadline expired while reading")
|
|
232
|
+
if not chunk:
|
|
233
|
+
return b"".join(chunks)
|
|
234
|
+
chunks.append(chunk)
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _retry_after(headers: Any) -> float | None:
|
|
238
|
+
if headers is None:
|
|
239
|
+
return None
|
|
240
|
+
value = headers.get("Retry-After")
|
|
241
|
+
if value is None:
|
|
242
|
+
return None
|
|
243
|
+
text = str(value).strip()
|
|
244
|
+
try:
|
|
245
|
+
return max(0.0, float(text))
|
|
246
|
+
except ValueError:
|
|
247
|
+
try:
|
|
248
|
+
parsed = parsedate_to_datetime(text)
|
|
249
|
+
return max(0.0, parsed.timestamp() - time.time())
|
|
250
|
+
except (TypeError, ValueError, OverflowError):
|
|
251
|
+
return None
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _response_number(value: Any, field: str) -> float:
|
|
255
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
256
|
+
raise TranscriptionError(f"deepgram {field} was not a number")
|
|
257
|
+
number = float(value)
|
|
258
|
+
if not math.isfinite(number):
|
|
259
|
+
raise TranscriptionError(f"deepgram {field} was not finite")
|
|
260
|
+
return number
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def parse_deepgram(response: dict[str, Any], *,
|
|
264
|
+
expected_duration: float | None = None) -> list[Word]:
|
|
265
|
+
"""Pull the word stream out of a Deepgram response.
|
|
266
|
+
|
|
267
|
+
`punctuated_word` is preferred over `word`: Premiere's end-of-sentence flag is
|
|
268
|
+
derived from trailing punctuation, so the punctuated form is what the exporter
|
|
269
|
+
needs to see.
|
|
270
|
+
|
|
271
|
+
Every level of the envelope is checked rather than defaulted away. This used
|
|
272
|
+
to be a chain of `or []`, so an error-shaped 200, a truncated body, or a
|
|
273
|
+
changed schema all produced an empty list -- indistinguishable from real
|
|
274
|
+
silence. The caller believes that: it advances its coverage cursor past the
|
|
275
|
+
audio and can retire the previous exports, so a passage of speech is
|
|
276
|
+
permanently published as having contained none. Only an explicit `words: []`
|
|
277
|
+
inside a well-formed alternative counts as silence; transcript text with no
|
|
278
|
+
word timings does not.
|
|
279
|
+
"""
|
|
280
|
+
if not isinstance(response, dict):
|
|
281
|
+
raise TranscriptionError("deepgram response was not a JSON object")
|
|
282
|
+
for key in ("err_code", "err_msg", "error", "message"):
|
|
283
|
+
if key not in response:
|
|
284
|
+
continue
|
|
285
|
+
value = response[key]
|
|
286
|
+
present = (bool(value.strip()) if isinstance(value, str)
|
|
287
|
+
else bool(value) if isinstance(value, (dict, list, tuple, set))
|
|
288
|
+
else value is not None)
|
|
289
|
+
if present:
|
|
290
|
+
detail = (json.dumps(value, ensure_ascii=False)
|
|
291
|
+
if isinstance(value, (dict, list)) else str(value))
|
|
292
|
+
raise TranscriptionError(
|
|
293
|
+
f"deepgram reported an error in {key}: {detail[:300]}")
|
|
294
|
+
|
|
295
|
+
results = response.get("results")
|
|
296
|
+
if not isinstance(results, dict):
|
|
297
|
+
raise TranscriptionError("deepgram response has no results object")
|
|
298
|
+
channels = results.get("channels")
|
|
299
|
+
if not isinstance(channels, list) or not channels:
|
|
300
|
+
raise TranscriptionError("deepgram response has no channels")
|
|
301
|
+
if not isinstance(channels[0], dict):
|
|
302
|
+
raise TranscriptionError("deepgram channel was not an object")
|
|
303
|
+
alternatives = channels[0].get("alternatives")
|
|
304
|
+
if not isinstance(alternatives, list) or not alternatives:
|
|
305
|
+
raise TranscriptionError("deepgram response has no alternatives")
|
|
306
|
+
if not isinstance(alternatives[0], dict):
|
|
307
|
+
raise TranscriptionError("deepgram alternative was not an object")
|
|
308
|
+
|
|
309
|
+
audio_duration: float | None = None
|
|
310
|
+
if "metadata" in response:
|
|
311
|
+
metadata = response["metadata"]
|
|
312
|
+
if not isinstance(metadata, dict):
|
|
313
|
+
raise TranscriptionError("deepgram metadata was not an object")
|
|
314
|
+
if "duration" in metadata:
|
|
315
|
+
audio_duration = _response_number(
|
|
316
|
+
metadata["duration"], "metadata duration")
|
|
317
|
+
if audio_duration < 0.0:
|
|
318
|
+
raise TranscriptionError("deepgram metadata duration was negative")
|
|
319
|
+
submitted_duration: float | None = None
|
|
320
|
+
if expected_duration is not None:
|
|
321
|
+
submitted_duration = _response_number(
|
|
322
|
+
expected_duration, "submitted audio duration")
|
|
323
|
+
if submitted_duration < 0.0:
|
|
324
|
+
raise TranscriptionError("submitted audio duration was negative")
|
|
325
|
+
|
|
326
|
+
raw = alternatives[0].get("words")
|
|
327
|
+
if raw is None:
|
|
328
|
+
raise TranscriptionError(
|
|
329
|
+
"deepgram response carried no word timings; refusing to treat that "
|
|
330
|
+
"as silence")
|
|
331
|
+
if not isinstance(raw, list):
|
|
332
|
+
raise TranscriptionError("deepgram words was not a list")
|
|
333
|
+
# AUD2-007: an empty word list is only silence when the transcript is also
|
|
334
|
+
# empty. A non-blank transcript with no word timings is a contradiction --
|
|
335
|
+
# the model heard speech but returned no timings -- and treating it as silence
|
|
336
|
+
# lets the caller advance its cursor and retire good exports over real speech.
|
|
337
|
+
if not raw:
|
|
338
|
+
transcript = alternatives[0].get("transcript")
|
|
339
|
+
if isinstance(transcript, str) and transcript.strip():
|
|
340
|
+
raise TranscriptionError(
|
|
341
|
+
"deepgram returned transcript text but no word timings; refusing "
|
|
342
|
+
"to treat that as silence")
|
|
343
|
+
|
|
344
|
+
words: list[Word] = []
|
|
345
|
+
previous_start = 0.0
|
|
346
|
+
for position, entry in enumerate(raw):
|
|
347
|
+
if not isinstance(entry, dict):
|
|
348
|
+
raise TranscriptionError(
|
|
349
|
+
f"deepgram word entry {position} was not an object")
|
|
350
|
+
if "punctuated_word" in entry:
|
|
351
|
+
text = entry["punctuated_word"]
|
|
352
|
+
elif "word" in entry:
|
|
353
|
+
text = entry["word"]
|
|
354
|
+
else:
|
|
355
|
+
raise TranscriptionError(
|
|
356
|
+
f"deepgram word entry {position} has no text")
|
|
357
|
+
if not isinstance(text, str) or not text.strip():
|
|
358
|
+
raise TranscriptionError(
|
|
359
|
+
f"deepgram word entry {position} has blank or invalid text")
|
|
360
|
+
missing = [key for key in ("start", "end", "confidence")
|
|
361
|
+
if key not in entry]
|
|
362
|
+
if missing:
|
|
363
|
+
raise TranscriptionError(
|
|
364
|
+
f"deepgram word {text.strip()!r} is missing {', '.join(missing)}")
|
|
365
|
+
start = _response_number(entry["start"], "word start")
|
|
366
|
+
end = _response_number(entry["end"], "word end")
|
|
367
|
+
confidence = _response_number(entry["confidence"], "word confidence")
|
|
368
|
+
if start < 0.0 or end < start:
|
|
369
|
+
raise TranscriptionError(
|
|
370
|
+
f"deepgram word {text.strip()!r} has a nonsensical span "
|
|
371
|
+
f"({start} -> {end})")
|
|
372
|
+
if start + 1e-6 < previous_start:
|
|
373
|
+
raise TranscriptionError(
|
|
374
|
+
f"deepgram word {text.strip()!r} precedes the previous word")
|
|
375
|
+
if not 0.0 <= confidence <= 1.0:
|
|
376
|
+
raise TranscriptionError(
|
|
377
|
+
f"deepgram word {text.strip()!r} has confidence outside 0..1")
|
|
378
|
+
# Our own ffprobe measurement of the file we uploaded outranks the
|
|
379
|
+
# response's self-report, which nova-3 rounds to the second.
|
|
380
|
+
limit = (submitted_duration if submitted_duration is not None
|
|
381
|
+
else audio_duration)
|
|
382
|
+
if limit is not None:
|
|
383
|
+
if start > limit + WORD_START_TOLERANCE:
|
|
384
|
+
raise TranscriptionError(
|
|
385
|
+
f"deepgram word {text.strip()!r} starts beyond the "
|
|
386
|
+
f"{'submitted' if submitted_duration is not None else 'response'}"
|
|
387
|
+
f" audio duration ({start} > {limit}); the response does not "
|
|
388
|
+
f"describe the audio that was sent")
|
|
389
|
+
# An overrunning end is an estimate, not extra audio. Keep the word
|
|
390
|
+
# -- it was spoken -- and cap it at the audio it was heard in.
|
|
391
|
+
end = max(start, min(end, limit))
|
|
392
|
+
words.append(Word(
|
|
393
|
+
text=text.strip(),
|
|
394
|
+
start=start,
|
|
395
|
+
duration=max(0.0, end - start),
|
|
396
|
+
confidence=confidence,
|
|
397
|
+
))
|
|
398
|
+
previous_start = start
|
|
399
|
+
return normalise(words)
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def build_provider(config, secret: str, *,
|
|
403
|
+
on_response: Callable[[dict[str, Any]], None] | None = None) -> ASRProvider:
|
|
404
|
+
name = (config.get("transcription.provider") or "deepgram").lower()
|
|
405
|
+
if name != "deepgram":
|
|
406
|
+
raise TranscriptionError(f"unknown transcription provider: {name}")
|
|
407
|
+
return DeepgramProvider(
|
|
408
|
+
secret,
|
|
409
|
+
on_response=on_response,
|
|
410
|
+
model=config.get("transcription.model", "nova-3"),
|
|
411
|
+
language=config.get("transcription.language", "en"),
|
|
412
|
+
filler_words=bool(config.get("transcription.filler_words", True)),
|
|
413
|
+
max_retries=int(config.get("transcription.max_retries", 4)),
|
|
414
|
+
timeout=float(config.get("transcription.request_timeout_seconds", 600)),
|
|
415
|
+
)
|