brief-spec-renderer-audio 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- brief_spec_renderer_audio-0.5.0/.gitignore +13 -0
- brief_spec_renderer_audio-0.5.0/PKG-INFO +14 -0
- brief_spec_renderer_audio-0.5.0/README.md +4 -0
- brief_spec_renderer_audio-0.5.0/pyproject.toml +22 -0
- brief_spec_renderer_audio-0.5.0/src/briefspec_renderer_audio/__init__.py +271 -0
- brief_spec_renderer_audio-0.5.0/tests/test_audio_renderer.py +89 -0
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: brief-spec-renderer-audio
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: Verified local macOS and opt-in OpenAI audio rendering for Brief-Spec.
|
|
5
|
+
Author: Luan Moreno Maciel
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Requires-Python: >=3.11
|
|
8
|
+
Requires-Dist: brief-spec<0.6,>=0.5
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
|
|
11
|
+
# Brief-Spec audio renderer
|
|
12
|
+
|
|
13
|
+
Optional MP3 renderer for Spoken Briefs. It supports offline macOS `say` and an explicitly
|
|
14
|
+
authorized OpenAI text-to-speech provider. It never falls back from local speech to the network.
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.27"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "brief-spec-renderer-audio"
|
|
7
|
+
version = "0.5.0"
|
|
8
|
+
description = "Verified local macOS and opt-in OpenAI audio rendering for Brief-Spec."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [{ name = "Luan Moreno Maciel" }]
|
|
13
|
+
dependencies = ["brief-spec>=0.5,<0.6"]
|
|
14
|
+
|
|
15
|
+
[project.entry-points."brief_spec.renderers"]
|
|
16
|
+
audio = "briefspec_renderer_audio:AudioRenderer"
|
|
17
|
+
|
|
18
|
+
[project.entry-points."briefspec.renderers"]
|
|
19
|
+
audio = "briefspec_renderer_audio:AudioRenderer"
|
|
20
|
+
|
|
21
|
+
[tool.hatch.build.targets.wheel]
|
|
22
|
+
packages = ["src/briefspec_renderer_audio"]
|
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import platform
|
|
6
|
+
import shutil
|
|
7
|
+
import subprocess
|
|
8
|
+
import tempfile
|
|
9
|
+
import urllib.error
|
|
10
|
+
import urllib.request
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from briefspec.delivery import render_spoken_text, sha256_bytes
|
|
15
|
+
from briefspec.state import atomic_write_public
|
|
16
|
+
|
|
17
|
+
__version__ = "0.5.0"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _tool_version(command: str) -> str:
|
|
21
|
+
result = subprocess.run(
|
|
22
|
+
[command, "-version"],
|
|
23
|
+
text=True,
|
|
24
|
+
capture_output=True,
|
|
25
|
+
timeout=30,
|
|
26
|
+
check=False,
|
|
27
|
+
)
|
|
28
|
+
output = result.stdout or result.stderr
|
|
29
|
+
return output.splitlines()[0].strip() if output else "unknown"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _probe_audio(artifact: Path, ffprobe: str) -> tuple[float, set[str], str | None]:
|
|
33
|
+
result = subprocess.run(
|
|
34
|
+
[
|
|
35
|
+
ffprobe,
|
|
36
|
+
"-v",
|
|
37
|
+
"error",
|
|
38
|
+
"-show_entries",
|
|
39
|
+
"format=duration:format_tags=comment:stream=codec_name,codec_type",
|
|
40
|
+
"-of",
|
|
41
|
+
"json",
|
|
42
|
+
str(artifact),
|
|
43
|
+
],
|
|
44
|
+
text=True,
|
|
45
|
+
capture_output=True,
|
|
46
|
+
timeout=60,
|
|
47
|
+
check=False,
|
|
48
|
+
)
|
|
49
|
+
if result.returncode != 0:
|
|
50
|
+
raise RuntimeError(result.stderr.strip() or "ffprobe failed")
|
|
51
|
+
try:
|
|
52
|
+
value = json.loads(result.stdout)
|
|
53
|
+
duration = float(value["format"]["duration"])
|
|
54
|
+
codecs = {
|
|
55
|
+
stream.get("codec_name")
|
|
56
|
+
for stream in value.get("streams", [])
|
|
57
|
+
if stream.get("codec_type") == "audio"
|
|
58
|
+
}
|
|
59
|
+
comment = value.get("format", {}).get("tags", {}).get("comment")
|
|
60
|
+
except (KeyError, TypeError, ValueError, json.JSONDecodeError) as exc:
|
|
61
|
+
raise ValueError(f"invalid ffprobe output: {exc}") from exc
|
|
62
|
+
return duration, codecs, str(comment) if comment is not None else None
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
class AudioRenderer:
|
|
66
|
+
name = "audio"
|
|
67
|
+
media_type = "audio/mpeg"
|
|
68
|
+
filename = "brief.mp3"
|
|
69
|
+
|
|
70
|
+
def capabilities(self) -> dict[str, Any]:
|
|
71
|
+
tools = {name: shutil.which(name) for name in ("say", "ffmpeg", "ffprobe")}
|
|
72
|
+
return {
|
|
73
|
+
"renderer_version": __version__,
|
|
74
|
+
"media_type": self.media_type,
|
|
75
|
+
"providers": {
|
|
76
|
+
"macos": bool(tools["say"] and tools["ffmpeg"]),
|
|
77
|
+
"openai": bool(os.environ.get("OPENAI_API_KEY") and tools["ffmpeg"]),
|
|
78
|
+
},
|
|
79
|
+
"verification_tools": tools,
|
|
80
|
+
"ready": bool(tools["ffmpeg"] and tools["ffprobe"]),
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
def setup(self, *, dry_run: bool = False) -> dict[str, Any]:
|
|
84
|
+
missing = [name for name in ("ffmpeg", "ffprobe") if shutil.which(name) is None]
|
|
85
|
+
status = "PASS" if not missing else "WARN"
|
|
86
|
+
return {
|
|
87
|
+
"status": "DRY-RUN" if dry_run else status,
|
|
88
|
+
"detail": "ready" if not missing else f"install manually: {', '.join(missing)}",
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
def _macos(self, script: str, target: Path, voice: str, rate: int) -> None:
|
|
92
|
+
say = shutil.which("say")
|
|
93
|
+
if say is None:
|
|
94
|
+
raise RuntimeError("macOS say is unavailable; no cloud fallback was attempted")
|
|
95
|
+
result = subprocess.run(
|
|
96
|
+
[say, "-v", voice, "-r", str(rate), "-o", str(target), script],
|
|
97
|
+
text=True,
|
|
98
|
+
capture_output=True,
|
|
99
|
+
timeout=300,
|
|
100
|
+
check=False,
|
|
101
|
+
)
|
|
102
|
+
if result.returncode != 0:
|
|
103
|
+
raise RuntimeError(result.stderr.strip() or "macOS say failed")
|
|
104
|
+
|
|
105
|
+
def _openai(self, script: str, target: Path, voice: str, options: dict[str, Any]) -> str:
|
|
106
|
+
if not options.get("consent_network"):
|
|
107
|
+
raise ValueError("OpenAI audio requires --consent-network")
|
|
108
|
+
api_key = os.environ.get("OPENAI_API_KEY")
|
|
109
|
+
if not api_key:
|
|
110
|
+
raise ValueError("OPENAI_API_KEY is required for the OpenAI audio provider")
|
|
111
|
+
model = str(options.get("model", "gpt-4o-mini-tts"))
|
|
112
|
+
payload = json.dumps(
|
|
113
|
+
{
|
|
114
|
+
"model": model,
|
|
115
|
+
"voice": voice,
|
|
116
|
+
"input": script,
|
|
117
|
+
"response_format": "mp3",
|
|
118
|
+
}
|
|
119
|
+
).encode("utf-8")
|
|
120
|
+
request = urllib.request.Request(
|
|
121
|
+
"https://api.openai.com/v1/audio/speech",
|
|
122
|
+
data=payload,
|
|
123
|
+
headers={
|
|
124
|
+
"Authorization": f"Bearer {api_key}",
|
|
125
|
+
"Content-Type": "application/json",
|
|
126
|
+
},
|
|
127
|
+
method="POST",
|
|
128
|
+
)
|
|
129
|
+
try:
|
|
130
|
+
with urllib.request.urlopen(request, timeout=180) as response:
|
|
131
|
+
target.write_bytes(response.read())
|
|
132
|
+
except urllib.error.HTTPError as exc:
|
|
133
|
+
raise RuntimeError(f"OpenAI audio request failed with HTTP {exc.code}") from exc
|
|
134
|
+
return model
|
|
135
|
+
|
|
136
|
+
def render(
|
|
137
|
+
self,
|
|
138
|
+
delivery: dict[str, Any],
|
|
139
|
+
output: Path,
|
|
140
|
+
options: dict[str, Any],
|
|
141
|
+
) -> dict[str, Any]:
|
|
142
|
+
script = render_spoken_text(delivery).strip()
|
|
143
|
+
record = render_script_document(
|
|
144
|
+
script,
|
|
145
|
+
output,
|
|
146
|
+
created_at=str(delivery.get("source", {}).get("created_at") or ""),
|
|
147
|
+
provider=str(options.get("provider", "macos")),
|
|
148
|
+
voice=options.get("voice"),
|
|
149
|
+
rate=int(options.get("rate", 190)),
|
|
150
|
+
consent_network=bool(options.get("consent_network")),
|
|
151
|
+
model=str(options.get("model", "gpt-4o-mini-tts")),
|
|
152
|
+
)
|
|
153
|
+
record["path"] = str(output)
|
|
154
|
+
return record
|
|
155
|
+
|
|
156
|
+
def verify(self, artifact: Path) -> dict[str, Any]:
|
|
157
|
+
ffprobe = shutil.which("ffprobe")
|
|
158
|
+
if ffprobe is None:
|
|
159
|
+
return {"status": "FAIL", "detail": "ffprobe is required for MP3 verification"}
|
|
160
|
+
try:
|
|
161
|
+
duration, codecs, comment = _probe_audio(artifact, ffprobe)
|
|
162
|
+
except (RuntimeError, ValueError) as exc:
|
|
163
|
+
return {"status": "FAIL", "detail": str(exc)}
|
|
164
|
+
if duration <= 0 or "mp3" not in codecs:
|
|
165
|
+
return {"status": "FAIL", "detail": "audio is empty or not MP3"}
|
|
166
|
+
if comment not in {
|
|
167
|
+
"AI-generated speech by Brief-Spec",
|
|
168
|
+
"AI-generated speech by BriefSpec", # legacy 0.x artifact compatibility
|
|
169
|
+
}:
|
|
170
|
+
return {"status": "FAIL", "detail": "AI-generated speech disclosure is missing"}
|
|
171
|
+
return {
|
|
172
|
+
"status": "PASS",
|
|
173
|
+
"detail": f"MP3 decoded; duration {duration:.2f}s; disclosure verified",
|
|
174
|
+
"duration_seconds": duration,
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def render_script_document(
|
|
179
|
+
script: str,
|
|
180
|
+
output: Path,
|
|
181
|
+
*,
|
|
182
|
+
created_at: str,
|
|
183
|
+
provider: str = "macos",
|
|
184
|
+
voice: str | None = None,
|
|
185
|
+
rate: int = 190,
|
|
186
|
+
consent_network: bool = False,
|
|
187
|
+
model: str = "gpt-4o-mini-tts",
|
|
188
|
+
force: bool = False,
|
|
189
|
+
) -> dict[str, Any]:
|
|
190
|
+
"""Render one canonical spoken script without reading screen-only evidence content."""
|
|
191
|
+
if output.exists() and not force:
|
|
192
|
+
raise FileExistsError(f"Refusing to overwrite existing output: {output}")
|
|
193
|
+
if not created_at:
|
|
194
|
+
raise ValueError("Audio rendering requires a canonical created_at")
|
|
195
|
+
ffmpeg = shutil.which("ffmpeg")
|
|
196
|
+
if ffmpeg is None:
|
|
197
|
+
raise RuntimeError("ffmpeg is required for Brief-Spec MP3 output")
|
|
198
|
+
ffprobe = shutil.which("ffprobe")
|
|
199
|
+
if ffprobe is None:
|
|
200
|
+
raise RuntimeError("ffprobe is required for Brief-Spec MP3 verification")
|
|
201
|
+
selected_voice = str(voice or ("marin" if provider == "openai" else "Samantha"))
|
|
202
|
+
if not 80 <= rate <= 400:
|
|
203
|
+
raise ValueError("Audio rate must be between 80 and 400 words per minute")
|
|
204
|
+
renderer = AudioRenderer()
|
|
205
|
+
selected_model: str | None = None
|
|
206
|
+
with tempfile.TemporaryDirectory(prefix="briefspec-audio-") as temporary:
|
|
207
|
+
root = Path(temporary)
|
|
208
|
+
source = root / ("source.aiff" if provider == "macos" else "source.mp3")
|
|
209
|
+
rendered = root / "brief.mp3"
|
|
210
|
+
if provider == "macos":
|
|
211
|
+
renderer._macos(script, source, selected_voice, rate)
|
|
212
|
+
elif provider == "openai":
|
|
213
|
+
selected_model = renderer._openai(
|
|
214
|
+
script,
|
|
215
|
+
source,
|
|
216
|
+
selected_voice,
|
|
217
|
+
{"consent_network": consent_network, "model": model},
|
|
218
|
+
)
|
|
219
|
+
else:
|
|
220
|
+
raise ValueError("Audio provider must be macos or openai")
|
|
221
|
+
result = subprocess.run(
|
|
222
|
+
[
|
|
223
|
+
ffmpeg,
|
|
224
|
+
"-y",
|
|
225
|
+
"-i",
|
|
226
|
+
str(source),
|
|
227
|
+
"-codec:a",
|
|
228
|
+
"libmp3lame",
|
|
229
|
+
"-q:a",
|
|
230
|
+
"2",
|
|
231
|
+
"-metadata",
|
|
232
|
+
"comment=AI-generated speech by Brief-Spec",
|
|
233
|
+
str(rendered),
|
|
234
|
+
],
|
|
235
|
+
text=True,
|
|
236
|
+
capture_output=True,
|
|
237
|
+
timeout=300,
|
|
238
|
+
check=False,
|
|
239
|
+
)
|
|
240
|
+
if result.returncode != 0:
|
|
241
|
+
raise RuntimeError(result.stderr.strip() or "ffmpeg MP3 conversion failed")
|
|
242
|
+
content = rendered.read_bytes()
|
|
243
|
+
duration, codecs, comment = _probe_audio(rendered, ffprobe)
|
|
244
|
+
if duration <= 0 or "mp3" not in codecs:
|
|
245
|
+
raise RuntimeError("rendered audio is empty or not MP3")
|
|
246
|
+
atomic_write_public(output, content, mode=0o644)
|
|
247
|
+
metadata = {
|
|
248
|
+
"renderer": "audio",
|
|
249
|
+
"renderer_version": __version__,
|
|
250
|
+
"provider": provider,
|
|
251
|
+
"model": selected_model,
|
|
252
|
+
"voice": selected_voice,
|
|
253
|
+
"rate": rate,
|
|
254
|
+
"source_script_sha256": sha256_bytes(script.encode("utf-8")),
|
|
255
|
+
"canonical_created_at": created_at,
|
|
256
|
+
"say_version": (f"macOS {platform.mac_ver()[0]} say" if provider == "macos" else None),
|
|
257
|
+
"ffmpeg_version": _tool_version(ffmpeg),
|
|
258
|
+
"ffprobe_version": _tool_version(ffprobe),
|
|
259
|
+
"duration_seconds": duration,
|
|
260
|
+
"ai_generated": True,
|
|
261
|
+
"disclosure": comment or "AI-generated speech",
|
|
262
|
+
}
|
|
263
|
+
return {
|
|
264
|
+
"format": "audio",
|
|
265
|
+
"path": output.name,
|
|
266
|
+
"media_type": "audio/mpeg",
|
|
267
|
+
"size_bytes": len(content),
|
|
268
|
+
"sha256": sha256_bytes(content),
|
|
269
|
+
"renderer_version": __version__,
|
|
270
|
+
"metadata": metadata,
|
|
271
|
+
}
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
import briefspec_renderer_audio as audio
|
|
7
|
+
import pytest
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class _Response:
|
|
11
|
+
def __enter__(self) -> _Response:
|
|
12
|
+
return self
|
|
13
|
+
|
|
14
|
+
def __exit__(self, *_args: object) -> None:
|
|
15
|
+
return None
|
|
16
|
+
|
|
17
|
+
def read(self) -> bytes:
|
|
18
|
+
return b"mock-mp3"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def test_openai_requires_explicit_consent_and_environment_key(
|
|
22
|
+
monkeypatch: pytest.MonkeyPatch, tmp_path: Path
|
|
23
|
+
) -> None:
|
|
24
|
+
renderer = audio.AudioRenderer()
|
|
25
|
+
target = tmp_path / "speech.mp3"
|
|
26
|
+
monkeypatch.delenv("OPENAI_API_KEY", raising=False)
|
|
27
|
+
with pytest.raises(ValueError, match="consent"):
|
|
28
|
+
renderer._openai("bounded script", target, "marin", {})
|
|
29
|
+
with pytest.raises(ValueError, match="OPENAI_API_KEY"):
|
|
30
|
+
renderer._openai("bounded script", target, "marin", {"consent_network": True})
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def test_openai_request_uses_documented_defaults_without_persisting_credentials(
|
|
34
|
+
monkeypatch: pytest.MonkeyPatch, tmp_path: Path
|
|
35
|
+
) -> None:
|
|
36
|
+
captured: dict[str, object] = {}
|
|
37
|
+
|
|
38
|
+
def urlopen(request: object, timeout: int) -> _Response:
|
|
39
|
+
captured["request"] = request
|
|
40
|
+
captured["timeout"] = timeout
|
|
41
|
+
return _Response()
|
|
42
|
+
|
|
43
|
+
monkeypatch.setenv("OPENAI_API_KEY", "test-secret-never-persist")
|
|
44
|
+
monkeypatch.setattr(audio.urllib.request, "urlopen", urlopen)
|
|
45
|
+
target = tmp_path / "speech.mp3"
|
|
46
|
+
model = audio.AudioRenderer()._openai(
|
|
47
|
+
"bounded script",
|
|
48
|
+
target,
|
|
49
|
+
"marin",
|
|
50
|
+
{"consent_network": True},
|
|
51
|
+
)
|
|
52
|
+
request = captured["request"]
|
|
53
|
+
payload = json.loads(request.data)
|
|
54
|
+
assert model == "gpt-4o-mini-tts"
|
|
55
|
+
assert payload == {
|
|
56
|
+
"model": "gpt-4o-mini-tts",
|
|
57
|
+
"voice": "marin",
|
|
58
|
+
"input": "bounded script",
|
|
59
|
+
"response_format": "mp3",
|
|
60
|
+
}
|
|
61
|
+
assert request.get_header("Authorization") == "Bearer test-secret-never-persist"
|
|
62
|
+
assert target.read_bytes() == b"mock-mp3"
|
|
63
|
+
assert b"test-secret-never-persist" not in target.read_bytes()
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def test_local_provider_never_falls_back_to_network(
|
|
67
|
+
monkeypatch: pytest.MonkeyPatch, tmp_path: Path
|
|
68
|
+
) -> None:
|
|
69
|
+
monkeypatch.setattr(audio.shutil, "which", lambda _name: None)
|
|
70
|
+
monkeypatch.setattr(
|
|
71
|
+
audio.urllib.request,
|
|
72
|
+
"urlopen",
|
|
73
|
+
lambda *_args, **_kwargs: (_ for _ in ()).throw(AssertionError("network attempted")),
|
|
74
|
+
)
|
|
75
|
+
with pytest.raises(RuntimeError, match="no cloud fallback"):
|
|
76
|
+
audio.AudioRenderer()._macos("script", tmp_path / "speech.aiff", "Samantha", 190)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def test_generic_script_helper_fails_closed_before_tool_use(tmp_path: Path) -> None:
|
|
80
|
+
target = tmp_path / "chronicle.mp3"
|
|
81
|
+
target.write_bytes(b"existing")
|
|
82
|
+
with pytest.raises(FileExistsError, match="Refusing to overwrite"):
|
|
83
|
+
audio.render_script_document(
|
|
84
|
+
"Bounded script",
|
|
85
|
+
target,
|
|
86
|
+
created_at="2026-08-14T12:00:00+00:00",
|
|
87
|
+
)
|
|
88
|
+
with pytest.raises(ValueError, match="canonical created_at"):
|
|
89
|
+
audio.render_script_document("Bounded script", tmp_path / "new.mp3", created_at="")
|