echoact 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- echoact/__init__.py +3 -0
- echoact/__main__.py +117 -0
- echoact/app.py +315 -0
- echoact/audio/__init__.py +0 -0
- echoact/audio/devices.py +192 -0
- echoact/audio/player.py +611 -0
- echoact/audio/wav.py +854 -0
- echoact/config/__init__.py +0 -0
- echoact/config/budget.py +370 -0
- echoact/config/settings.py +1244 -0
- echoact/db/__init__.py +0 -0
- echoact/db/backup.py +2429 -0
- echoact/db/migrations.py +434 -0
- echoact/db/schema.sql +214 -0
- echoact/db/store.py +2062 -0
- echoact/diagnostics.py +902 -0
- echoact/domain.py +487 -0
- echoact/engine/__init__.py +0 -0
- echoact/engine/container.py +843 -0
- echoact/engine/protocol.py +241 -0
- echoact/engine/runtime.py +324 -0
- echoact/engine/supervisor.py +961 -0
- echoact/engine/worker.py +659 -0
- echoact/errors.py +281 -0
- echoact/instance.py +172 -0
- echoact/jobs/__init__.py +0 -0
- echoact/jobs/engine.py +776 -0
- echoact/jobs/request.py +300 -0
- echoact/mcp/__init__.py +0 -0
- echoact/mcp/__main__.py +50 -0
- echoact/mcp/client.py +202 -0
- echoact/mcp/config.py +112 -0
- echoact/mcp/server.py +340 -0
- echoact/models/__init__.py +0 -0
- echoact/models/catalog.py +273 -0
- echoact/models/manifest.py +278 -0
- echoact/models/registry.py +1551 -0
- echoact/paths.py +93 -0
- echoact/policy.py +189 -0
- echoact/security/__init__.py +0 -0
- echoact/security/credentials.py +930 -0
- echoact/security/ratelimit.py +534 -0
- echoact/service/__init__.py +20 -0
- echoact/service/app.py +182 -0
- echoact/service/deps.py +563 -0
- echoact/service/errors.py +241 -0
- echoact/service/routes.py +1125 -0
- echoact/service/schemas.py +509 -0
- echoact/service/server.py +270 -0
- echoact/text/__init__.py +0 -0
- echoact/text/language.py +44 -0
- echoact/text/loader.py +577 -0
- echoact/text/normalize.py +924 -0
- echoact/text/segment.py +499 -0
- echoact/text/sniff.py +1202 -0
- echoact/ui/__init__.py +0 -0
- echoact/ui/bridge.py +50 -0
- echoact/ui/controls.py +360 -0
- echoact/ui/credential_dialog.py +131 -0
- echoact/ui/fonts.py +94 -0
- echoact/ui/i18n.py +260 -0
- echoact/ui/icons.py +440 -0
- echoact/ui/library.py +1642 -0
- echoact/ui/licence.py +162 -0
- echoact/ui/main_window.py +1202 -0
- echoact/ui/mcp_setup.py +494 -0
- echoact/ui/models_view.py +1142 -0
- echoact/ui/notifications.py +202 -0
- echoact/ui/reading.py +494 -0
- echoact/ui/settings_view.py +2258 -0
- echoact/ui/status_view.py +1193 -0
- echoact/ui/theme.py +579 -0
- echoact/util/__init__.py +0 -0
- echoact/util/ids.py +62 -0
- echoact/util/logging.py +127 -0
- echoact-0.1.0.dist-info/METADATA +162 -0
- echoact-0.1.0.dist-info/RECORD +80 -0
- echoact-0.1.0.dist-info/WHEEL +4 -0
- echoact-0.1.0.dist-info/entry_points.txt +3 -0
- echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
echoact/mcp/server.py
ADDED
|
@@ -0,0 +1,340 @@
|
|
|
1
|
+
"""The MCP server: eight tools, each one REST call, and nothing else.
|
|
2
|
+
|
|
3
|
+
F-59 is unusually strict about this and it is worth restating, because the
|
|
4
|
+
temptation to be helpful here is what would break it: each tool maps onto
|
|
5
|
+
one REST operation and adds no capability REST does not already expose.
|
|
6
|
+
N-24 makes the REST contract canonical and MCP a projection of it, so a
|
|
7
|
+
behaviour difference between the two is a defect -- which is why these
|
|
8
|
+
tools pass the service's JSON through rather than re-modelling it. A
|
|
9
|
+
second model of the same data is a second thing to keep in step, and it
|
|
10
|
+
would not stay in step.
|
|
11
|
+
|
|
12
|
+
The revision this implements, 2026-07-28, is stateless and has no
|
|
13
|
+
connection-establishing handshake, so:
|
|
14
|
+
|
|
15
|
+
* Nothing is cached between calls and no session is opened. The
|
|
16
|
+
connection is read once at start-up because it is configuration, not
|
|
17
|
+
state.
|
|
18
|
+
* Cross-call state travels as a server-minted job identifier passed as an
|
|
19
|
+
ordinary tool argument -- which is what ``job_id`` is.
|
|
20
|
+
* Result references are MCP resource URIs, and they are identifiers
|
|
21
|
+
rather than addresses. ``echoact://job/<id>/audio`` names a thing; it
|
|
22
|
+
carries no host, no filesystem path, and no credential, and resolving
|
|
23
|
+
it requires this process's own credential. §2.11 forbids the
|
|
24
|
+
alternatives by name.
|
|
25
|
+
* Resource reads are cacheable on this revision, so every reference
|
|
26
|
+
carries a freshness hint no longer than the result's remaining lifetime
|
|
27
|
+
and is marked private, so no shared intermediary may hold one owner's
|
|
28
|
+
audio. The hint travels on the reference the tool returns. The
|
|
29
|
+
protocol's own ``ttlMs``/``cacheScope`` on a read default to 0 and
|
|
30
|
+
private in this SDK, which is stricter than the requirement rather than
|
|
31
|
+
looser, so the two agree and neither has to be argued about.
|
|
32
|
+
|
|
33
|
+
Diagnostics go to stderr. The revision deprecates the protocol's own
|
|
34
|
+
logging and §2.11 says so explicitly.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import sys
|
|
40
|
+
from typing import Annotated, Any
|
|
41
|
+
|
|
42
|
+
from ..errors import Code, EchoActError
|
|
43
|
+
from ..policy import BOUNDED_WAIT_CEILING_S, MCP_PROTOCOL_REVISION
|
|
44
|
+
from ..util import ids
|
|
45
|
+
from .client import RestClient, describe
|
|
46
|
+
from .config import Connection
|
|
47
|
+
|
|
48
|
+
SERVER_NAME = "echoact"
|
|
49
|
+
RESOURCE_SCHEME = "echoact"
|
|
50
|
+
|
|
51
|
+
INSTRUCTIONS = """\
|
|
52
|
+
EchoAct generates Korean and English speech locally, on this machine.
|
|
53
|
+
|
|
54
|
+
One generation runs at a time across the whole application, including the
|
|
55
|
+
person sitting at it, so contention is normal: a busy answer carries a
|
|
56
|
+
retry_after_s and giving up is a reasonable response to it.
|
|
57
|
+
|
|
58
|
+
Call estimate_speech before create_speech for anything long: it validates,
|
|
59
|
+
counts segments, gives the expected length, and says whether the slot is
|
|
60
|
+
free, without creating a job or loading a model.
|
|
61
|
+
|
|
62
|
+
create_speech needs an idempotency_key. Reuse the same key when you retry
|
|
63
|
+
and you will get the same job back rather than a second rendering.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _stderr(message: str) -> None:
|
|
68
|
+
print(f"[echoact-mcp] {message}", file=sys.stderr, flush=True)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def audio_uri(job_id: str, segment_id: str | None = None) -> str:
|
|
72
|
+
if segment_id:
|
|
73
|
+
return f"{RESOURCE_SCHEME}://job/{job_id}/segment/{segment_id}"
|
|
74
|
+
return f"{RESOURCE_SCHEME}://job/{job_id}/audio"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _ttl_ms(result: dict[str, Any], now: float | None = None) -> int:
|
|
78
|
+
"""A freshness hint that cannot outlive the result.
|
|
79
|
+
|
|
80
|
+
F-60 caps the hint at the remaining lifetime in 4.1. A retained
|
|
81
|
+
result has no expiry, and a hint that never goes stale would still be
|
|
82
|
+
wrong -- the owner may delete it -- so it is capped at the one-hour
|
|
83
|
+
figure the same section uses for a one-off.
|
|
84
|
+
"""
|
|
85
|
+
moment = ids.now() if now is None else now
|
|
86
|
+
expires = result.get("expires_at")
|
|
87
|
+
if expires is None:
|
|
88
|
+
return 3_600_000
|
|
89
|
+
remaining = float(expires) - moment
|
|
90
|
+
return max(0, int(remaining * 1000))
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def build(connection: Connection, *, client: RestClient | None = None):
|
|
94
|
+
"""Construct the server. Takes a client so a test can supply one."""
|
|
95
|
+
from fastmcp import FastMCP
|
|
96
|
+
|
|
97
|
+
rest = client or RestClient(connection)
|
|
98
|
+
mcp = FastMCP(
|
|
99
|
+
name=SERVER_NAME,
|
|
100
|
+
instructions=INSTRUCTIONS,
|
|
101
|
+
version="0.1.0",
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
def call(fn):
|
|
105
|
+
"""Run one REST call, and turn a failure into a described result.
|
|
106
|
+
|
|
107
|
+
A tool that raised would give the client an exception where F-57
|
|
108
|
+
promises a code, whether a retry can succeed, and a request
|
|
109
|
+
identifier. So failures come back as data.
|
|
110
|
+
"""
|
|
111
|
+
try:
|
|
112
|
+
return fn()
|
|
113
|
+
except EchoActError as exc:
|
|
114
|
+
if exc.code in {Code.APP_NOT_RUNNING, Code.SERVICE_OFF, Code.MCP_DISABLED}:
|
|
115
|
+
_stderr(f"{exc.code.value}: {exc.message}")
|
|
116
|
+
return describe(exc)
|
|
117
|
+
|
|
118
|
+
# -- discovery ------------------------------------------------------
|
|
119
|
+
|
|
120
|
+
@mcp.tool
|
|
121
|
+
def list_models() -> dict[str, Any]:
|
|
122
|
+
"""Models, voices, languages, speaking styles and input limits.
|
|
123
|
+
|
|
124
|
+
Every voice carries a description as well as an identifier,
|
|
125
|
+
because an identifier alone gives a caller that is not a person
|
|
126
|
+
nothing to choose on (F-53).
|
|
127
|
+
"""
|
|
128
|
+
return call(lambda: rest.get("/models"))
|
|
129
|
+
|
|
130
|
+
@mcp.tool
|
|
131
|
+
def estimate_speech(
|
|
132
|
+
text: Annotated[str, "The text to be spoken."],
|
|
133
|
+
model_id: Annotated[str, "From list_models."],
|
|
134
|
+
voice_id: Annotated[str, "A voice belonging to that model."],
|
|
135
|
+
gender: Annotated[str, "female or male; must match the voice's own."],
|
|
136
|
+
language: Annotated[str, "auto, ko or en."] = "auto",
|
|
137
|
+
style: Annotated[str, "natural, calm, bright or narration."] = "natural",
|
|
138
|
+
tempo: Annotated[float, "0.70 to 1.50."] = 1.0,
|
|
139
|
+
) -> dict[str, Any]:
|
|
140
|
+
"""Validate and estimate without creating a job or loading a model.
|
|
141
|
+
|
|
142
|
+
Answers with the segment count, the expected audio length and
|
|
143
|
+
synthesis time, and whether the single generation slot is free.
|
|
144
|
+
The figures are approximate by construction: they come from the
|
|
145
|
+
model's recorded throughput, not from a trial run (F-88).
|
|
146
|
+
"""
|
|
147
|
+
body = {
|
|
148
|
+
"text": text,
|
|
149
|
+
"kind": "speech",
|
|
150
|
+
"voice": _voice_body(model_id, voice_id, gender, language, style, tempo),
|
|
151
|
+
}
|
|
152
|
+
return call(lambda: rest.post("/estimate", body))
|
|
153
|
+
|
|
154
|
+
# -- generation -----------------------------------------------------
|
|
155
|
+
|
|
156
|
+
@mcp.tool
|
|
157
|
+
def create_speech(
|
|
158
|
+
text: Annotated[str, "The text to be spoken. Files and URLs are not accepted."],
|
|
159
|
+
idempotency_key: Annotated[
|
|
160
|
+
str, "Required. Reuse it on a retry to get the same job back rather than a second one."
|
|
161
|
+
],
|
|
162
|
+
model_id: Annotated[str, "From list_models."],
|
|
163
|
+
voice_id: Annotated[str, "A voice belonging to that model."],
|
|
164
|
+
gender: Annotated[str, "female or male; must match the voice's own."],
|
|
165
|
+
language: Annotated[str, "auto, ko or en."] = "auto",
|
|
166
|
+
style: Annotated[str, "natural, calm, bright or narration."] = "natural",
|
|
167
|
+
tempo: Annotated[float, "0.70 to 1.50."] = 1.0,
|
|
168
|
+
wait_seconds: Annotated[
|
|
169
|
+
float | None,
|
|
170
|
+
"Wait up to this long for the job to finish before answering. "
|
|
171
|
+
"The job is untouched either way; if the bound passes you get the job id.",
|
|
172
|
+
] = None,
|
|
173
|
+
retain: Annotated[bool, "Keep the result in EchoAct's library."] = False,
|
|
174
|
+
) -> dict[str, Any]:
|
|
175
|
+
"""Start generating speech.
|
|
176
|
+
|
|
177
|
+
Returns the job's state. If it finished inside ``wait_seconds``
|
|
178
|
+
the answer is terminal; otherwise it carries the job id and the
|
|
179
|
+
job continues untouched. Waiting never covers model preparation.
|
|
180
|
+
|
|
181
|
+
File and URL inputs are not accepted: N-18 keeps arbitrary paths
|
|
182
|
+
and external URLs out of what an integration can ask EchoAct to
|
|
183
|
+
read.
|
|
184
|
+
"""
|
|
185
|
+
body: dict[str, Any] = {
|
|
186
|
+
"kind": "speech",
|
|
187
|
+
"idempotency_key": idempotency_key,
|
|
188
|
+
"text": text,
|
|
189
|
+
"retain": retain,
|
|
190
|
+
"voice": _voice_body(model_id, voice_id, gender, language, style, tempo),
|
|
191
|
+
}
|
|
192
|
+
if wait_seconds is not None:
|
|
193
|
+
body["wait_s"] = max(0.0, min(float(wait_seconds), BOUNDED_WAIT_CEILING_S))
|
|
194
|
+
return call(lambda: rest.post("/jobs", body))
|
|
195
|
+
|
|
196
|
+
@mcp.tool
|
|
197
|
+
def get_speech_job(job_id: Annotated[str, "From create_speech."]) -> dict[str, Any]:
|
|
198
|
+
"""State, progress and any error for one of your own jobs."""
|
|
199
|
+
return call(lambda: rest.get(f"/jobs/{job_id}"))
|
|
200
|
+
|
|
201
|
+
@mcp.tool
|
|
202
|
+
def cancel_speech_job(job_id: Annotated[str, "From create_speech."]) -> dict[str, Any]:
|
|
203
|
+
"""Cancel one of your own jobs.
|
|
204
|
+
|
|
205
|
+
Cancelling twice adds nothing, and cancelling a job that already
|
|
206
|
+
finished returns its final state rather than changing it. This is
|
|
207
|
+
not the same as cancelling the MCP request itself.
|
|
208
|
+
"""
|
|
209
|
+
return call(lambda: rest.post(f"/jobs/{job_id}/cancel"))
|
|
210
|
+
|
|
211
|
+
@mcp.tool
|
|
212
|
+
def list_speech_segments(job_id: Annotated[str, "From create_speech."]) -> dict[str, Any]:
|
|
213
|
+
"""Ready segments, with their source-text range and their times.
|
|
214
|
+
|
|
215
|
+
Ranges are Unicode code point offsets into the text you supplied,
|
|
216
|
+
start inclusive and end exclusive, and times are milliseconds from
|
|
217
|
+
the start of the audio.
|
|
218
|
+
"""
|
|
219
|
+
return call(lambda: rest.get(f"/jobs/{job_id}/segments"))
|
|
220
|
+
|
|
221
|
+
@mcp.tool
|
|
222
|
+
def get_speech_result(
|
|
223
|
+
job_id: Annotated[str, "From create_speech."],
|
|
224
|
+
segment_id: Annotated[str | None, "From list_speech_segments, for one segment."] = None,
|
|
225
|
+
) -> dict[str, Any]:
|
|
226
|
+
"""A reference to the finished audio, and what it is.
|
|
227
|
+
|
|
228
|
+
The audio is not embedded here. You get a resource URI to read
|
|
229
|
+
over this same session, plus the format, rate, size, length and
|
|
230
|
+
expiry. The URI is an identifier, not an address: it holds no
|
|
231
|
+
path and no credential, and only this server can resolve it.
|
|
232
|
+
"""
|
|
233
|
+
|
|
234
|
+
def fetch() -> dict[str, Any]:
|
|
235
|
+
result = rest.get(f"/jobs/{job_id}/result")
|
|
236
|
+
uri = audio_uri(job_id, segment_id)
|
|
237
|
+
ttl = _ttl_ms(result)
|
|
238
|
+
return {
|
|
239
|
+
**result,
|
|
240
|
+
"resource": {
|
|
241
|
+
"uri": uri,
|
|
242
|
+
"mime_type": "audio/wav",
|
|
243
|
+
# F-60: no longer than the remaining lifetime, and
|
|
244
|
+
# private so no shared cache may hold one owner's audio.
|
|
245
|
+
"ttl_ms": ttl,
|
|
246
|
+
"cache_scope": "private",
|
|
247
|
+
},
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
return call(fetch)
|
|
251
|
+
|
|
252
|
+
@mcp.tool
|
|
253
|
+
def list_speech_history(
|
|
254
|
+
limit: Annotated[int | None, "Up to 100."] = None,
|
|
255
|
+
model: Annotated[str | None, "Filter by model id."] = None,
|
|
256
|
+
state: Annotated[str | None, "Filter by job state."] = None,
|
|
257
|
+
) -> dict[str, Any]:
|
|
258
|
+
"""Your own retained jobs, if your credential may read history.
|
|
259
|
+
|
|
260
|
+
A summary without body text is what you get; the source text of a
|
|
261
|
+
retained job needs its own permission and its own request.
|
|
262
|
+
"""
|
|
263
|
+
return call(lambda: rest.get("/jobs", limit=limit, model=model, state=state))
|
|
264
|
+
|
|
265
|
+
# -- resources ------------------------------------------------------
|
|
266
|
+
|
|
267
|
+
@mcp.resource(
|
|
268
|
+
f"{RESOURCE_SCHEME}://job/{{job_id}}/audio",
|
|
269
|
+
name="Finished audio",
|
|
270
|
+
mime_type="audio/wav",
|
|
271
|
+
description="The complete WAV for a job, resolved on your behalf.",
|
|
272
|
+
)
|
|
273
|
+
def job_audio(job_id: str) -> bytes:
|
|
274
|
+
payload, _ = rest.get_bytes(f"/jobs/{job_id}/audio")
|
|
275
|
+
return payload
|
|
276
|
+
|
|
277
|
+
@mcp.resource(
|
|
278
|
+
f"{RESOURCE_SCHEME}://job/{{job_id}}/segment/{{segment_id}}",
|
|
279
|
+
name="Segment audio",
|
|
280
|
+
mime_type="audio/wav",
|
|
281
|
+
description="One ready segment's WAV, resolved on your behalf.",
|
|
282
|
+
)
|
|
283
|
+
def segment_audio(job_id: str, segment_id: str) -> bytes:
|
|
284
|
+
payload, _ = rest.get_bytes(f"/jobs/{job_id}/segments/{segment_id}/audio")
|
|
285
|
+
return payload
|
|
286
|
+
|
|
287
|
+
return mcp
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _voice_body(
|
|
291
|
+
model_id: str,
|
|
292
|
+
voice_id: str,
|
|
293
|
+
gender: str,
|
|
294
|
+
language: str,
|
|
295
|
+
style: str,
|
|
296
|
+
tempo: float,
|
|
297
|
+
) -> dict[str, Any]:
|
|
298
|
+
"""The settings, stated in full, exactly as the REST contract wants them.
|
|
299
|
+
|
|
300
|
+
Nothing here falls back to the owner's current selection, and that is
|
|
301
|
+
the service's decision rather than this module's: an automated
|
|
302
|
+
caller's output must not depend on what the person at the keyboard
|
|
303
|
+
last clicked. So the tools require the three fields that identify a
|
|
304
|
+
voice and let ``list_models`` be the place a caller learns them --
|
|
305
|
+
which is also why they are required *parameters* rather than a
|
|
306
|
+
runtime error, since a tool schema can say so and an error cannot.
|
|
307
|
+
"""
|
|
308
|
+
return {
|
|
309
|
+
"model_id": model_id,
|
|
310
|
+
"voice_id": voice_id,
|
|
311
|
+
"gender": gender,
|
|
312
|
+
"language": language,
|
|
313
|
+
"style": style,
|
|
314
|
+
"tempo": tempo,
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def preflight(rest: RestClient) -> dict[str, Any]:
|
|
319
|
+
"""Check that this process can be useful before serving anything.
|
|
320
|
+
|
|
321
|
+
F-52 requires an app that is not running, a REST service that is off,
|
|
322
|
+
and a disabled MCP integration to be reported to the client as
|
|
323
|
+
distinct, non-retrying errors -- and requires this process never to
|
|
324
|
+
launch the app. Doing it here means the client is told once, clearly,
|
|
325
|
+
rather than on every tool call.
|
|
326
|
+
"""
|
|
327
|
+
status = rest.get("/status")
|
|
328
|
+
if not status.get("mcp_enabled", True):
|
|
329
|
+
raise EchoActError(
|
|
330
|
+
Code.MCP_DISABLED,
|
|
331
|
+
"MCP is turned off in EchoAct. Enable it under Settings; "
|
|
332
|
+
"this server cannot enable it.",
|
|
333
|
+
)
|
|
334
|
+
revision = status.get("mcp_protocol_revision")
|
|
335
|
+
if revision and revision != MCP_PROTOCOL_REVISION:
|
|
336
|
+
_stderr(
|
|
337
|
+
f"EchoAct reports MCP revision {revision}; this server implements "
|
|
338
|
+
f"{MCP_PROTOCOL_REVISION}."
|
|
339
|
+
)
|
|
340
|
+
return status
|
|
File without changes
|
|
@@ -0,0 +1,273 @@
|
|
|
1
|
+
"""The shipped manifest (F-84): one model, Supertonic 3.
|
|
2
|
+
|
|
3
|
+
A.3 fixes the lineup at a single model, and F-04 keeps the selection
|
|
4
|
+
machinery anyway so that a second one is data rather than structure. This
|
|
5
|
+
file is that data.
|
|
6
|
+
|
|
7
|
+
It is a Python literal on purpose. F-84 says manifest changes ship with the
|
|
8
|
+
application and are never applied as a silent remote update; a manifest that
|
|
9
|
+
could be fetched would let whoever served it redefine which bytes count as
|
|
10
|
+
sound for weights already on this disk. Changing anything here is a release.
|
|
11
|
+
|
|
12
|
+
The digests below were computed over the files of
|
|
13
|
+
``Supertone/supertonic-3`` at the pinned revision, on disk, with SHA-256 --
|
|
14
|
+
not copied from a listing. ``tests/test_manifest.py`` re-derives them from
|
|
15
|
+
the real weights when they are present, because F-84 makes a mismatch mean
|
|
16
|
+
"corrupted": an invented digest would turn every honest copy into a fake
|
|
17
|
+
corruption report, which is worse than shipping no manifest at all.
|
|
18
|
+
|
|
19
|
+
The licence text is quoted from the ``LICENSE`` file of that same revision.
|
|
20
|
+
N-11 treats the pass-through clause as a release blocker, so it is recorded
|
|
21
|
+
as an obligation the product must carry to its own users, not as a notice.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from typing import Final
|
|
27
|
+
|
|
28
|
+
from ..domain import Gender, Language
|
|
29
|
+
from ..policy import CPU_PERCENT_MIN, MEMORY_FLOOR_BYTES
|
|
30
|
+
from .manifest import (
|
|
31
|
+
LicenseTerms,
|
|
32
|
+
Manifest,
|
|
33
|
+
MinimumBudget,
|
|
34
|
+
ModelEntry,
|
|
35
|
+
ModelFile,
|
|
36
|
+
VoiceEntry,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
SUPERTONIC_3_ID: Final = "supertonic-3"
|
|
40
|
+
|
|
41
|
+
#: Attachment A of the BigScience OpenRAIL-M licence, quoted from the
|
|
42
|
+
#: ``LICENSE`` file at the pinned revision, in its own order and wording.
|
|
43
|
+
#: N-11 requires these to reach the end user unchanged, and a paraphrase is
|
|
44
|
+
#: what a reader would hold us to instead of the real term.
|
|
45
|
+
_USE_RESTRICTIONS: Final[tuple[str, ...]] = (
|
|
46
|
+
"(a) In any way that violates any applicable national, federal, state, local or "
|
|
47
|
+
"international law or regulation;",
|
|
48
|
+
"(b) For the purpose of exploiting, harming or attempting to exploit or harm minors in "
|
|
49
|
+
"any way;",
|
|
50
|
+
"(c) To generate or disseminate verifiably false information and/or content with the "
|
|
51
|
+
"purpose of harming others;",
|
|
52
|
+
"(d) To generate or disseminate personal identifiable information that can be used to "
|
|
53
|
+
"harm an individual;",
|
|
54
|
+
"(e) To generate or disseminate information and/or content (e.g. images, code, posts, "
|
|
55
|
+
"articles), and place the information and/or content in any context (e.g. bot generating "
|
|
56
|
+
"tweets) without expressly and intelligibly disclaiming that the information and/or "
|
|
57
|
+
"content is machine generated;",
|
|
58
|
+
"(f) To defame, disparage or otherwise harass others;",
|
|
59
|
+
"(g) To impersonate or attempt to impersonate (e.g. deepfakes) others without their "
|
|
60
|
+
"consent;",
|
|
61
|
+
"(h) For fully automated decision making that adversely impacts an individual’s legal "
|
|
62
|
+
"rights or otherwise creates or modifies a binding, enforceable obligation;",
|
|
63
|
+
"(i) For any use intended to or which has the effect of discriminating against or harming "
|
|
64
|
+
"individuals or groups based on online or offline social behavior or known or predicted "
|
|
65
|
+
"personal or personality characteristics;",
|
|
66
|
+
"(j) To exploit any of the vulnerabilities of a specific group of persons based on their "
|
|
67
|
+
"age, social, physical or mental characteristics, in order to materially distort the "
|
|
68
|
+
"behavior of a person pertaining to that group in a manner that causes or is likely to "
|
|
69
|
+
"cause that person or another person physical or psychological harm;",
|
|
70
|
+
"(k) For any use intended to or which has the effect of discriminating against individuals "
|
|
71
|
+
"or groups based on legally protected characteristics or categories;",
|
|
72
|
+
"(l) To provide medical advice and medical results interpretation;",
|
|
73
|
+
"(m) To generate or disseminate information for the purpose to be used for administration "
|
|
74
|
+
"of justice, law enforcement, immigration or asylum processes, such as predicting an "
|
|
75
|
+
"individual will commit fraud/crime commitment (e.g. by text profiling, drawing causal "
|
|
76
|
+
"relationships between assertions made in documents, indiscriminate and "
|
|
77
|
+
"arbitrarily-targeted use).",
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
SUPERTONIC_3_LICENSE: Final = LicenseTerms(
|
|
81
|
+
name="BigScience OpenRAIL-M",
|
|
82
|
+
restrictions=_USE_RESTRICTIONS,
|
|
83
|
+
# Paragraph 5, quoted. This sentence is why N-11 calls shipping the
|
|
84
|
+
# model without carrying the restrictions through a release blocker.
|
|
85
|
+
pass_through_obligation=(
|
|
86
|
+
"Paragraph 5, Use-based restrictions: “You shall require all of Your users who use "
|
|
87
|
+
"the Model or a Derivative of the Model to comply with the terms of this paragraph "
|
|
88
|
+
"(paragraph 5).” The restrictions in Attachment A therefore bind everyone who uses "
|
|
89
|
+
"EchoAct's speech generation, not only whoever installed it."
|
|
90
|
+
),
|
|
91
|
+
acceptance_required=True,
|
|
92
|
+
notes=(
|
|
93
|
+
"Full title as it appears in the file: “BigScience Open RAIL-M License, dated "
|
|
94
|
+
"August 18, 2022”.",
|
|
95
|
+
"Paragraph 6, the Output: “Except as set forth herein, Licensor claims no rights in "
|
|
96
|
+
"the Output You generate using the Model. You are accountable for the Output you "
|
|
97
|
+
"generate and its subsequent uses. No use of the output can contravene any provision "
|
|
98
|
+
"as stated in the License.”",
|
|
99
|
+
"Paragraph 7, Updates and Runtime Restrictions: the licensor “reserves the right to "
|
|
100
|
+
"restrict (remotely or otherwise) usage of the Model” and asks that You “undertake "
|
|
101
|
+
"reasonable efforts to use the latest version of the Model”. EchoAct implements no "
|
|
102
|
+
"remote control and checks for updates only on the user's action (N-01, F-75); A.4 "
|
|
103
|
+
"leaves reading this clause against a deliberately offline product to a lawyer.",
|
|
104
|
+
"Being able to run the model locally is not a right to redistribute it (N-11); the "
|
|
105
|
+
"weights are downloaded from the source repository, never bundled.",
|
|
106
|
+
),
|
|
107
|
+
source_file="LICENSE",
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
#: Every voice description ends with this. A.4 records that nobody has
|
|
111
|
+
#: listened to this model in either language yet, so a description that
|
|
112
|
+
#: claimed a character would be invented. F-53 still needs a caller to have
|
|
113
|
+
#: something to choose on, so the descriptions carry what is verifiable and
|
|
114
|
+
#: say plainly that the rest is unknown.
|
|
115
|
+
_UNREVIEWED: Final = (
|
|
116
|
+
"Its voice character has not been reviewed yet, so nothing is claimed here about tone, "
|
|
117
|
+
"age or intended use (N-08)."
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _voice(voice_id: str, gender: Gender, index: int, count: int) -> VoiceEntry:
|
|
122
|
+
"""One built-in style, described only as far as it has been verified."""
|
|
123
|
+
word = "male" if gender is Gender.MALE else "female"
|
|
124
|
+
return VoiceEntry(
|
|
125
|
+
voice_id=voice_id,
|
|
126
|
+
gender=gender,
|
|
127
|
+
display_name=f"{word.capitalize()} {index}",
|
|
128
|
+
description=(
|
|
129
|
+
f"Built-in Supertonic 3 style {voice_id}: {word} voice {index} of {count}. "
|
|
130
|
+
f"Reads both Korean and English, since a voice is not specific to one language "
|
|
131
|
+
f"(F-06). {_UNREVIEWED}"
|
|
132
|
+
),
|
|
133
|
+
characterised=False,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
SUPERTONIC_3_VOICES: Final[tuple[VoiceEntry, ...]] = (
|
|
138
|
+
_voice("M1", Gender.MALE, 1, 5),
|
|
139
|
+
_voice("M2", Gender.MALE, 2, 5),
|
|
140
|
+
_voice("M3", Gender.MALE, 3, 5),
|
|
141
|
+
_voice("M4", Gender.MALE, 4, 5),
|
|
142
|
+
_voice("M5", Gender.MALE, 5, 5),
|
|
143
|
+
_voice("F1", Gender.FEMALE, 1, 5),
|
|
144
|
+
_voice("F2", Gender.FEMALE, 2, 5),
|
|
145
|
+
_voice("F3", Gender.FEMALE, 3, 5),
|
|
146
|
+
_voice("F4", Gender.FEMALE, 4, 5),
|
|
147
|
+
_voice("F5", Gender.FEMALE, 5, 5),
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
#: The four ONNX graphs and two tables the runtime loads, plus one style
|
|
151
|
+
#: file per voice. Nothing else in the repository is needed to synthesise,
|
|
152
|
+
#: so nothing else is downloaded: F-73 sizes this cache and N-01 keeps the
|
|
153
|
+
#: network traffic to what preparation actually requires.
|
|
154
|
+
SUPERTONIC_3_FILES: Final[tuple[ModelFile, ...]] = (
|
|
155
|
+
ModelFile(
|
|
156
|
+
"onnx/duration_predictor.onnx",
|
|
157
|
+
"c3eb91414d5ff8a7a239b7fe9e34e7e2bf8a8140d8375ffb14718b1c639325db",
|
|
158
|
+
3_700_147,
|
|
159
|
+
),
|
|
160
|
+
ModelFile(
|
|
161
|
+
"onnx/text_encoder.onnx",
|
|
162
|
+
"c7befd5ea8c3119769e8a6c1486c4edc6a3bc8365c67621c881bbb774b9902ff",
|
|
163
|
+
36_416_150,
|
|
164
|
+
),
|
|
165
|
+
ModelFile(
|
|
166
|
+
"onnx/vector_estimator.onnx",
|
|
167
|
+
"883ac868ea0275ef0e991524dc64f16b3c0376efd7c320af6b53f5b780d7c61c",
|
|
168
|
+
256_534_781,
|
|
169
|
+
),
|
|
170
|
+
ModelFile(
|
|
171
|
+
"onnx/vocoder.onnx",
|
|
172
|
+
"085de76dd8e8d5836d6ca66826601f615939218f90e519f70ee8a36ed2a4c4ba",
|
|
173
|
+
101_424_195,
|
|
174
|
+
),
|
|
175
|
+
ModelFile(
|
|
176
|
+
"onnx/tts.json",
|
|
177
|
+
"42078d3aef1cd43ab43021f3c54f47d2d75ceb4e75f627f118890128b06a0d09",
|
|
178
|
+
8_253,
|
|
179
|
+
),
|
|
180
|
+
ModelFile(
|
|
181
|
+
"onnx/unicode_indexer.json",
|
|
182
|
+
"9bf7346e43883a81f8645c81224f786d43c5b57f3641f6e7671a7d6c493cb24f",
|
|
183
|
+
277_676,
|
|
184
|
+
),
|
|
185
|
+
ModelFile(
|
|
186
|
+
"voice_styles/M1.json",
|
|
187
|
+
"e35604687f5d23694b8e91593a93eec0e4eca6c0b02bb8ed69139ab2ea6b0a5b",
|
|
188
|
+
291_748,
|
|
189
|
+
),
|
|
190
|
+
ModelFile(
|
|
191
|
+
"voice_styles/M2.json",
|
|
192
|
+
"b76cbf62bac707c710cf0ae5aba5e31eea1a6339a9734bfae33ab98499534a50",
|
|
193
|
+
292_055,
|
|
194
|
+
),
|
|
195
|
+
ModelFile(
|
|
196
|
+
"voice_styles/M3.json",
|
|
197
|
+
"ea1ac35ccb91b0d7ecad533a2fbd0eec10c91513d8951e3b25fbba99954e159b",
|
|
198
|
+
290_198,
|
|
199
|
+
),
|
|
200
|
+
ModelFile(
|
|
201
|
+
"voice_styles/M4.json",
|
|
202
|
+
"ca8eefad4fcd989c9379032ff3e50738adc547eeb5e221b82593a6d7b3bac303",
|
|
203
|
+
291_522,
|
|
204
|
+
),
|
|
205
|
+
ModelFile(
|
|
206
|
+
"voice_styles/M5.json",
|
|
207
|
+
"dd22b92740314321f8ae11c5e87f8dd60d060f15dd3a632b5adf77f471f77af2",
|
|
208
|
+
291_469,
|
|
209
|
+
),
|
|
210
|
+
ModelFile(
|
|
211
|
+
"voice_styles/F1.json",
|
|
212
|
+
"bbdec6ee00231c2c742ad05483df5334cab3b52fda3ba38e6a07059c4563dbc2",
|
|
213
|
+
292_046,
|
|
214
|
+
),
|
|
215
|
+
ModelFile(
|
|
216
|
+
"voice_styles/F2.json",
|
|
217
|
+
"7c722c6a72707b1a77f035d67f0d1351ba187738e06f7683e8c72b1df3477fc6",
|
|
218
|
+
292_423,
|
|
219
|
+
),
|
|
220
|
+
ModelFile(
|
|
221
|
+
"voice_styles/F3.json",
|
|
222
|
+
"12f6ef2573baa2defa1128069cb59f203e3ab67c92af77b42df8a0e3a2f7c6ab",
|
|
223
|
+
290_794,
|
|
224
|
+
),
|
|
225
|
+
ModelFile(
|
|
226
|
+
"voice_styles/F4.json",
|
|
227
|
+
"c2fa764c1225a76dfc3e2c73e8aa4f70d9ee48793860eb34c295fff01c2e032b",
|
|
228
|
+
291_808,
|
|
229
|
+
),
|
|
230
|
+
ModelFile(
|
|
231
|
+
"voice_styles/F5.json",
|
|
232
|
+
"45966e73316415626cf41a7d1c6f3b4c70dbc1ba2bee5c1978ef0ce33244fc8d",
|
|
233
|
+
291_479,
|
|
234
|
+
),
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
SUPERTONIC_3: Final = ModelEntry(
|
|
238
|
+
model_id=SUPERTONIC_3_ID,
|
|
239
|
+
display_name="Supertonic 3",
|
|
240
|
+
repo_id="Supertone/supertonic-3",
|
|
241
|
+
revision="724fb5abbf5502583fb520898d45929e62f02c0b",
|
|
242
|
+
files=SUPERTONIC_3_FILES,
|
|
243
|
+
license=SUPERTONIC_3_LICENSE,
|
|
244
|
+
# A.5 measured 44,100 Hz mono, which F-82 then forbids resampling away.
|
|
245
|
+
sample_rate=44_100,
|
|
246
|
+
# The engine advertises 31 languages; the product supports two (F-05),
|
|
247
|
+
# and this list is what F-53 answers and what a request is checked
|
|
248
|
+
# against. Claiming the other 29 would promise pronunciation nobody has
|
|
249
|
+
# tested and N-08 has not reviewed even for these two.
|
|
250
|
+
languages=(Language.KO, Language.EN),
|
|
251
|
+
# F-23's floor is also this model's approved minimum: A.5 measured 0.57
|
|
252
|
+
# GB of working memory, so 2 GiB is comfortable rather than tight. At
|
|
253
|
+
# F-20's lowest CPU setting the baseline is about one thread, which A.5
|
|
254
|
+
# measured at a real-time factor under 0.36 -- still faster than
|
|
255
|
+
# playback, so the model is approved down to the bottom of the range.
|
|
256
|
+
minimum_budget=MinimumBudget(memory_bytes=MEMORY_FLOOR_BYTES, cpu_percent=CPU_PERCENT_MIN),
|
|
257
|
+
voices=SUPERTONIC_3_VOICES,
|
|
258
|
+
engine_model_name="supertonic-3",
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
#: The lineup this build ships. F-04: one model, chosen through the same
|
|
262
|
+
#: machinery a longer list would use.
|
|
263
|
+
MANIFEST: Final = Manifest((SUPERTONIC_3,))
|
|
264
|
+
|
|
265
|
+
DEFAULT_MODEL_ID: Final = SUPERTONIC_3_ID
|
|
266
|
+
|
|
267
|
+
__all__ = [
|
|
268
|
+
"DEFAULT_MODEL_ID",
|
|
269
|
+
"MANIFEST",
|
|
270
|
+
"SUPERTONIC_3",
|
|
271
|
+
"SUPERTONIC_3_ID",
|
|
272
|
+
"SUPERTONIC_3_LICENSE",
|
|
273
|
+
]
|