echoact 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- echoact/__init__.py +3 -0
- echoact/__main__.py +117 -0
- echoact/app.py +315 -0
- echoact/audio/__init__.py +0 -0
- echoact/audio/devices.py +192 -0
- echoact/audio/player.py +611 -0
- echoact/audio/wav.py +854 -0
- echoact/config/__init__.py +0 -0
- echoact/config/budget.py +370 -0
- echoact/config/settings.py +1244 -0
- echoact/db/__init__.py +0 -0
- echoact/db/backup.py +2429 -0
- echoact/db/migrations.py +434 -0
- echoact/db/schema.sql +214 -0
- echoact/db/store.py +2062 -0
- echoact/diagnostics.py +902 -0
- echoact/domain.py +487 -0
- echoact/engine/__init__.py +0 -0
- echoact/engine/container.py +843 -0
- echoact/engine/protocol.py +241 -0
- echoact/engine/runtime.py +324 -0
- echoact/engine/supervisor.py +961 -0
- echoact/engine/worker.py +659 -0
- echoact/errors.py +281 -0
- echoact/instance.py +172 -0
- echoact/jobs/__init__.py +0 -0
- echoact/jobs/engine.py +776 -0
- echoact/jobs/request.py +300 -0
- echoact/mcp/__init__.py +0 -0
- echoact/mcp/__main__.py +50 -0
- echoact/mcp/client.py +202 -0
- echoact/mcp/config.py +112 -0
- echoact/mcp/server.py +340 -0
- echoact/models/__init__.py +0 -0
- echoact/models/catalog.py +273 -0
- echoact/models/manifest.py +278 -0
- echoact/models/registry.py +1551 -0
- echoact/paths.py +93 -0
- echoact/policy.py +189 -0
- echoact/security/__init__.py +0 -0
- echoact/security/credentials.py +930 -0
- echoact/security/ratelimit.py +534 -0
- echoact/service/__init__.py +20 -0
- echoact/service/app.py +182 -0
- echoact/service/deps.py +563 -0
- echoact/service/errors.py +241 -0
- echoact/service/routes.py +1125 -0
- echoact/service/schemas.py +509 -0
- echoact/service/server.py +270 -0
- echoact/text/__init__.py +0 -0
- echoact/text/language.py +44 -0
- echoact/text/loader.py +577 -0
- echoact/text/normalize.py +924 -0
- echoact/text/segment.py +499 -0
- echoact/text/sniff.py +1202 -0
- echoact/ui/__init__.py +0 -0
- echoact/ui/bridge.py +50 -0
- echoact/ui/controls.py +360 -0
- echoact/ui/credential_dialog.py +131 -0
- echoact/ui/fonts.py +94 -0
- echoact/ui/i18n.py +260 -0
- echoact/ui/icons.py +440 -0
- echoact/ui/library.py +1642 -0
- echoact/ui/licence.py +162 -0
- echoact/ui/main_window.py +1202 -0
- echoact/ui/mcp_setup.py +494 -0
- echoact/ui/models_view.py +1142 -0
- echoact/ui/notifications.py +202 -0
- echoact/ui/reading.py +494 -0
- echoact/ui/settings_view.py +2258 -0
- echoact/ui/status_view.py +1193 -0
- echoact/ui/theme.py +579 -0
- echoact/util/__init__.py +0 -0
- echoact/util/ids.py +62 -0
- echoact/util/logging.py +127 -0
- echoact-0.1.0.dist-info/METADATA +162 -0
- echoact-0.1.0.dist-info/RECORD +80 -0
- echoact-0.1.0.dist-info/WHEEL +4 -0
- echoact-0.1.0.dist-info/entry_points.txt +3 -0
- echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
echoact/jobs/engine.py
ADDED
|
@@ -0,0 +1,776 @@
|
|
|
1
|
+
"""The single generation slot, and the machine that drives one job through it.
|
|
2
|
+
|
|
3
|
+
F-47 allows one generation across the GUI and every client combined, so
|
|
4
|
+
this class is the product's narrowest resource and the one place that owns
|
|
5
|
+
it. Everything else -- the window, the REST service, the MCP server --
|
|
6
|
+
asks here and is told yes, or busy.
|
|
7
|
+
|
|
8
|
+
The shape follows from three requirements that pull against each other:
|
|
9
|
+
|
|
10
|
+
* F-12 wants audio as early as possible, so segments are rendered one at a
|
|
11
|
+
time and published the moment each file is closed, rather than at the end.
|
|
12
|
+
* N-22 wants the slot released within five seconds of a cancellation, which
|
|
13
|
+
means the worker can be killed mid-segment. That is only safe because the
|
|
14
|
+
worker owns nothing durable and because the parent named every output file
|
|
15
|
+
before asking for it, so an interrupted write leaves a file this module
|
|
16
|
+
already knows to discard.
|
|
17
|
+
* Section 5.1 says Complete means the audio is ready *and* any requested
|
|
18
|
+
retention succeeded. So the terminal transition happens after the result
|
|
19
|
+
is written and recorded, never when the last segment lands.
|
|
20
|
+
|
|
21
|
+
Cancellation and completion can race. 5.1 keeps whichever terminal state
|
|
22
|
+
was confirmed first and forbids Canceled from overwriting Complete, and
|
|
23
|
+
that is enforced in the store's transition check rather than by ordering
|
|
24
|
+
here -- a rule that depends on two threads interleaving politely is not a
|
|
25
|
+
rule.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import os
|
|
31
|
+
import shutil
|
|
32
|
+
import threading
|
|
33
|
+
from collections.abc import Callable, Iterable
|
|
34
|
+
from dataclasses import dataclass, field
|
|
35
|
+
from enum import StrEnum
|
|
36
|
+
from pathlib import Path
|
|
37
|
+
from typing import Any
|
|
38
|
+
|
|
39
|
+
from ..audio import wav
|
|
40
|
+
from ..config.budget import HaltReason, ResourceSampler, halt_decision, resolve_budget_from_system
|
|
41
|
+
from ..config.settings import Settings
|
|
42
|
+
from ..db.backup import RESTORE_GATE
|
|
43
|
+
from ..db.store import Store
|
|
44
|
+
from ..domain import (
|
|
45
|
+
Budget,
|
|
46
|
+
Job,
|
|
47
|
+
JobState,
|
|
48
|
+
RequestPath,
|
|
49
|
+
RetentionMode,
|
|
50
|
+
Segment,
|
|
51
|
+
TimeRange,
|
|
52
|
+
VoiceSettings,
|
|
53
|
+
)
|
|
54
|
+
from ..engine.supervisor import WorkerSupervisor
|
|
55
|
+
from ..errors import Code, EchoActError
|
|
56
|
+
from ..models.manifest import Manifest
|
|
57
|
+
from ..models.registry import ModelRegistry, ModelState
|
|
58
|
+
from ..paths import audio_dir, temp_dir
|
|
59
|
+
from ..policy import (
|
|
60
|
+
BOUNDED_WAIT_CEILING_S,
|
|
61
|
+
ENGINE_TOTAL_STEPS,
|
|
62
|
+
ONEOFF_RESULT_TTL_S,
|
|
63
|
+
WORKER_RELEASE_DEADLINE_S,
|
|
64
|
+
)
|
|
65
|
+
from ..util import ids
|
|
66
|
+
from ..util.logging import get_logger, job_context
|
|
67
|
+
from .request import JobRequest, plan_segments, validate_request
|
|
68
|
+
|
|
69
|
+
log = get_logger("jobs.engine")
|
|
70
|
+
|
|
71
|
+
#: What a busy caller is told to wait. A.5 measured a real-time factor
|
|
72
|
+
#: near 0.2, so a typical short job is seconds rather than minutes; the
|
|
73
|
+
#: hint is a floor on politeness, not a prediction.
|
|
74
|
+
BUSY_RETRY_AFTER_S = 3.0
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class EventKind(StrEnum):
|
|
78
|
+
ACCEPTED = "accepted"
|
|
79
|
+
STATE = "state"
|
|
80
|
+
SEGMENT = "segment"
|
|
81
|
+
USAGE = "usage"
|
|
82
|
+
FINISHED = "finished"
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass(frozen=True, slots=True)
|
|
86
|
+
class Event:
|
|
87
|
+
"""What happened, for anyone watching.
|
|
88
|
+
|
|
89
|
+
Carries identifiers and numbers, never audio and never body text: the
|
|
90
|
+
GUI, the log, and a notification all consume these, and N-20 keeps body
|
|
91
|
+
text out of the last two.
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
kind: EventKind
|
|
95
|
+
job_id: str
|
|
96
|
+
state: JobState | None = None
|
|
97
|
+
segment_index: int | None = None
|
|
98
|
+
generated: int = 0
|
|
99
|
+
total: int = 0
|
|
100
|
+
request_path: RequestPath | None = None
|
|
101
|
+
client_label: str | None = None
|
|
102
|
+
error_code: str | None = None
|
|
103
|
+
detail: dict[str, Any] = field(default_factory=dict)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@dataclass(slots=True)
|
|
107
|
+
class _Run:
|
|
108
|
+
"""The mutable state of the one job that is running."""
|
|
109
|
+
|
|
110
|
+
job: Job
|
|
111
|
+
segments: list[Segment]
|
|
112
|
+
budget: Budget
|
|
113
|
+
settings: VoiceSettings
|
|
114
|
+
cancel: threading.Event
|
|
115
|
+
finished: threading.Event
|
|
116
|
+
thread: threading.Thread | None = None
|
|
117
|
+
sample_rate: int = 0
|
|
118
|
+
segment_paths: list[str] = field(default_factory=list)
|
|
119
|
+
gaps: list[int] = field(default_factory=list)
|
|
120
|
+
halt: EchoActError | None = None
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
class JobEngine:
|
|
124
|
+
"""One slot, one worker, one job at a time."""
|
|
125
|
+
|
|
126
|
+
def __init__(
|
|
127
|
+
self,
|
|
128
|
+
*,
|
|
129
|
+
store: Store,
|
|
130
|
+
supervisor: WorkerSupervisor,
|
|
131
|
+
registry: ModelRegistry,
|
|
132
|
+
manifest: Manifest,
|
|
133
|
+
settings: Settings,
|
|
134
|
+
work_dir: Path | None = None,
|
|
135
|
+
result_dir: Path | None = None,
|
|
136
|
+
total_steps: int = ENGINE_TOTAL_STEPS,
|
|
137
|
+
) -> None:
|
|
138
|
+
self._store = store
|
|
139
|
+
self._supervisor = supervisor
|
|
140
|
+
self._registry = registry
|
|
141
|
+
self._manifest = manifest
|
|
142
|
+
self._settings = settings
|
|
143
|
+
self._work_dir = Path(work_dir) if work_dir else temp_dir()
|
|
144
|
+
self._result_dir = Path(result_dir) if result_dir else audio_dir()
|
|
145
|
+
self._total_steps = total_steps
|
|
146
|
+
|
|
147
|
+
self._lock = threading.RLock()
|
|
148
|
+
self._run: _Run | None = None
|
|
149
|
+
self._listeners: list[Callable[[Event], None]] = []
|
|
150
|
+
self._closing = False
|
|
151
|
+
|
|
152
|
+
# ------------------------------------------------------------------
|
|
153
|
+
# Observation
|
|
154
|
+
# ------------------------------------------------------------------
|
|
155
|
+
|
|
156
|
+
def listen(self, fn: Callable[[Event], None]) -> Callable[[], None]:
|
|
157
|
+
"""Subscribe. Returns an unsubscribe callable.
|
|
158
|
+
|
|
159
|
+
Listeners are called on the run thread, so a GUI listener must
|
|
160
|
+
marshal to the main thread rather than touch a widget here.
|
|
161
|
+
"""
|
|
162
|
+
self._listeners.append(fn)
|
|
163
|
+
|
|
164
|
+
def off() -> None:
|
|
165
|
+
with self._lock:
|
|
166
|
+
if fn in self._listeners:
|
|
167
|
+
self._listeners.remove(fn)
|
|
168
|
+
|
|
169
|
+
return off
|
|
170
|
+
|
|
171
|
+
def _emit(self, event: Event) -> None:
|
|
172
|
+
for fn in list(self._listeners):
|
|
173
|
+
try:
|
|
174
|
+
fn(event)
|
|
175
|
+
except Exception as exc: # noqa: BLE001 - a listener must not stop a job
|
|
176
|
+
log.warning("job listener failed: %s", type(exc).__name__)
|
|
177
|
+
|
|
178
|
+
# ------------------------------------------------------------------
|
|
179
|
+
# State
|
|
180
|
+
# ------------------------------------------------------------------
|
|
181
|
+
|
|
182
|
+
@property
|
|
183
|
+
def busy(self) -> bool:
|
|
184
|
+
with self._lock:
|
|
185
|
+
return self._run is not None
|
|
186
|
+
|
|
187
|
+
def current(self) -> Job | None:
|
|
188
|
+
with self._lock:
|
|
189
|
+
return self._run.job if self._run else None
|
|
190
|
+
|
|
191
|
+
def apply_settings(self, settings: Settings) -> None:
|
|
192
|
+
"""F-78: a change during a job applies to the *next* job.
|
|
193
|
+
|
|
194
|
+
Nothing here touches a running job, which is the whole point: the
|
|
195
|
+
job recorded the budget it started under and keeps it.
|
|
196
|
+
"""
|
|
197
|
+
self._settings = settings
|
|
198
|
+
|
|
199
|
+
def would_need_load(self, settings: VoiceSettings | None = None) -> bool:
|
|
200
|
+
"""F-17: whether the next job pays for a model load."""
|
|
201
|
+
voice = settings or self._settings.voice
|
|
202
|
+
try:
|
|
203
|
+
budget = resolve_budget_from_system(self._settings)
|
|
204
|
+
except EchoActError:
|
|
205
|
+
return True
|
|
206
|
+
return self._supervisor.would_need_load(voice.model_id, budget)
|
|
207
|
+
|
|
208
|
+
# ------------------------------------------------------------------
|
|
209
|
+
# Submission (F-47, F-49)
|
|
210
|
+
# ------------------------------------------------------------------
|
|
211
|
+
|
|
212
|
+
def submit(self, request: JobRequest) -> tuple[Job, bool]:
|
|
213
|
+
"""Accept a job, or refuse it. Returns ``(job, created)``.
|
|
214
|
+
|
|
215
|
+
Order matters and is not arbitrary. The slot is taken *before* the
|
|
216
|
+
duplicate-prevention key is claimed, because a refusal for busy must
|
|
217
|
+
not consume the key -- F-49's record is meant to identify a job that
|
|
218
|
+
exists, and a busy response creates none. If the claim then finds
|
|
219
|
+
an existing job, the slot is handed straight back.
|
|
220
|
+
"""
|
|
221
|
+
# 5.3: during a restore, new generation and edits are blocked.
|
|
222
|
+
# Checked before anything else because it is the cheapest refusal
|
|
223
|
+
# and the only one that is temporary by construction -- the
|
|
224
|
+
# restore will finish, so this is retryable where the others are
|
|
225
|
+
# the caller's to fix.
|
|
226
|
+
RESTORE_GATE.require_idle("Generating speech")
|
|
227
|
+
entry = validate_request(request, self._manifest)
|
|
228
|
+
budget = resolve_budget_from_system(self._settings)
|
|
229
|
+
|
|
230
|
+
runnable, why = self._registry.can_run(request.settings.model_id, budget)
|
|
231
|
+
if not runnable:
|
|
232
|
+
# F-04: unavailable with the reason, never silently substituted.
|
|
233
|
+
raise EchoActError(Code.MODEL_OVER_BUDGET, why or None)
|
|
234
|
+
self._check_model_preparable(request.settings.model_id)
|
|
235
|
+
|
|
236
|
+
with self._lock:
|
|
237
|
+
if self._closing:
|
|
238
|
+
raise EchoActError(Code.SHUTTING_DOWN)
|
|
239
|
+
taken = self._run is None
|
|
240
|
+
if taken:
|
|
241
|
+
self._run = _PLACEHOLDER
|
|
242
|
+
if not taken:
|
|
243
|
+
raise EchoActError(Code.BUSY, retry_after_s=BUSY_RETRY_AFTER_S)
|
|
244
|
+
|
|
245
|
+
try:
|
|
246
|
+
job = Job(
|
|
247
|
+
job_id=ids.job_id(),
|
|
248
|
+
kind=request.kind,
|
|
249
|
+
request_path=request.request_path,
|
|
250
|
+
owner_client_id=request.owner_client_id,
|
|
251
|
+
state=JobState.ACCEPTED,
|
|
252
|
+
source_text=request.text,
|
|
253
|
+
settings=request.settings,
|
|
254
|
+
budget=budget,
|
|
255
|
+
retention=request.retention,
|
|
256
|
+
created_at=ids.now(),
|
|
257
|
+
client_label=request.client_label,
|
|
258
|
+
idempotency_key=request.idempotency_key,
|
|
259
|
+
)
|
|
260
|
+
stored, created = self._store.claim_job(
|
|
261
|
+
job,
|
|
262
|
+
client_id=request.owner_client_id,
|
|
263
|
+
key=request.idempotency_key,
|
|
264
|
+
request_digest=request.digest(),
|
|
265
|
+
)
|
|
266
|
+
if not created:
|
|
267
|
+
# 4.2: a repeat returns the existing job and its state,
|
|
268
|
+
# including a terminal or expired one, and regenerates
|
|
269
|
+
# nothing. A new key is needed to generate again.
|
|
270
|
+
self._release_slot()
|
|
271
|
+
return stored, False
|
|
272
|
+
|
|
273
|
+
segments = plan_segments(request.text, request.settings)
|
|
274
|
+
stored.segments = list(self._store.insert_segments(stored.job_id, segments))
|
|
275
|
+
stored.total_segments = len(stored.segments)
|
|
276
|
+
self._store.set_job_budget(stored.job_id, budget)
|
|
277
|
+
|
|
278
|
+
run = _Run(
|
|
279
|
+
job=stored,
|
|
280
|
+
segments=stored.segments,
|
|
281
|
+
budget=budget,
|
|
282
|
+
settings=request.settings,
|
|
283
|
+
cancel=threading.Event(),
|
|
284
|
+
finished=threading.Event(),
|
|
285
|
+
sample_rate=entry.sample_rate,
|
|
286
|
+
)
|
|
287
|
+
with self._lock:
|
|
288
|
+
self._run = run
|
|
289
|
+
self._emit(
|
|
290
|
+
Event(
|
|
291
|
+
EventKind.ACCEPTED,
|
|
292
|
+
stored.job_id,
|
|
293
|
+
state=stored.state,
|
|
294
|
+
total=len(segments),
|
|
295
|
+
request_path=stored.request_path,
|
|
296
|
+
client_label=stored.client_label,
|
|
297
|
+
)
|
|
298
|
+
)
|
|
299
|
+
run.thread = threading.Thread(
|
|
300
|
+
target=self._run_job, args=(run,), name=f"echoact-job-{stored.job_id}", daemon=True
|
|
301
|
+
)
|
|
302
|
+
run.thread.start()
|
|
303
|
+
return stored, True
|
|
304
|
+
except BaseException:
|
|
305
|
+
self._release_slot()
|
|
306
|
+
raise
|
|
307
|
+
|
|
308
|
+
def _check_model_preparable(self, model_id: str) -> None:
|
|
309
|
+
"""Refuse at acceptance what could only fail later anyway.
|
|
310
|
+
|
|
311
|
+
5.3 says a request for a model that is not downloaded is *refused*,
|
|
312
|
+
not accepted and then failed, and the same reasoning covers a
|
|
313
|
+
licence nobody has accepted: both are knowable now, neither can
|
|
314
|
+
change while the job waits, and accepting the job would consume
|
|
315
|
+
F-49's key on something that can never run and hand the caller a
|
|
316
|
+
job id to poll instead of an answer.
|
|
317
|
+
|
|
318
|
+
Deliberately the shallow check. N-22 gives acceptance one second
|
|
319
|
+
at p95 and a deep verify hashes 385 MB; the deep pass still runs in
|
|
320
|
+
``_prepare``, where its cost belongs.
|
|
321
|
+
"""
|
|
322
|
+
status = self._registry.status(model_id, deep=False)
|
|
323
|
+
if status.state is ModelState.CORRUPT:
|
|
324
|
+
raise EchoActError(Code.MODEL_CORRUPT, detail={"model_id": model_id})
|
|
325
|
+
if status.state is not ModelState.READY:
|
|
326
|
+
raise EchoActError(Code.MODEL_NOT_READY, detail={"model_id": model_id})
|
|
327
|
+
if status.license_acceptance_required and not status.license_accepted:
|
|
328
|
+
raise EchoActError(
|
|
329
|
+
Code.MODEL_LICENSE_NOT_ACCEPTED,
|
|
330
|
+
detail={"model_id": model_id, "license": status.license_name},
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
def _release_slot(self) -> None:
|
|
334
|
+
with self._lock:
|
|
335
|
+
self._run = None
|
|
336
|
+
|
|
337
|
+
# ------------------------------------------------------------------
|
|
338
|
+
# Waiting (F-88)
|
|
339
|
+
# ------------------------------------------------------------------
|
|
340
|
+
|
|
341
|
+
def wait(self, job_id: str, timeout_s: float) -> Job:
|
|
342
|
+
"""Wait up to ``timeout_s`` for a terminal state, then answer anyway.
|
|
343
|
+
|
|
344
|
+
F-88 makes this an optimisation and never a different lifecycle:
|
|
345
|
+
the job is untouched whether the bound passes or not, and the caller
|
|
346
|
+
gets the same job either way. It also never waits on model
|
|
347
|
+
preparation -- the wait begins only once generation is under way,
|
|
348
|
+
which is why the deadline is re-checked against the job's state
|
|
349
|
+
rather than simply slept through.
|
|
350
|
+
"""
|
|
351
|
+
bound = max(0.0, min(float(timeout_s), BOUNDED_WAIT_CEILING_S))
|
|
352
|
+
with self._lock:
|
|
353
|
+
run = self._run
|
|
354
|
+
if run is None or run is _PLACEHOLDER or run.job.job_id != job_id:
|
|
355
|
+
return self._store.get_job(job_id)
|
|
356
|
+
if bound > 0:
|
|
357
|
+
run.finished.wait(bound)
|
|
358
|
+
return self._store.get_job(job_id)
|
|
359
|
+
|
|
360
|
+
# ------------------------------------------------------------------
|
|
361
|
+
# Cancellation (F-15, F-49, N-22)
|
|
362
|
+
# ------------------------------------------------------------------
|
|
363
|
+
|
|
364
|
+
def cancel(self, job_id: str) -> Job:
|
|
365
|
+
"""Cancel a job. Repeating it adds no further side effects (F-49)."""
|
|
366
|
+
job = self._store.get_job(job_id, include_segments=False)
|
|
367
|
+
if job.state.is_terminal:
|
|
368
|
+
return job
|
|
369
|
+
with self._lock:
|
|
370
|
+
run = self._run
|
|
371
|
+
running = run is not None and run is not _PLACEHOLDER and run.job.job_id == job_id
|
|
372
|
+
if not running:
|
|
373
|
+
# Accepted but not the current job: only possible after an
|
|
374
|
+
# abnormal termination, which F-45 reconciles to Interrupted.
|
|
375
|
+
self._store.update_job_state(job_id, JobState.CANCELING, force=True)
|
|
376
|
+
self._store.update_job_state(job_id, JobState.CANCELED)
|
|
377
|
+
return self._store.get_job(job_id, include_segments=False)
|
|
378
|
+
|
|
379
|
+
assert run is not None
|
|
380
|
+
self._set_state(run, JobState.CANCELING)
|
|
381
|
+
run.cancel.set()
|
|
382
|
+
# Mid-segment is the case N-22's five seconds is written for: the
|
|
383
|
+
# worker is inside an ONNX call that will not return promptly, so
|
|
384
|
+
# asking politely is not enough.
|
|
385
|
+
elapsed = self._supervisor.kill()
|
|
386
|
+
log.info(job_context(job_id, "canceling", release_s=round(elapsed, 3)))
|
|
387
|
+
run.finished.wait(WORKER_RELEASE_DEADLINE_S)
|
|
388
|
+
return self._store.get_job(job_id, include_segments=False)
|
|
389
|
+
|
|
390
|
+
def release_model(self) -> None:
|
|
391
|
+
"""F-19's explicit release. Refused while a job is running, because
|
|
392
|
+
the job would then fail rather than the model being freed."""
|
|
393
|
+
if self.busy:
|
|
394
|
+
raise EchoActError(Code.BUSY, retry_after_s=BUSY_RETRY_AFTER_S)
|
|
395
|
+
self._supervisor.unload()
|
|
396
|
+
|
|
397
|
+
def shutdown(self) -> None:
|
|
398
|
+
"""F-52: report and stop. The worker goes first, so nothing is
|
|
399
|
+
still writing when the database closes."""
|
|
400
|
+
with self._lock:
|
|
401
|
+
self._closing = True
|
|
402
|
+
run = self._run
|
|
403
|
+
if run is not None and run is not _PLACEHOLDER:
|
|
404
|
+
run.cancel.set()
|
|
405
|
+
self._supervisor.kill()
|
|
406
|
+
run.finished.wait(WORKER_RELEASE_DEADLINE_S)
|
|
407
|
+
else:
|
|
408
|
+
self._supervisor.kill()
|
|
409
|
+
|
|
410
|
+
# ------------------------------------------------------------------
|
|
411
|
+
# The run
|
|
412
|
+
# ------------------------------------------------------------------
|
|
413
|
+
|
|
414
|
+
def _run_job(self, run: _Run) -> None:
|
|
415
|
+
job_id = run.job.job_id
|
|
416
|
+
try:
|
|
417
|
+
self._prepare(run)
|
|
418
|
+
self._generate(run)
|
|
419
|
+
self._finish(run)
|
|
420
|
+
except EchoActError as exc:
|
|
421
|
+
self._fail(run, exc)
|
|
422
|
+
except Exception as exc: # noqa: BLE001 - a run thread must not vanish
|
|
423
|
+
log.exception("job %s failed unexpectedly", job_id)
|
|
424
|
+
self._fail(run, EchoActError(Code.GENERATION_FAILED, cause=exc))
|
|
425
|
+
finally:
|
|
426
|
+
self._cleanup(run)
|
|
427
|
+
run.finished.set()
|
|
428
|
+
self._release_slot()
|
|
429
|
+
self._announce_finished(run)
|
|
430
|
+
|
|
431
|
+
def _announce_finished(self, run: _Run) -> None:
|
|
432
|
+
"""The last word on a job, read back from the database.
|
|
433
|
+
|
|
434
|
+
Read back rather than assumed, because the terminal state may not
|
|
435
|
+
be the one this thread chose: 5.1 lets a cancellation and a
|
|
436
|
+
completion race and keeps whichever was confirmed first.
|
|
437
|
+
|
|
438
|
+
Tolerant of a closed database on purpose. Shutdown kills the worker
|
|
439
|
+
and then closes the store, so this can be the last thing running
|
|
440
|
+
during an exit, and a job that has already ended is not worth an
|
|
441
|
+
exception on the way out.
|
|
442
|
+
"""
|
|
443
|
+
job_id = run.job.job_id
|
|
444
|
+
try:
|
|
445
|
+
final = self._store.get_job(
|
|
446
|
+
job_id, include_source_text=False, include_segments=False
|
|
447
|
+
)
|
|
448
|
+
state, generated = final.state, final.generated_segments
|
|
449
|
+
total, error_code = final.total_segments, final.error_code
|
|
450
|
+
path, label = final.request_path, final.client_label
|
|
451
|
+
except EchoActError:
|
|
452
|
+
state = JobState.CANCELED if run.cancel.is_set() else run.job.state
|
|
453
|
+
generated = sum(1 for s in run.segments if s.ready)
|
|
454
|
+
total, error_code = len(run.segments), None
|
|
455
|
+
path, label = run.job.request_path, run.job.client_label
|
|
456
|
+
self._emit(
|
|
457
|
+
Event(
|
|
458
|
+
EventKind.FINISHED,
|
|
459
|
+
job_id,
|
|
460
|
+
state=state,
|
|
461
|
+
generated=generated,
|
|
462
|
+
total=total,
|
|
463
|
+
request_path=path,
|
|
464
|
+
client_label=label,
|
|
465
|
+
error_code=error_code,
|
|
466
|
+
)
|
|
467
|
+
)
|
|
468
|
+
|
|
469
|
+
def _prepare(self, run: _Run) -> None:
|
|
470
|
+
"""Resolve the model and load it, per F-09, F-17, F-84."""
|
|
471
|
+
self._check_cancelled(run)
|
|
472
|
+
self._set_state(run, JobState.PREPARING_MODEL)
|
|
473
|
+
|
|
474
|
+
model_id = run.settings.model_id
|
|
475
|
+
# F-09/5.3: the engine never downloads. A model that is not present
|
|
476
|
+
# is refused here, and preparing it is the owner's explicit action
|
|
477
|
+
# in the model screen.
|
|
478
|
+
model_dir, from_package = self._registry.resolve_dir(model_id)
|
|
479
|
+
if from_package:
|
|
480
|
+
log.info("model %s served from the package cache", model_id)
|
|
481
|
+
|
|
482
|
+
if self._supervisor.would_need_load(model_id, run.budget):
|
|
483
|
+
loaded = self._supervisor.load(model_id, model_dir, run.budget)
|
|
484
|
+
run.sample_rate = loaded.sample_rate
|
|
485
|
+
else:
|
|
486
|
+
current = self._supervisor.loaded_model
|
|
487
|
+
if current is not None:
|
|
488
|
+
run.sample_rate = current.sample_rate
|
|
489
|
+
self._check_cancelled(run)
|
|
490
|
+
|
|
491
|
+
def _generate(self, run: _Run) -> None:
|
|
492
|
+
self._set_state(run, JobState.GENERATING)
|
|
493
|
+
job_id = run.job.job_id
|
|
494
|
+
sampler = ResourceSampler(self._supervisor.worker_pid)
|
|
495
|
+
start_frame = 0
|
|
496
|
+
|
|
497
|
+
for seg in run.segments:
|
|
498
|
+
self._check_cancelled(run)
|
|
499
|
+
self._check_resources(run, sampler)
|
|
500
|
+
|
|
501
|
+
gap_frames = wav.frames_for_ms(seg.trailing_silence_ms, run.sample_rate)
|
|
502
|
+
if not seg.is_spoken:
|
|
503
|
+
# F-27: a range that produces no audio still exists on the
|
|
504
|
+
# timeline, attached to its neighbour. Nothing is sent to
|
|
505
|
+
# the engine, which A.5 showed would otherwise vocalise it.
|
|
506
|
+
span = TimeRange(
|
|
507
|
+
wav.ms_for_frames(start_frame, run.sample_rate),
|
|
508
|
+
wav.ms_for_frames(start_frame + gap_frames, run.sample_rate),
|
|
509
|
+
)
|
|
510
|
+
self._store.mark_segment_ready(
|
|
511
|
+
job_id, seg.index, time=span, audio_path=None, frame_count=0
|
|
512
|
+
)
|
|
513
|
+
seg.time, seg.ready, seg.frame_count = span, True, 0
|
|
514
|
+
run.segment_paths.append("")
|
|
515
|
+
run.gaps.append(seg.trailing_silence_ms)
|
|
516
|
+
start_frame += gap_frames
|
|
517
|
+
self._emit_segment(run, seg)
|
|
518
|
+
continue
|
|
519
|
+
|
|
520
|
+
out = self._work_dir / job_id / f"{seg.index:05d}.wav"
|
|
521
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
522
|
+
reply = self._supervisor.synthesize(
|
|
523
|
+
job_id=job_id,
|
|
524
|
+
segment_index=seg.index,
|
|
525
|
+
text=seg.spoken_text,
|
|
526
|
+
lang=seg.language,
|
|
527
|
+
voice_id=run.settings.voice_id,
|
|
528
|
+
speed=_engine_speed(run.settings),
|
|
529
|
+
out_path=out,
|
|
530
|
+
total_steps=self._total_steps,
|
|
531
|
+
)
|
|
532
|
+
self._check_cancelled(run)
|
|
533
|
+
|
|
534
|
+
span = TimeRange(
|
|
535
|
+
wav.ms_for_frames(start_frame, run.sample_rate),
|
|
536
|
+
wav.ms_for_frames(start_frame + reply.frame_count + gap_frames, run.sample_rate),
|
|
537
|
+
)
|
|
538
|
+
self._store.mark_segment_ready(
|
|
539
|
+
job_id,
|
|
540
|
+
seg.index,
|
|
541
|
+
time=span,
|
|
542
|
+
audio_path=str(out),
|
|
543
|
+
frame_count=reply.frame_count,
|
|
544
|
+
)
|
|
545
|
+
seg.time = span
|
|
546
|
+
seg.ready = True
|
|
547
|
+
seg.frame_count = reply.frame_count
|
|
548
|
+
seg.audio_path = str(out)
|
|
549
|
+
run.segment_paths.append(str(out))
|
|
550
|
+
run.gaps.append(seg.trailing_silence_ms)
|
|
551
|
+
start_frame += reply.frame_count + gap_frames
|
|
552
|
+
self._emit_segment(run, seg)
|
|
553
|
+
|
|
554
|
+
def _finish(self, run: _Run) -> None:
|
|
555
|
+
"""Concatenate, record, and only then call the job Complete.
|
|
556
|
+
|
|
557
|
+
5.1: "If generation finishes but saving fails, the job is not marked
|
|
558
|
+
Complete." So the transition is last, after the file exists and the
|
|
559
|
+
row is written.
|
|
560
|
+
"""
|
|
561
|
+
self._check_cancelled(run)
|
|
562
|
+
job_id = run.job.job_id
|
|
563
|
+
retained = run.job.retention is RetentionMode.RETAINED
|
|
564
|
+
root = self._result_dir if retained else self._work_dir
|
|
565
|
+
out = root / job_id / "result.wav"
|
|
566
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
567
|
+
|
|
568
|
+
spoken = [(p, g) for p, g in zip(run.segment_paths, run.gaps, strict=True) if p]
|
|
569
|
+
if not spoken:
|
|
570
|
+
raise EchoActError(
|
|
571
|
+
Code.GENERATION_FAILED, "The text produced no audio.", detail={"job_id": job_id}
|
|
572
|
+
)
|
|
573
|
+
report = wav.concatenate(
|
|
574
|
+
[p for p, _ in spoken],
|
|
575
|
+
gaps_ms=[g for _, g in spoken],
|
|
576
|
+
out_path=out,
|
|
577
|
+
sample_rate=run.sample_rate,
|
|
578
|
+
)
|
|
579
|
+
now = ids.now()
|
|
580
|
+
from ..domain import Result
|
|
581
|
+
|
|
582
|
+
result = Result(
|
|
583
|
+
result_id=ids.result_id(),
|
|
584
|
+
job_id=job_id,
|
|
585
|
+
sample_rate=report.sample_rate,
|
|
586
|
+
channels=1,
|
|
587
|
+
sample_width_bits=16,
|
|
588
|
+
frame_count=report.frame_count,
|
|
589
|
+
byte_size=report.byte_size,
|
|
590
|
+
digest=wav.digest(out),
|
|
591
|
+
created_at=now,
|
|
592
|
+
expires_at=None if retained else now + ONEOFF_RESULT_TTL_S,
|
|
593
|
+
relative_path=str(out),
|
|
594
|
+
)
|
|
595
|
+
self._store.attach_result(result)
|
|
596
|
+
self._set_state(run, JobState.COMPLETE)
|
|
597
|
+
|
|
598
|
+
def _fail(self, run: _Run, error: EchoActError) -> None:
|
|
599
|
+
job_id = run.job.job_id
|
|
600
|
+
if run.cancel.is_set():
|
|
601
|
+
# A cancellation in flight looks like a failure from inside the
|
|
602
|
+
# loop. 5.1 wants Canceled, and Complete is never overwritten
|
|
603
|
+
# because the store refuses that transition.
|
|
604
|
+
try:
|
|
605
|
+
self._store.update_job_state(job_id, JobState.CANCELED)
|
|
606
|
+
except (AssertionError, EchoActError):
|
|
607
|
+
pass
|
|
608
|
+
self._emit_state(run, JobState.CANCELED)
|
|
609
|
+
return
|
|
610
|
+
log.warning(job_context(job_id, "failed", code=error.code.value))
|
|
611
|
+
try:
|
|
612
|
+
self._store.record_job_error(job_id, error.code, error.message)
|
|
613
|
+
self._store.update_job_state(job_id, JobState.FAILED)
|
|
614
|
+
except (AssertionError, EchoActError):
|
|
615
|
+
pass
|
|
616
|
+
# F-19: the model is released on error as well as on cancellation.
|
|
617
|
+
self._supervisor.kill()
|
|
618
|
+
self._emit_state(run, JobState.FAILED, error_code=error.code.value)
|
|
619
|
+
|
|
620
|
+
def _cleanup(self, run: _Run) -> None:
|
|
621
|
+
"""Remove the per-job scratch directory once nothing needs it.
|
|
622
|
+
|
|
623
|
+
A retained job's segment audio stays: F-55 lets a client fetch
|
|
624
|
+
individual segments, and F-31 replays a retained job with its
|
|
625
|
+
mapping. A one-off job's scratch is the result's own home until it
|
|
626
|
+
expires, so it is left for the sweeper rather than deleted here.
|
|
627
|
+
"""
|
|
628
|
+
if run.cancel.is_set() and run.job.retention is RetentionMode.ONE_OFF:
|
|
629
|
+
scratch = self._work_dir / run.job.job_id
|
|
630
|
+
shutil.rmtree(scratch, ignore_errors=True)
|
|
631
|
+
|
|
632
|
+
# -- helpers ---------------------------------------------------------
|
|
633
|
+
|
|
634
|
+
def _check_cancelled(self, run: _Run) -> None:
|
|
635
|
+
if run.cancel.is_set():
|
|
636
|
+
raise EchoActError(Code.GENERATION_FAILED, "Canceled.")
|
|
637
|
+
|
|
638
|
+
def _check_resources(self, run: _Run, sampler: ResourceSampler) -> None:
|
|
639
|
+
"""F-23 and N-04, checked between segments.
|
|
640
|
+
|
|
641
|
+
Between rather than during: the container is what stops a spike
|
|
642
|
+
inside a single ONNX call, and N-03 is explicit that the enforced
|
|
643
|
+
ceiling is the operating system's job. This is the part that
|
|
644
|
+
reports *why* and releases the model.
|
|
645
|
+
"""
|
|
646
|
+
sample = sampler.poll()
|
|
647
|
+
if sample is None:
|
|
648
|
+
return
|
|
649
|
+
decision = halt_decision(sample, run.budget)
|
|
650
|
+
if decision.reason is HaltReason.NONE:
|
|
651
|
+
return
|
|
652
|
+
raise decision.to_error()
|
|
653
|
+
|
|
654
|
+
def _set_state(self, run: _Run, state: JobState) -> None:
|
|
655
|
+
self._store.update_job_state(run.job.job_id, state)
|
|
656
|
+
run.job.state = state
|
|
657
|
+
self._emit_state(run, state)
|
|
658
|
+
|
|
659
|
+
def _emit_state(self, run: _Run, state: JobState, error_code: str | None = None) -> None:
|
|
660
|
+
self._emit(
|
|
661
|
+
Event(
|
|
662
|
+
EventKind.STATE,
|
|
663
|
+
run.job.job_id,
|
|
664
|
+
state=state,
|
|
665
|
+
generated=sum(1 for s in run.segments if s.ready),
|
|
666
|
+
total=len(run.segments),
|
|
667
|
+
request_path=run.job.request_path,
|
|
668
|
+
client_label=run.job.client_label,
|
|
669
|
+
error_code=error_code,
|
|
670
|
+
)
|
|
671
|
+
)
|
|
672
|
+
|
|
673
|
+
def _emit_segment(self, run: _Run, seg: Segment) -> None:
|
|
674
|
+
self._emit(
|
|
675
|
+
Event(
|
|
676
|
+
EventKind.SEGMENT,
|
|
677
|
+
run.job.job_id,
|
|
678
|
+
segment_index=seg.index,
|
|
679
|
+
generated=sum(1 for s in run.segments if s.ready),
|
|
680
|
+
total=len(run.segments),
|
|
681
|
+
request_path=run.job.request_path,
|
|
682
|
+
client_label=run.job.client_label,
|
|
683
|
+
detail={
|
|
684
|
+
"start_ms": seg.time.start_ms if seg.time else 0,
|
|
685
|
+
"end_ms": seg.time.end_ms if seg.time else 0,
|
|
686
|
+
"frame_count": seg.frame_count,
|
|
687
|
+
"audio_path": seg.audio_path or "",
|
|
688
|
+
},
|
|
689
|
+
)
|
|
690
|
+
)
|
|
691
|
+
|
|
692
|
+
|
|
693
|
+
def _engine_speed(settings: VoiceSettings) -> float:
|
|
694
|
+
"""F-07 and F-08 combined into the one number the engine takes.
|
|
695
|
+
|
|
696
|
+
A style is a preset over tempo among other things, so the two multiply
|
|
697
|
+
-- and the product is clamped back into F-07's advertised range, since
|
|
698
|
+
a style must never take tempo somewhere the user is told is impossible.
|
|
699
|
+
"""
|
|
700
|
+
from ..text.segment import effective_tempo
|
|
701
|
+
|
|
702
|
+
return effective_tempo(settings)
|
|
703
|
+
|
|
704
|
+
|
|
705
|
+
class _Placeholder:
|
|
706
|
+
"""Marks the slot as taken between the check and the real run object.
|
|
707
|
+
|
|
708
|
+
Without it, two submissions could both find the slot free while the
|
|
709
|
+
first was still building its job -- the window is small and, with one
|
|
710
|
+
slot and clients that retry, exactly the window that gets hit.
|
|
711
|
+
"""
|
|
712
|
+
|
|
713
|
+
__slots__ = ()
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
_PLACEHOLDER: Any = _Placeholder()
|
|
717
|
+
|
|
718
|
+
|
|
719
|
+
def expire_one_off_results(store: Store, work_root: Path | None = None) -> int:
|
|
720
|
+
"""4.1's one-off lifetime, swept.
|
|
721
|
+
|
|
722
|
+
Returns how many results were removed. Retained data is never touched:
|
|
723
|
+
N-16 forbids deleting explicitly retained results to reclaim space, and
|
|
724
|
+
this is a lifetime sweep rather than a space one.
|
|
725
|
+
"""
|
|
726
|
+
root = Path(work_root) if work_root else temp_dir()
|
|
727
|
+
now = ids.now()
|
|
728
|
+
removed = 0
|
|
729
|
+
for job_id, result_id, _expires_at, path in _expired(store, now):
|
|
730
|
+
try:
|
|
731
|
+
store.delete_result(result_id)
|
|
732
|
+
except EchoActError:
|
|
733
|
+
continue
|
|
734
|
+
removed += 1
|
|
735
|
+
try:
|
|
736
|
+
if path:
|
|
737
|
+
Path(path).unlink(missing_ok=True)
|
|
738
|
+
scratch = root / job_id
|
|
739
|
+
if scratch.is_dir():
|
|
740
|
+
shutil.rmtree(scratch, ignore_errors=True)
|
|
741
|
+
except OSError:
|
|
742
|
+
pass
|
|
743
|
+
return removed
|
|
744
|
+
|
|
745
|
+
|
|
746
|
+
def _expired(store: Store, now: float) -> Iterable[tuple[str, str, float, str]]:
|
|
747
|
+
conn = store._conn() # noqa: SLF001 - the sweeper is part of the storage layer
|
|
748
|
+
rows = conn.execute(
|
|
749
|
+
"SELECT job_id, result_id, expires_at, relative_path FROM results"
|
|
750
|
+
" WHERE expires_at IS NOT NULL AND expires_at <= ?",
|
|
751
|
+
(now,),
|
|
752
|
+
).fetchall()
|
|
753
|
+
return [(r["job_id"], r["result_id"], r["expires_at"], r["relative_path"]) for r in rows]
|
|
754
|
+
|
|
755
|
+
|
|
756
|
+
def clear_temp_tree(work_root: Path | None = None) -> int:
|
|
757
|
+
"""N-02: whatever a forced termination left behind, on relaunch.
|
|
758
|
+
|
|
759
|
+
Returns the number of entries removed. Only the scratch tree is
|
|
760
|
+
touched; retained audio lives elsewhere precisely so that this can be
|
|
761
|
+
unconditional.
|
|
762
|
+
"""
|
|
763
|
+
root = Path(work_root) if work_root else temp_dir()
|
|
764
|
+
if not root.is_dir():
|
|
765
|
+
return 0
|
|
766
|
+
removed = 0
|
|
767
|
+
for child in root.iterdir():
|
|
768
|
+
try:
|
|
769
|
+
if child.is_dir():
|
|
770
|
+
shutil.rmtree(child, ignore_errors=True)
|
|
771
|
+
else:
|
|
772
|
+
os.unlink(child)
|
|
773
|
+
removed += 1
|
|
774
|
+
except OSError:
|
|
775
|
+
continue
|
|
776
|
+
return removed
|