echoact 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. echoact/__init__.py +3 -0
  2. echoact/__main__.py +117 -0
  3. echoact/app.py +315 -0
  4. echoact/audio/__init__.py +0 -0
  5. echoact/audio/devices.py +192 -0
  6. echoact/audio/player.py +611 -0
  7. echoact/audio/wav.py +854 -0
  8. echoact/config/__init__.py +0 -0
  9. echoact/config/budget.py +370 -0
  10. echoact/config/settings.py +1244 -0
  11. echoact/db/__init__.py +0 -0
  12. echoact/db/backup.py +2429 -0
  13. echoact/db/migrations.py +434 -0
  14. echoact/db/schema.sql +214 -0
  15. echoact/db/store.py +2062 -0
  16. echoact/diagnostics.py +902 -0
  17. echoact/domain.py +487 -0
  18. echoact/engine/__init__.py +0 -0
  19. echoact/engine/container.py +843 -0
  20. echoact/engine/protocol.py +241 -0
  21. echoact/engine/runtime.py +324 -0
  22. echoact/engine/supervisor.py +961 -0
  23. echoact/engine/worker.py +659 -0
  24. echoact/errors.py +281 -0
  25. echoact/instance.py +172 -0
  26. echoact/jobs/__init__.py +0 -0
  27. echoact/jobs/engine.py +776 -0
  28. echoact/jobs/request.py +300 -0
  29. echoact/mcp/__init__.py +0 -0
  30. echoact/mcp/__main__.py +50 -0
  31. echoact/mcp/client.py +202 -0
  32. echoact/mcp/config.py +112 -0
  33. echoact/mcp/server.py +340 -0
  34. echoact/models/__init__.py +0 -0
  35. echoact/models/catalog.py +273 -0
  36. echoact/models/manifest.py +278 -0
  37. echoact/models/registry.py +1551 -0
  38. echoact/paths.py +93 -0
  39. echoact/policy.py +189 -0
  40. echoact/security/__init__.py +0 -0
  41. echoact/security/credentials.py +930 -0
  42. echoact/security/ratelimit.py +534 -0
  43. echoact/service/__init__.py +20 -0
  44. echoact/service/app.py +182 -0
  45. echoact/service/deps.py +563 -0
  46. echoact/service/errors.py +241 -0
  47. echoact/service/routes.py +1125 -0
  48. echoact/service/schemas.py +509 -0
  49. echoact/service/server.py +270 -0
  50. echoact/text/__init__.py +0 -0
  51. echoact/text/language.py +44 -0
  52. echoact/text/loader.py +577 -0
  53. echoact/text/normalize.py +924 -0
  54. echoact/text/segment.py +499 -0
  55. echoact/text/sniff.py +1202 -0
  56. echoact/ui/__init__.py +0 -0
  57. echoact/ui/bridge.py +50 -0
  58. echoact/ui/controls.py +360 -0
  59. echoact/ui/credential_dialog.py +131 -0
  60. echoact/ui/fonts.py +94 -0
  61. echoact/ui/i18n.py +260 -0
  62. echoact/ui/icons.py +440 -0
  63. echoact/ui/library.py +1642 -0
  64. echoact/ui/licence.py +162 -0
  65. echoact/ui/main_window.py +1202 -0
  66. echoact/ui/mcp_setup.py +494 -0
  67. echoact/ui/models_view.py +1142 -0
  68. echoact/ui/notifications.py +202 -0
  69. echoact/ui/reading.py +494 -0
  70. echoact/ui/settings_view.py +2258 -0
  71. echoact/ui/status_view.py +1193 -0
  72. echoact/ui/theme.py +579 -0
  73. echoact/util/__init__.py +0 -0
  74. echoact/util/ids.py +62 -0
  75. echoact/util/logging.py +127 -0
  76. echoact-0.1.0.dist-info/METADATA +162 -0
  77. echoact-0.1.0.dist-info/RECORD +80 -0
  78. echoact-0.1.0.dist-info/WHEEL +4 -0
  79. echoact-0.1.0.dist-info/entry_points.txt +3 -0
  80. echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
echoact/jobs/engine.py ADDED
@@ -0,0 +1,776 @@
1
+ """The single generation slot, and the machine that drives one job through it.
2
+
3
+ F-47 allows one generation across the GUI and every client combined, so
4
+ this class is the product's narrowest resource and the one place that owns
5
+ it. Everything else -- the window, the REST service, the MCP server --
6
+ asks here and is told yes, or busy.
7
+
8
+ The shape follows from three requirements that pull against each other:
9
+
10
+ * F-12 wants audio as early as possible, so segments are rendered one at a
11
+ time and published the moment each file is closed, rather than at the end.
12
+ * N-22 wants the slot released within five seconds of a cancellation, which
13
+ means the worker can be killed mid-segment. That is only safe because the
14
+ worker owns nothing durable and because the parent named every output file
15
+ before asking for it, so an interrupted write leaves a file this module
16
+ already knows to discard.
17
+ * Section 5.1 says Complete means the audio is ready *and* any requested
18
+ retention succeeded. So the terminal transition happens after the result
19
+ is written and recorded, never when the last segment lands.
20
+
21
+ Cancellation and completion can race. 5.1 keeps whichever terminal state
22
+ was confirmed first and forbids Canceled from overwriting Complete, and
23
+ that is enforced in the store's transition check rather than by ordering
24
+ here -- a rule that depends on two threads interleaving politely is not a
25
+ rule.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import os
31
+ import shutil
32
+ import threading
33
+ from collections.abc import Callable, Iterable
34
+ from dataclasses import dataclass, field
35
+ from enum import StrEnum
36
+ from pathlib import Path
37
+ from typing import Any
38
+
39
+ from ..audio import wav
40
+ from ..config.budget import HaltReason, ResourceSampler, halt_decision, resolve_budget_from_system
41
+ from ..config.settings import Settings
42
+ from ..db.backup import RESTORE_GATE
43
+ from ..db.store import Store
44
+ from ..domain import (
45
+ Budget,
46
+ Job,
47
+ JobState,
48
+ RequestPath,
49
+ RetentionMode,
50
+ Segment,
51
+ TimeRange,
52
+ VoiceSettings,
53
+ )
54
+ from ..engine.supervisor import WorkerSupervisor
55
+ from ..errors import Code, EchoActError
56
+ from ..models.manifest import Manifest
57
+ from ..models.registry import ModelRegistry, ModelState
58
+ from ..paths import audio_dir, temp_dir
59
+ from ..policy import (
60
+ BOUNDED_WAIT_CEILING_S,
61
+ ENGINE_TOTAL_STEPS,
62
+ ONEOFF_RESULT_TTL_S,
63
+ WORKER_RELEASE_DEADLINE_S,
64
+ )
65
+ from ..util import ids
66
+ from ..util.logging import get_logger, job_context
67
+ from .request import JobRequest, plan_segments, validate_request
68
+
69
+ log = get_logger("jobs.engine")
70
+
71
+ #: What a busy caller is told to wait. A.5 measured a real-time factor
72
+ #: near 0.2, so a typical short job is seconds rather than minutes; the
73
+ #: hint is a floor on politeness, not a prediction.
74
+ BUSY_RETRY_AFTER_S = 3.0
75
+
76
+
77
+ class EventKind(StrEnum):
78
+ ACCEPTED = "accepted"
79
+ STATE = "state"
80
+ SEGMENT = "segment"
81
+ USAGE = "usage"
82
+ FINISHED = "finished"
83
+
84
+
85
+ @dataclass(frozen=True, slots=True)
86
+ class Event:
87
+ """What happened, for anyone watching.
88
+
89
+ Carries identifiers and numbers, never audio and never body text: the
90
+ GUI, the log, and a notification all consume these, and N-20 keeps body
91
+ text out of the last two.
92
+ """
93
+
94
+ kind: EventKind
95
+ job_id: str
96
+ state: JobState | None = None
97
+ segment_index: int | None = None
98
+ generated: int = 0
99
+ total: int = 0
100
+ request_path: RequestPath | None = None
101
+ client_label: str | None = None
102
+ error_code: str | None = None
103
+ detail: dict[str, Any] = field(default_factory=dict)
104
+
105
+
106
+ @dataclass(slots=True)
107
+ class _Run:
108
+ """The mutable state of the one job that is running."""
109
+
110
+ job: Job
111
+ segments: list[Segment]
112
+ budget: Budget
113
+ settings: VoiceSettings
114
+ cancel: threading.Event
115
+ finished: threading.Event
116
+ thread: threading.Thread | None = None
117
+ sample_rate: int = 0
118
+ segment_paths: list[str] = field(default_factory=list)
119
+ gaps: list[int] = field(default_factory=list)
120
+ halt: EchoActError | None = None
121
+
122
+
123
+ class JobEngine:
124
+ """One slot, one worker, one job at a time."""
125
+
126
+ def __init__(
127
+ self,
128
+ *,
129
+ store: Store,
130
+ supervisor: WorkerSupervisor,
131
+ registry: ModelRegistry,
132
+ manifest: Manifest,
133
+ settings: Settings,
134
+ work_dir: Path | None = None,
135
+ result_dir: Path | None = None,
136
+ total_steps: int = ENGINE_TOTAL_STEPS,
137
+ ) -> None:
138
+ self._store = store
139
+ self._supervisor = supervisor
140
+ self._registry = registry
141
+ self._manifest = manifest
142
+ self._settings = settings
143
+ self._work_dir = Path(work_dir) if work_dir else temp_dir()
144
+ self._result_dir = Path(result_dir) if result_dir else audio_dir()
145
+ self._total_steps = total_steps
146
+
147
+ self._lock = threading.RLock()
148
+ self._run: _Run | None = None
149
+ self._listeners: list[Callable[[Event], None]] = []
150
+ self._closing = False
151
+
152
+ # ------------------------------------------------------------------
153
+ # Observation
154
+ # ------------------------------------------------------------------
155
+
156
+ def listen(self, fn: Callable[[Event], None]) -> Callable[[], None]:
157
+ """Subscribe. Returns an unsubscribe callable.
158
+
159
+ Listeners are called on the run thread, so a GUI listener must
160
+ marshal to the main thread rather than touch a widget here.
161
+ """
162
+ self._listeners.append(fn)
163
+
164
+ def off() -> None:
165
+ with self._lock:
166
+ if fn in self._listeners:
167
+ self._listeners.remove(fn)
168
+
169
+ return off
170
+
171
+ def _emit(self, event: Event) -> None:
172
+ for fn in list(self._listeners):
173
+ try:
174
+ fn(event)
175
+ except Exception as exc: # noqa: BLE001 - a listener must not stop a job
176
+ log.warning("job listener failed: %s", type(exc).__name__)
177
+
178
+ # ------------------------------------------------------------------
179
+ # State
180
+ # ------------------------------------------------------------------
181
+
182
+ @property
183
+ def busy(self) -> bool:
184
+ with self._lock:
185
+ return self._run is not None
186
+
187
+ def current(self) -> Job | None:
188
+ with self._lock:
189
+ return self._run.job if self._run else None
190
+
191
+ def apply_settings(self, settings: Settings) -> None:
192
+ """F-78: a change during a job applies to the *next* job.
193
+
194
+ Nothing here touches a running job, which is the whole point: the
195
+ job recorded the budget it started under and keeps it.
196
+ """
197
+ self._settings = settings
198
+
199
+ def would_need_load(self, settings: VoiceSettings | None = None) -> bool:
200
+ """F-17: whether the next job pays for a model load."""
201
+ voice = settings or self._settings.voice
202
+ try:
203
+ budget = resolve_budget_from_system(self._settings)
204
+ except EchoActError:
205
+ return True
206
+ return self._supervisor.would_need_load(voice.model_id, budget)
207
+
208
+ # ------------------------------------------------------------------
209
+ # Submission (F-47, F-49)
210
+ # ------------------------------------------------------------------
211
+
212
+ def submit(self, request: JobRequest) -> tuple[Job, bool]:
213
+ """Accept a job, or refuse it. Returns ``(job, created)``.
214
+
215
+ Order matters and is not arbitrary. The slot is taken *before* the
216
+ duplicate-prevention key is claimed, because a refusal for busy must
217
+ not consume the key -- F-49's record is meant to identify a job that
218
+ exists, and a busy response creates none. If the claim then finds
219
+ an existing job, the slot is handed straight back.
220
+ """
221
+ # 5.3: during a restore, new generation and edits are blocked.
222
+ # Checked before anything else because it is the cheapest refusal
223
+ # and the only one that is temporary by construction -- the
224
+ # restore will finish, so this is retryable where the others are
225
+ # the caller's to fix.
226
+ RESTORE_GATE.require_idle("Generating speech")
227
+ entry = validate_request(request, self._manifest)
228
+ budget = resolve_budget_from_system(self._settings)
229
+
230
+ runnable, why = self._registry.can_run(request.settings.model_id, budget)
231
+ if not runnable:
232
+ # F-04: unavailable with the reason, never silently substituted.
233
+ raise EchoActError(Code.MODEL_OVER_BUDGET, why or None)
234
+ self._check_model_preparable(request.settings.model_id)
235
+
236
+ with self._lock:
237
+ if self._closing:
238
+ raise EchoActError(Code.SHUTTING_DOWN)
239
+ taken = self._run is None
240
+ if taken:
241
+ self._run = _PLACEHOLDER
242
+ if not taken:
243
+ raise EchoActError(Code.BUSY, retry_after_s=BUSY_RETRY_AFTER_S)
244
+
245
+ try:
246
+ job = Job(
247
+ job_id=ids.job_id(),
248
+ kind=request.kind,
249
+ request_path=request.request_path,
250
+ owner_client_id=request.owner_client_id,
251
+ state=JobState.ACCEPTED,
252
+ source_text=request.text,
253
+ settings=request.settings,
254
+ budget=budget,
255
+ retention=request.retention,
256
+ created_at=ids.now(),
257
+ client_label=request.client_label,
258
+ idempotency_key=request.idempotency_key,
259
+ )
260
+ stored, created = self._store.claim_job(
261
+ job,
262
+ client_id=request.owner_client_id,
263
+ key=request.idempotency_key,
264
+ request_digest=request.digest(),
265
+ )
266
+ if not created:
267
+ # 4.2: a repeat returns the existing job and its state,
268
+ # including a terminal or expired one, and regenerates
269
+ # nothing. A new key is needed to generate again.
270
+ self._release_slot()
271
+ return stored, False
272
+
273
+ segments = plan_segments(request.text, request.settings)
274
+ stored.segments = list(self._store.insert_segments(stored.job_id, segments))
275
+ stored.total_segments = len(stored.segments)
276
+ self._store.set_job_budget(stored.job_id, budget)
277
+
278
+ run = _Run(
279
+ job=stored,
280
+ segments=stored.segments,
281
+ budget=budget,
282
+ settings=request.settings,
283
+ cancel=threading.Event(),
284
+ finished=threading.Event(),
285
+ sample_rate=entry.sample_rate,
286
+ )
287
+ with self._lock:
288
+ self._run = run
289
+ self._emit(
290
+ Event(
291
+ EventKind.ACCEPTED,
292
+ stored.job_id,
293
+ state=stored.state,
294
+ total=len(segments),
295
+ request_path=stored.request_path,
296
+ client_label=stored.client_label,
297
+ )
298
+ )
299
+ run.thread = threading.Thread(
300
+ target=self._run_job, args=(run,), name=f"echoact-job-{stored.job_id}", daemon=True
301
+ )
302
+ run.thread.start()
303
+ return stored, True
304
+ except BaseException:
305
+ self._release_slot()
306
+ raise
307
+
308
+ def _check_model_preparable(self, model_id: str) -> None:
309
+ """Refuse at acceptance what could only fail later anyway.
310
+
311
+ 5.3 says a request for a model that is not downloaded is *refused*,
312
+ not accepted and then failed, and the same reasoning covers a
313
+ licence nobody has accepted: both are knowable now, neither can
314
+ change while the job waits, and accepting the job would consume
315
+ F-49's key on something that can never run and hand the caller a
316
+ job id to poll instead of an answer.
317
+
318
+ Deliberately the shallow check. N-22 gives acceptance one second
319
+ at p95 and a deep verify hashes 385 MB; the deep pass still runs in
320
+ ``_prepare``, where its cost belongs.
321
+ """
322
+ status = self._registry.status(model_id, deep=False)
323
+ if status.state is ModelState.CORRUPT:
324
+ raise EchoActError(Code.MODEL_CORRUPT, detail={"model_id": model_id})
325
+ if status.state is not ModelState.READY:
326
+ raise EchoActError(Code.MODEL_NOT_READY, detail={"model_id": model_id})
327
+ if status.license_acceptance_required and not status.license_accepted:
328
+ raise EchoActError(
329
+ Code.MODEL_LICENSE_NOT_ACCEPTED,
330
+ detail={"model_id": model_id, "license": status.license_name},
331
+ )
332
+
333
+ def _release_slot(self) -> None:
334
+ with self._lock:
335
+ self._run = None
336
+
337
+ # ------------------------------------------------------------------
338
+ # Waiting (F-88)
339
+ # ------------------------------------------------------------------
340
+
341
+ def wait(self, job_id: str, timeout_s: float) -> Job:
342
+ """Wait up to ``timeout_s`` for a terminal state, then answer anyway.
343
+
344
+ F-88 makes this an optimisation and never a different lifecycle:
345
+ the job is untouched whether the bound passes or not, and the caller
346
+ gets the same job either way. It also never waits on model
347
+ preparation -- the wait begins only once generation is under way,
348
+ which is why the deadline is re-checked against the job's state
349
+ rather than simply slept through.
350
+ """
351
+ bound = max(0.0, min(float(timeout_s), BOUNDED_WAIT_CEILING_S))
352
+ with self._lock:
353
+ run = self._run
354
+ if run is None or run is _PLACEHOLDER or run.job.job_id != job_id:
355
+ return self._store.get_job(job_id)
356
+ if bound > 0:
357
+ run.finished.wait(bound)
358
+ return self._store.get_job(job_id)
359
+
360
+ # ------------------------------------------------------------------
361
+ # Cancellation (F-15, F-49, N-22)
362
+ # ------------------------------------------------------------------
363
+
364
+ def cancel(self, job_id: str) -> Job:
365
+ """Cancel a job. Repeating it adds no further side effects (F-49)."""
366
+ job = self._store.get_job(job_id, include_segments=False)
367
+ if job.state.is_terminal:
368
+ return job
369
+ with self._lock:
370
+ run = self._run
371
+ running = run is not None and run is not _PLACEHOLDER and run.job.job_id == job_id
372
+ if not running:
373
+ # Accepted but not the current job: only possible after an
374
+ # abnormal termination, which F-45 reconciles to Interrupted.
375
+ self._store.update_job_state(job_id, JobState.CANCELING, force=True)
376
+ self._store.update_job_state(job_id, JobState.CANCELED)
377
+ return self._store.get_job(job_id, include_segments=False)
378
+
379
+ assert run is not None
380
+ self._set_state(run, JobState.CANCELING)
381
+ run.cancel.set()
382
+ # Mid-segment is the case N-22's five seconds is written for: the
383
+ # worker is inside an ONNX call that will not return promptly, so
384
+ # asking politely is not enough.
385
+ elapsed = self._supervisor.kill()
386
+ log.info(job_context(job_id, "canceling", release_s=round(elapsed, 3)))
387
+ run.finished.wait(WORKER_RELEASE_DEADLINE_S)
388
+ return self._store.get_job(job_id, include_segments=False)
389
+
390
+ def release_model(self) -> None:
391
+ """F-19's explicit release. Refused while a job is running, because
392
+ the job would then fail rather than the model being freed."""
393
+ if self.busy:
394
+ raise EchoActError(Code.BUSY, retry_after_s=BUSY_RETRY_AFTER_S)
395
+ self._supervisor.unload()
396
+
397
+ def shutdown(self) -> None:
398
+ """F-52: report and stop. The worker goes first, so nothing is
399
+ still writing when the database closes."""
400
+ with self._lock:
401
+ self._closing = True
402
+ run = self._run
403
+ if run is not None and run is not _PLACEHOLDER:
404
+ run.cancel.set()
405
+ self._supervisor.kill()
406
+ run.finished.wait(WORKER_RELEASE_DEADLINE_S)
407
+ else:
408
+ self._supervisor.kill()
409
+
410
+ # ------------------------------------------------------------------
411
+ # The run
412
+ # ------------------------------------------------------------------
413
+
414
+ def _run_job(self, run: _Run) -> None:
415
+ job_id = run.job.job_id
416
+ try:
417
+ self._prepare(run)
418
+ self._generate(run)
419
+ self._finish(run)
420
+ except EchoActError as exc:
421
+ self._fail(run, exc)
422
+ except Exception as exc: # noqa: BLE001 - a run thread must not vanish
423
+ log.exception("job %s failed unexpectedly", job_id)
424
+ self._fail(run, EchoActError(Code.GENERATION_FAILED, cause=exc))
425
+ finally:
426
+ self._cleanup(run)
427
+ run.finished.set()
428
+ self._release_slot()
429
+ self._announce_finished(run)
430
+
431
+ def _announce_finished(self, run: _Run) -> None:
432
+ """The last word on a job, read back from the database.
433
+
434
+ Read back rather than assumed, because the terminal state may not
435
+ be the one this thread chose: 5.1 lets a cancellation and a
436
+ completion race and keeps whichever was confirmed first.
437
+
438
+ Tolerant of a closed database on purpose. Shutdown kills the worker
439
+ and then closes the store, so this can be the last thing running
440
+ during an exit, and a job that has already ended is not worth an
441
+ exception on the way out.
442
+ """
443
+ job_id = run.job.job_id
444
+ try:
445
+ final = self._store.get_job(
446
+ job_id, include_source_text=False, include_segments=False
447
+ )
448
+ state, generated = final.state, final.generated_segments
449
+ total, error_code = final.total_segments, final.error_code
450
+ path, label = final.request_path, final.client_label
451
+ except EchoActError:
452
+ state = JobState.CANCELED if run.cancel.is_set() else run.job.state
453
+ generated = sum(1 for s in run.segments if s.ready)
454
+ total, error_code = len(run.segments), None
455
+ path, label = run.job.request_path, run.job.client_label
456
+ self._emit(
457
+ Event(
458
+ EventKind.FINISHED,
459
+ job_id,
460
+ state=state,
461
+ generated=generated,
462
+ total=total,
463
+ request_path=path,
464
+ client_label=label,
465
+ error_code=error_code,
466
+ )
467
+ )
468
+
469
+ def _prepare(self, run: _Run) -> None:
470
+ """Resolve the model and load it, per F-09, F-17, F-84."""
471
+ self._check_cancelled(run)
472
+ self._set_state(run, JobState.PREPARING_MODEL)
473
+
474
+ model_id = run.settings.model_id
475
+ # F-09/5.3: the engine never downloads. A model that is not present
476
+ # is refused here, and preparing it is the owner's explicit action
477
+ # in the model screen.
478
+ model_dir, from_package = self._registry.resolve_dir(model_id)
479
+ if from_package:
480
+ log.info("model %s served from the package cache", model_id)
481
+
482
+ if self._supervisor.would_need_load(model_id, run.budget):
483
+ loaded = self._supervisor.load(model_id, model_dir, run.budget)
484
+ run.sample_rate = loaded.sample_rate
485
+ else:
486
+ current = self._supervisor.loaded_model
487
+ if current is not None:
488
+ run.sample_rate = current.sample_rate
489
+ self._check_cancelled(run)
490
+
491
+ def _generate(self, run: _Run) -> None:
492
+ self._set_state(run, JobState.GENERATING)
493
+ job_id = run.job.job_id
494
+ sampler = ResourceSampler(self._supervisor.worker_pid)
495
+ start_frame = 0
496
+
497
+ for seg in run.segments:
498
+ self._check_cancelled(run)
499
+ self._check_resources(run, sampler)
500
+
501
+ gap_frames = wav.frames_for_ms(seg.trailing_silence_ms, run.sample_rate)
502
+ if not seg.is_spoken:
503
+ # F-27: a range that produces no audio still exists on the
504
+ # timeline, attached to its neighbour. Nothing is sent to
505
+ # the engine, which A.5 showed would otherwise vocalise it.
506
+ span = TimeRange(
507
+ wav.ms_for_frames(start_frame, run.sample_rate),
508
+ wav.ms_for_frames(start_frame + gap_frames, run.sample_rate),
509
+ )
510
+ self._store.mark_segment_ready(
511
+ job_id, seg.index, time=span, audio_path=None, frame_count=0
512
+ )
513
+ seg.time, seg.ready, seg.frame_count = span, True, 0
514
+ run.segment_paths.append("")
515
+ run.gaps.append(seg.trailing_silence_ms)
516
+ start_frame += gap_frames
517
+ self._emit_segment(run, seg)
518
+ continue
519
+
520
+ out = self._work_dir / job_id / f"{seg.index:05d}.wav"
521
+ out.parent.mkdir(parents=True, exist_ok=True)
522
+ reply = self._supervisor.synthesize(
523
+ job_id=job_id,
524
+ segment_index=seg.index,
525
+ text=seg.spoken_text,
526
+ lang=seg.language,
527
+ voice_id=run.settings.voice_id,
528
+ speed=_engine_speed(run.settings),
529
+ out_path=out,
530
+ total_steps=self._total_steps,
531
+ )
532
+ self._check_cancelled(run)
533
+
534
+ span = TimeRange(
535
+ wav.ms_for_frames(start_frame, run.sample_rate),
536
+ wav.ms_for_frames(start_frame + reply.frame_count + gap_frames, run.sample_rate),
537
+ )
538
+ self._store.mark_segment_ready(
539
+ job_id,
540
+ seg.index,
541
+ time=span,
542
+ audio_path=str(out),
543
+ frame_count=reply.frame_count,
544
+ )
545
+ seg.time = span
546
+ seg.ready = True
547
+ seg.frame_count = reply.frame_count
548
+ seg.audio_path = str(out)
549
+ run.segment_paths.append(str(out))
550
+ run.gaps.append(seg.trailing_silence_ms)
551
+ start_frame += reply.frame_count + gap_frames
552
+ self._emit_segment(run, seg)
553
+
554
+ def _finish(self, run: _Run) -> None:
555
+ """Concatenate, record, and only then call the job Complete.
556
+
557
+ 5.1: "If generation finishes but saving fails, the job is not marked
558
+ Complete." So the transition is last, after the file exists and the
559
+ row is written.
560
+ """
561
+ self._check_cancelled(run)
562
+ job_id = run.job.job_id
563
+ retained = run.job.retention is RetentionMode.RETAINED
564
+ root = self._result_dir if retained else self._work_dir
565
+ out = root / job_id / "result.wav"
566
+ out.parent.mkdir(parents=True, exist_ok=True)
567
+
568
+ spoken = [(p, g) for p, g in zip(run.segment_paths, run.gaps, strict=True) if p]
569
+ if not spoken:
570
+ raise EchoActError(
571
+ Code.GENERATION_FAILED, "The text produced no audio.", detail={"job_id": job_id}
572
+ )
573
+ report = wav.concatenate(
574
+ [p for p, _ in spoken],
575
+ gaps_ms=[g for _, g in spoken],
576
+ out_path=out,
577
+ sample_rate=run.sample_rate,
578
+ )
579
+ now = ids.now()
580
+ from ..domain import Result
581
+
582
+ result = Result(
583
+ result_id=ids.result_id(),
584
+ job_id=job_id,
585
+ sample_rate=report.sample_rate,
586
+ channels=1,
587
+ sample_width_bits=16,
588
+ frame_count=report.frame_count,
589
+ byte_size=report.byte_size,
590
+ digest=wav.digest(out),
591
+ created_at=now,
592
+ expires_at=None if retained else now + ONEOFF_RESULT_TTL_S,
593
+ relative_path=str(out),
594
+ )
595
+ self._store.attach_result(result)
596
+ self._set_state(run, JobState.COMPLETE)
597
+
598
+ def _fail(self, run: _Run, error: EchoActError) -> None:
599
+ job_id = run.job.job_id
600
+ if run.cancel.is_set():
601
+ # A cancellation in flight looks like a failure from inside the
602
+ # loop. 5.1 wants Canceled, and Complete is never overwritten
603
+ # because the store refuses that transition.
604
+ try:
605
+ self._store.update_job_state(job_id, JobState.CANCELED)
606
+ except (AssertionError, EchoActError):
607
+ pass
608
+ self._emit_state(run, JobState.CANCELED)
609
+ return
610
+ log.warning(job_context(job_id, "failed", code=error.code.value))
611
+ try:
612
+ self._store.record_job_error(job_id, error.code, error.message)
613
+ self._store.update_job_state(job_id, JobState.FAILED)
614
+ except (AssertionError, EchoActError):
615
+ pass
616
+ # F-19: the model is released on error as well as on cancellation.
617
+ self._supervisor.kill()
618
+ self._emit_state(run, JobState.FAILED, error_code=error.code.value)
619
+
620
+ def _cleanup(self, run: _Run) -> None:
621
+ """Remove the per-job scratch directory once nothing needs it.
622
+
623
+ A retained job's segment audio stays: F-55 lets a client fetch
624
+ individual segments, and F-31 replays a retained job with its
625
+ mapping. A one-off job's scratch is the result's own home until it
626
+ expires, so it is left for the sweeper rather than deleted here.
627
+ """
628
+ if run.cancel.is_set() and run.job.retention is RetentionMode.ONE_OFF:
629
+ scratch = self._work_dir / run.job.job_id
630
+ shutil.rmtree(scratch, ignore_errors=True)
631
+
632
+ # -- helpers ---------------------------------------------------------
633
+
634
+ def _check_cancelled(self, run: _Run) -> None:
635
+ if run.cancel.is_set():
636
+ raise EchoActError(Code.GENERATION_FAILED, "Canceled.")
637
+
638
+ def _check_resources(self, run: _Run, sampler: ResourceSampler) -> None:
639
+ """F-23 and N-04, checked between segments.
640
+
641
+ Between rather than during: the container is what stops a spike
642
+ inside a single ONNX call, and N-03 is explicit that the enforced
643
+ ceiling is the operating system's job. This is the part that
644
+ reports *why* and releases the model.
645
+ """
646
+ sample = sampler.poll()
647
+ if sample is None:
648
+ return
649
+ decision = halt_decision(sample, run.budget)
650
+ if decision.reason is HaltReason.NONE:
651
+ return
652
+ raise decision.to_error()
653
+
654
+ def _set_state(self, run: _Run, state: JobState) -> None:
655
+ self._store.update_job_state(run.job.job_id, state)
656
+ run.job.state = state
657
+ self._emit_state(run, state)
658
+
659
+ def _emit_state(self, run: _Run, state: JobState, error_code: str | None = None) -> None:
660
+ self._emit(
661
+ Event(
662
+ EventKind.STATE,
663
+ run.job.job_id,
664
+ state=state,
665
+ generated=sum(1 for s in run.segments if s.ready),
666
+ total=len(run.segments),
667
+ request_path=run.job.request_path,
668
+ client_label=run.job.client_label,
669
+ error_code=error_code,
670
+ )
671
+ )
672
+
673
+ def _emit_segment(self, run: _Run, seg: Segment) -> None:
674
+ self._emit(
675
+ Event(
676
+ EventKind.SEGMENT,
677
+ run.job.job_id,
678
+ segment_index=seg.index,
679
+ generated=sum(1 for s in run.segments if s.ready),
680
+ total=len(run.segments),
681
+ request_path=run.job.request_path,
682
+ client_label=run.job.client_label,
683
+ detail={
684
+ "start_ms": seg.time.start_ms if seg.time else 0,
685
+ "end_ms": seg.time.end_ms if seg.time else 0,
686
+ "frame_count": seg.frame_count,
687
+ "audio_path": seg.audio_path or "",
688
+ },
689
+ )
690
+ )
691
+
692
+
693
+ def _engine_speed(settings: VoiceSettings) -> float:
694
+ """F-07 and F-08 combined into the one number the engine takes.
695
+
696
+ A style is a preset over tempo among other things, so the two multiply
697
+ -- and the product is clamped back into F-07's advertised range, since
698
+ a style must never take tempo somewhere the user is told is impossible.
699
+ """
700
+ from ..text.segment import effective_tempo
701
+
702
+ return effective_tempo(settings)
703
+
704
+
705
+ class _Placeholder:
706
+ """Marks the slot as taken between the check and the real run object.
707
+
708
+ Without it, two submissions could both find the slot free while the
709
+ first was still building its job -- the window is small and, with one
710
+ slot and clients that retry, exactly the window that gets hit.
711
+ """
712
+
713
+ __slots__ = ()
714
+
715
+
716
+ _PLACEHOLDER: Any = _Placeholder()
717
+
718
+
719
+ def expire_one_off_results(store: Store, work_root: Path | None = None) -> int:
720
+ """4.1's one-off lifetime, swept.
721
+
722
+ Returns how many results were removed. Retained data is never touched:
723
+ N-16 forbids deleting explicitly retained results to reclaim space, and
724
+ this is a lifetime sweep rather than a space one.
725
+ """
726
+ root = Path(work_root) if work_root else temp_dir()
727
+ now = ids.now()
728
+ removed = 0
729
+ for job_id, result_id, _expires_at, path in _expired(store, now):
730
+ try:
731
+ store.delete_result(result_id)
732
+ except EchoActError:
733
+ continue
734
+ removed += 1
735
+ try:
736
+ if path:
737
+ Path(path).unlink(missing_ok=True)
738
+ scratch = root / job_id
739
+ if scratch.is_dir():
740
+ shutil.rmtree(scratch, ignore_errors=True)
741
+ except OSError:
742
+ pass
743
+ return removed
744
+
745
+
746
+ def _expired(store: Store, now: float) -> Iterable[tuple[str, str, float, str]]:
747
+ conn = store._conn() # noqa: SLF001 - the sweeper is part of the storage layer
748
+ rows = conn.execute(
749
+ "SELECT job_id, result_id, expires_at, relative_path FROM results"
750
+ " WHERE expires_at IS NOT NULL AND expires_at <= ?",
751
+ (now,),
752
+ ).fetchall()
753
+ return [(r["job_id"], r["result_id"], r["expires_at"], r["relative_path"]) for r in rows]
754
+
755
+
756
+ def clear_temp_tree(work_root: Path | None = None) -> int:
757
+ """N-02: whatever a forced termination left behind, on relaunch.
758
+
759
+ Returns the number of entries removed. Only the scratch tree is
760
+ touched; retained audio lives elsewhere precisely so that this can be
761
+ unconditional.
762
+ """
763
+ root = Path(work_root) if work_root else temp_dir()
764
+ if not root.is_dir():
765
+ return 0
766
+ removed = 0
767
+ for child in root.iterdir():
768
+ try:
769
+ if child.is_dir():
770
+ shutil.rmtree(child, ignore_errors=True)
771
+ else:
772
+ os.unlink(child)
773
+ removed += 1
774
+ except OSError:
775
+ continue
776
+ return removed