echoact 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. echoact/__init__.py +3 -0
  2. echoact/__main__.py +117 -0
  3. echoact/app.py +315 -0
  4. echoact/audio/__init__.py +0 -0
  5. echoact/audio/devices.py +192 -0
  6. echoact/audio/player.py +611 -0
  7. echoact/audio/wav.py +854 -0
  8. echoact/config/__init__.py +0 -0
  9. echoact/config/budget.py +370 -0
  10. echoact/config/settings.py +1244 -0
  11. echoact/db/__init__.py +0 -0
  12. echoact/db/backup.py +2429 -0
  13. echoact/db/migrations.py +434 -0
  14. echoact/db/schema.sql +214 -0
  15. echoact/db/store.py +2062 -0
  16. echoact/diagnostics.py +902 -0
  17. echoact/domain.py +487 -0
  18. echoact/engine/__init__.py +0 -0
  19. echoact/engine/container.py +843 -0
  20. echoact/engine/protocol.py +241 -0
  21. echoact/engine/runtime.py +324 -0
  22. echoact/engine/supervisor.py +961 -0
  23. echoact/engine/worker.py +659 -0
  24. echoact/errors.py +281 -0
  25. echoact/instance.py +172 -0
  26. echoact/jobs/__init__.py +0 -0
  27. echoact/jobs/engine.py +776 -0
  28. echoact/jobs/request.py +300 -0
  29. echoact/mcp/__init__.py +0 -0
  30. echoact/mcp/__main__.py +50 -0
  31. echoact/mcp/client.py +202 -0
  32. echoact/mcp/config.py +112 -0
  33. echoact/mcp/server.py +340 -0
  34. echoact/models/__init__.py +0 -0
  35. echoact/models/catalog.py +273 -0
  36. echoact/models/manifest.py +278 -0
  37. echoact/models/registry.py +1551 -0
  38. echoact/paths.py +93 -0
  39. echoact/policy.py +189 -0
  40. echoact/security/__init__.py +0 -0
  41. echoact/security/credentials.py +930 -0
  42. echoact/security/ratelimit.py +534 -0
  43. echoact/service/__init__.py +20 -0
  44. echoact/service/app.py +182 -0
  45. echoact/service/deps.py +563 -0
  46. echoact/service/errors.py +241 -0
  47. echoact/service/routes.py +1125 -0
  48. echoact/service/schemas.py +509 -0
  49. echoact/service/server.py +270 -0
  50. echoact/text/__init__.py +0 -0
  51. echoact/text/language.py +44 -0
  52. echoact/text/loader.py +577 -0
  53. echoact/text/normalize.py +924 -0
  54. echoact/text/segment.py +499 -0
  55. echoact/text/sniff.py +1202 -0
  56. echoact/ui/__init__.py +0 -0
  57. echoact/ui/bridge.py +50 -0
  58. echoact/ui/controls.py +360 -0
  59. echoact/ui/credential_dialog.py +131 -0
  60. echoact/ui/fonts.py +94 -0
  61. echoact/ui/i18n.py +260 -0
  62. echoact/ui/icons.py +440 -0
  63. echoact/ui/library.py +1642 -0
  64. echoact/ui/licence.py +162 -0
  65. echoact/ui/main_window.py +1202 -0
  66. echoact/ui/mcp_setup.py +494 -0
  67. echoact/ui/models_view.py +1142 -0
  68. echoact/ui/notifications.py +202 -0
  69. echoact/ui/reading.py +494 -0
  70. echoact/ui/settings_view.py +2258 -0
  71. echoact/ui/status_view.py +1193 -0
  72. echoact/ui/theme.py +579 -0
  73. echoact/util/__init__.py +0 -0
  74. echoact/util/ids.py +62 -0
  75. echoact/util/logging.py +127 -0
  76. echoact-0.1.0.dist-info/METADATA +162 -0
  77. echoact-0.1.0.dist-info/RECORD +80 -0
  78. echoact-0.1.0.dist-info/WHEEL +4 -0
  79. echoact-0.1.0.dist-info/entry_points.txt +3 -0
  80. echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
echoact/db/store.py ADDED
@@ -0,0 +1,2062 @@
1
+ """The local library: documents, job history, segments, results, and the
2
+ bookkeeping that keeps them honest.
3
+
4
+ SQLite in write-ahead mode with a *bounded* busy timeout, per A.2 and N-14.
5
+ The bound is the whole point: N-14 forbids both waiting indefinitely on a
6
+ lock and reporting a save that did not happen, so a lock that outlives the
7
+ timeout leaves here as ``DB_LOCKED`` -- a refusal a caller can retry -- and
8
+ never as a stall or a false success.
9
+
10
+ Three habits run through every method:
11
+
12
+ * One connection per thread. ``sqlite3.Connection`` objects are not
13
+ shareable, and the GUI thread, the job engine, and the REST worker all
14
+ read this store.
15
+ * Every write goes through :meth:`Store.transaction`, which begins
16
+ ``IMMEDIATE``. A deferred transaction takes its write lock only at the
17
+ first write and can then fail with ``SQLITE_BUSY`` *without* honouring the
18
+ busy timeout at all; taking the lock up front is what makes the bound real.
19
+ * No method reads an audio file except the two that exist to verify one
20
+ (:meth:`Store.verify_result` and :meth:`Store.reconcile_on_start`). F-40
21
+ requires a list to be answerable without reading audio, and N-21 forbids a
22
+ list query from loading unbounded data into memory, so summaries carry the
23
+ duration computed from stored frame counts and never open anything.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ import hashlib
29
+ import json
30
+ import re
31
+ import sqlite3
32
+ import threading
33
+ from collections.abc import Callable, Iterable, Iterator, Sequence
34
+ from contextlib import contextmanager
35
+ from dataclasses import dataclass, replace
36
+ from enum import StrEnum
37
+ from functools import wraps
38
+ from pathlib import Path
39
+ from typing import Any, Final
40
+
41
+ from .. import __version__
42
+ from ..domain import (
43
+ Budget,
44
+ Capability,
45
+ Document,
46
+ Job,
47
+ JobKind,
48
+ JobState,
49
+ RequestPath,
50
+ Result,
51
+ RetentionMode,
52
+ Segment,
53
+ TextRange,
54
+ TimeRange,
55
+ VoiceSettings,
56
+ check_transition,
57
+ )
58
+ from ..errors import Code, EchoActError
59
+ from ..paths import audio_dir, db_path
60
+ from ..policy import (
61
+ IDEMPOTENCY_TTL_S,
62
+ LIST_PAGE_DEFAULT,
63
+ LIST_PAGE_MAX,
64
+ RETENTION_DEFAULT_BYTES,
65
+ WORKER_RELEASE_DEADLINE_S,
66
+ )
67
+ from ..util import ids
68
+ from .migrations import (
69
+ SUPPORTED_SCHEMA_VERSION,
70
+ assert_compatible,
71
+ current_version,
72
+ has_fts_index,
73
+ migrate,
74
+ translate_sqlite_error,
75
+ )
76
+
77
+ #: How long a write may wait for another writer before it is refused.
78
+ #:
79
+ #: ``echoact.policy`` has no database timeout of its own, so this borrows
80
+ #: N-22's five-second resource-release deadline as the nearest bounded
81
+ #: deadline the requirements fix. It is an upper bound on a pathological
82
+ #: case, not a latency: the app has one writer thread in normal operation,
83
+ #: and N-22's one-second p95 for queries is met with room to spare.
84
+ DEFAULT_BUSY_TIMEOUT_S: Final = WORKER_RELEASE_DEADLINE_S
85
+
86
+ #: Trigram FTS5 cannot match a term shorter than one trigram; below this the
87
+ #: store answers with LIKE instead of returning a confidently empty page.
88
+ _MIN_FTS_QUERY_CODEPOINTS: Final = 3
89
+
90
+ _LIKE_SPECIAL = re.compile(r"([\\%_])")
91
+
92
+ def _guard[**P, R](fn: Callable[P, R]) -> Callable[P, R]:
93
+ """Turn every driver failure into the one exception type (rule 3)."""
94
+
95
+ @wraps(fn)
96
+ def wrapper(*args: P.args, **kwargs: P.kwargs) -> R:
97
+ try:
98
+ return fn(*args, **kwargs)
99
+ except EchoActError:
100
+ raise
101
+ except (sqlite3.Error, OSError) as exc:
102
+ raise translate_sqlite_error(exc, retry_after_s=DEFAULT_BUSY_TIMEOUT_S) from exc
103
+
104
+ return wrapper
105
+
106
+
107
+ def text_bytes(text: str | None) -> int:
108
+ """UTF-8 size, which is what 4.1's storage ceiling counts."""
109
+ return 0 if text is None else len(text.encode("utf-8"))
110
+
111
+
112
+ def request_match_digest(*parts: str) -> str:
113
+ """4.2's request-match discriminator.
114
+
115
+ A digest rather than the request itself, because 4.2 allows a re-request
116
+ record to keep only what duplicate prevention needs and explicitly no
117
+ source text. Parts are length-prefixed so that ("ab", "c") and
118
+ ("a", "bc") cannot collide.
119
+ """
120
+ h = hashlib.sha256()
121
+ for part in parts:
122
+ raw = part.encode("utf-8")
123
+ h.update(str(len(raw)).encode("ascii"))
124
+ h.update(b":")
125
+ h.update(raw)
126
+ return h.hexdigest()
127
+
128
+
129
+ # ======================================================================
130
+ # Read models
131
+ # ======================================================================
132
+
133
+
134
+ class ResultIntegrity(StrEnum):
135
+ """F-45's finding about a result's file, persisted so the GUI can offer
136
+ delete and regenerate at any time and not only in the session that
137
+ happened to notice."""
138
+
139
+ UNVERIFIED = "unverified"
140
+ OK = "ok"
141
+ MISSING = "missing"
142
+ CORRUPT = "corrupt"
143
+
144
+
145
+ @dataclass(frozen=True)
146
+ class Page[T]:
147
+ """One page of a list query (4.1: 20 by default, 100 at most)."""
148
+
149
+ items: tuple[T, ...]
150
+ total: int
151
+ limit: int
152
+ offset: int
153
+
154
+ @property
155
+ def has_more(self) -> bool:
156
+ return self.offset + len(self.items) < self.total
157
+
158
+
159
+ @dataclass(frozen=True, slots=True)
160
+ class DocumentSummary:
161
+ """A document without its body.
162
+
163
+ N-21 forbids a list query from loading unbounded data: a hundred
164
+ documents of fifty thousand code points each is five million characters
165
+ nobody asked for.
166
+ """
167
+
168
+ document_id: str
169
+ title: str
170
+ created_at: float
171
+ modified_at: float
172
+ version: int
173
+ body_codepoints: int
174
+
175
+
176
+ @dataclass(frozen=True, slots=True)
177
+ class JobSummary:
178
+ """F-39's history row, and F-56's default projection of it.
179
+
180
+ ``source_text`` stays ``None`` unless a caller asks for it, because F-56
181
+ makes the summary the default and puts the snapshot behind a separate,
182
+ separately authorised request. ``audio_duration_ms`` comes from the
183
+ stored frame count, so F-40's list never opens a WAV.
184
+ """
185
+
186
+ job_id: str
187
+ kind: JobKind
188
+ request_path: RequestPath
189
+ owner_client_id: str
190
+ client_label: str | None
191
+ state: JobState
192
+ retention: RetentionMode
193
+ settings: VoiceSettings
194
+ budget: Budget | None
195
+ created_at: float
196
+ started_at: float | None
197
+ ended_at: float | None
198
+ error_code: str | None
199
+ error_message: str | None
200
+ generated_segments: int
201
+ total_segments: int
202
+ has_result: bool
203
+ result_expired: bool
204
+ result_integrity: ResultIntegrity | None
205
+ audio_duration_ms: int
206
+ source_text: str | None = None
207
+
208
+ @property
209
+ def model_id(self) -> str:
210
+ return self.settings.model_id
211
+
212
+
213
+ @dataclass(frozen=True, slots=True)
214
+ class ClientRecord:
215
+ """4.2's integration permission. Holds no credential and no verifier:
216
+ N-17 keeps those in ``echoact.security`` and out of backups."""
217
+
218
+ client_id: str
219
+ label: str
220
+ capabilities: frozenset[Capability]
221
+ active: bool
222
+ created_at: float
223
+ revoked_at: float | None = None
224
+ last_seen_at: float | None = None
225
+
226
+
227
+ @dataclass(frozen=True, slots=True)
228
+ class IdempotencyRecord:
229
+ """4.2's re-request record."""
230
+
231
+ client_id: str
232
+ key: str
233
+ request_digest: str
234
+ job_id: str
235
+ created_at: float
236
+ expires_at: float
237
+
238
+
239
+ @dataclass(frozen=True, slots=True)
240
+ class BackupRecord:
241
+ backup_id: str
242
+ kind: str
243
+ created_at: float
244
+ location: str
245
+ byte_size: int
246
+ item_count: int
247
+ schema_version: int
248
+ app_version: str
249
+ verified: bool
250
+ note: str | None
251
+
252
+
253
+ @dataclass(frozen=True, slots=True)
254
+ class StorageUsage:
255
+ """N-16's "ceiling and remaining space", in the units 4.1 fixes.
256
+
257
+ Model cache, user backups, and exported WAV files are deliberately absent:
258
+ 4.1 shows those separately and they are not the store's to count.
259
+ """
260
+
261
+ document_bytes: int
262
+ job_text_bytes: int
263
+ audio_bytes: int
264
+ limit_bytes: int
265
+
266
+ @property
267
+ def total_bytes(self) -> int:
268
+ return self.document_bytes + self.job_text_bytes + self.audio_bytes
269
+
270
+ @property
271
+ def remaining_bytes(self) -> int:
272
+ return max(0, self.limit_bytes - self.total_bytes)
273
+
274
+ def fits(self, additional_bytes: int) -> bool:
275
+ return self.total_bytes + max(0, additional_bytes) <= self.limit_bytes
276
+
277
+
278
+ @dataclass(frozen=True, slots=True)
279
+ class DeletionScope:
280
+ """F-43's preview. ``blocked_job_ids`` are jobs that are still active;
281
+ 5.3 requires cancellation first and forbids deleting one in part."""
282
+
283
+ jobs: int
284
+ documents: int
285
+ segments: int
286
+ results: int
287
+ audio_bytes: int
288
+ text_bytes: int
289
+ blocked_job_ids: tuple[str, ...]
290
+
291
+
292
+ @dataclass(frozen=True, slots=True)
293
+ class Deletion:
294
+ """What a delete actually removed, and the audio files left for the
295
+ caller to unlink.
296
+
297
+ The store never deletes a file. F-43 keeps user-exported WAVs out of any
298
+ deletion, and the only way to be sure of that is for file removal to
299
+ happen where the audio directory's layout is understood, not here.
300
+ """
301
+
302
+ documents: int
303
+ jobs: int
304
+ results: int
305
+ audio_paths: tuple[str, ...]
306
+
307
+
308
+ @dataclass(frozen=True, slots=True)
309
+ class ReconcileReport:
310
+ """F-45's start-up findings. Nothing here starts, regenerates, or plays
311
+ anything; it records what an abnormal termination left behind."""
312
+
313
+ interrupted_job_ids: tuple[str, ...]
314
+ canceled_job_ids: tuple[str, ...]
315
+ expired_result_ids: tuple[str, ...]
316
+ missing_result_ids: tuple[str, ...]
317
+ corrupt_result_ids: tuple[str, ...]
318
+
319
+ @property
320
+ def has_findings(self) -> bool:
321
+ return bool(
322
+ self.interrupted_job_ids
323
+ or self.canceled_job_ids
324
+ or self.missing_result_ids
325
+ or self.corrupt_result_ids
326
+ )
327
+
328
+
329
+ # ======================================================================
330
+ # Column lists
331
+ # ======================================================================
332
+
333
+ # Enumerated rather than ``SELECT *`` so that F-56's summary cannot pick up
334
+ # the source-text snapshot by accident when a column is added later.
335
+ _JOB_COLUMNS: Final = (
336
+ "job_id",
337
+ "kind",
338
+ "request_path",
339
+ "owner_client_id",
340
+ "client_label",
341
+ "state",
342
+ "retention",
343
+ "model_id",
344
+ "settings_json",
345
+ "budget_json",
346
+ "created_at",
347
+ "started_at",
348
+ "ended_at",
349
+ "error_code",
350
+ "error_message",
351
+ "idempotency_key",
352
+ "generated_segments",
353
+ "total_segments",
354
+ )
355
+
356
+ _SEGMENT_COLUMNS: Final = (
357
+ "segment_id",
358
+ "job_id",
359
+ "seq",
360
+ "source_start_codepoint_inclusive",
361
+ "source_end_codepoint_exclusive",
362
+ "spoken_text",
363
+ "language",
364
+ "audio_start_ms",
365
+ "audio_end_ms",
366
+ "trailing_silence_ms",
367
+ "audio_path",
368
+ "frame_count",
369
+ "ready",
370
+ )
371
+
372
+ _RESULT_COLUMNS: Final = (
373
+ "result_id",
374
+ "job_id",
375
+ "sample_rate",
376
+ "channels",
377
+ "sample_width_bits",
378
+ "frame_count",
379
+ "byte_size",
380
+ "digest",
381
+ "relative_path",
382
+ "created_at",
383
+ "expires_at",
384
+ "integrity_state",
385
+ "verified_at",
386
+ )
387
+
388
+ _ACTIVE_STATE_VALUES: Final = tuple(s.value for s in JobState if s.is_active)
389
+
390
+
391
+ # ======================================================================
392
+ # Row mapping
393
+ # ======================================================================
394
+
395
+
396
+ def _to_document(row: sqlite3.Row) -> Document:
397
+ return Document(
398
+ document_id=row["document_id"],
399
+ title=row["title"],
400
+ body=row["body"],
401
+ created_at=row["created_at"],
402
+ modified_at=row["modified_at"],
403
+ version=row["version"],
404
+ )
405
+
406
+
407
+ def _to_segment(row: sqlite3.Row) -> Segment:
408
+ start_ms, end_ms = row["audio_start_ms"], row["audio_end_ms"]
409
+ return Segment(
410
+ index=row["seq"],
411
+ source=TextRange(
412
+ row["source_start_codepoint_inclusive"], row["source_end_codepoint_exclusive"]
413
+ ),
414
+ spoken_text=row["spoken_text"],
415
+ language=row["language"],
416
+ time=None if start_ms is None or end_ms is None else TimeRange(start_ms, end_ms),
417
+ trailing_silence_ms=row["trailing_silence_ms"],
418
+ audio_path=row["audio_path"],
419
+ frame_count=row["frame_count"],
420
+ ready=bool(row["ready"]),
421
+ segment_id=row["segment_id"],
422
+ )
423
+
424
+
425
+ def _to_result(row: sqlite3.Row) -> Result:
426
+ return Result(
427
+ result_id=row["result_id"],
428
+ job_id=row["job_id"],
429
+ sample_rate=row["sample_rate"],
430
+ channels=row["channels"],
431
+ sample_width_bits=row["sample_width_bits"],
432
+ frame_count=row["frame_count"],
433
+ byte_size=row["byte_size"],
434
+ digest=row["digest"],
435
+ created_at=row["created_at"],
436
+ expires_at=row["expires_at"],
437
+ relative_path=row["relative_path"],
438
+ )
439
+
440
+
441
+ def _to_client(row: sqlite3.Row) -> ClientRecord:
442
+ return ClientRecord(
443
+ client_id=row["client_id"],
444
+ label=row["label"],
445
+ capabilities=frozenset(Capability(c) for c in json.loads(row["capabilities"])),
446
+ active=bool(row["active"]),
447
+ created_at=row["created_at"],
448
+ revoked_at=row["revoked_at"],
449
+ last_seen_at=row["last_seen_at"],
450
+ )
451
+
452
+
453
+ def _to_backup(row: sqlite3.Row) -> BackupRecord:
454
+ return BackupRecord(
455
+ backup_id=row["backup_id"],
456
+ kind=row["kind"],
457
+ created_at=row["created_at"],
458
+ location=row["location"],
459
+ byte_size=row["byte_size"],
460
+ item_count=row["item_count"],
461
+ schema_version=row["schema_version"],
462
+ app_version=row["app_version"],
463
+ verified=bool(row["verified"]),
464
+ note=row["note"],
465
+ )
466
+
467
+
468
+ def _to_idempotency(row: sqlite3.Row) -> IdempotencyRecord:
469
+ return IdempotencyRecord(
470
+ client_id=row["client_id"],
471
+ key=row["key"],
472
+ request_digest=row["request_digest"],
473
+ job_id=row["job_id"],
474
+ created_at=row["created_at"],
475
+ expires_at=row["expires_at"],
476
+ )
477
+
478
+
479
+ def _duration_ms(frame_count: int | None, sample_rate: int | None) -> int:
480
+ if not frame_count or not sample_rate:
481
+ return 0
482
+ return int(round(frame_count * 1000 / sample_rate))
483
+
484
+
485
+ def _clamp_page(limit: int | None, offset: int) -> tuple[int, int]:
486
+ size = LIST_PAGE_DEFAULT if limit is None else int(limit)
487
+ size = max(1, min(size, LIST_PAGE_MAX))
488
+ return size, max(0, int(offset))
489
+
490
+
491
+ def _like_pattern(query: str) -> str:
492
+ return "%" + _LIKE_SPECIAL.sub(r"\\\1", query) + "%"
493
+
494
+
495
+ def _fts_phrase(query: str) -> str:
496
+ """Quote a user's words as one FTS5 phrase.
497
+
498
+ N-18 treats text a user or a client supplies as data. Unquoted, a query
499
+ containing ``AND``, ``*`` or ``"`` is FTS5 *syntax*, and at best it makes
500
+ the search behave in a way nobody asked for.
501
+ """
502
+ return '"' + query.replace('"', '""') + '"'
503
+
504
+
505
+ # ======================================================================
506
+ # Store
507
+ # ======================================================================
508
+
509
+
510
+ class Store:
511
+ """The database, as the rest of the application sees it.
512
+
513
+ Construct one per process. It is safe to use from several threads: each
514
+ gets its own connection, and every write is serialised by SQLite itself.
515
+ """
516
+
517
+ def __init__(
518
+ self,
519
+ path: str | Path | None = None,
520
+ *,
521
+ busy_timeout_s: float = DEFAULT_BUSY_TIMEOUT_S,
522
+ retention_limit_bytes: int = RETENTION_DEFAULT_BYTES,
523
+ audio_root: str | Path | None = None,
524
+ migrate_on_open: bool = True,
525
+ backup_dir: Path | None = None,
526
+ ) -> None:
527
+ self._path = Path(path) if path is not None else db_path()
528
+ self._busy_timeout_s = max(0.0, float(busy_timeout_s))
529
+ self._retention_limit_bytes = int(retention_limit_bytes)
530
+ self._audio_root = Path(audio_root) if audio_root is not None else None
531
+ self._local = threading.local()
532
+ self._lock = threading.Lock()
533
+ self._connections: list[sqlite3.Connection] = []
534
+ self._closed = False
535
+
536
+ conn = self._conn()
537
+ try:
538
+ if migrate_on_open:
539
+ migrate(conn, backup_dir=backup_dir)
540
+ else:
541
+ assert_compatible(conn)
542
+ self._fts = has_fts_index(conn)
543
+ self._schema_version = current_version(conn)
544
+ except BaseException:
545
+ # A database this build refuses to touch (N-15) must not leave a
546
+ # handle open on it either.
547
+ self.close()
548
+ raise
549
+
550
+ # -- lifecycle ------------------------------------------------------
551
+
552
+ @property
553
+ def path(self) -> Path:
554
+ return self._path
555
+
556
+ @property
557
+ def fts_enabled(self) -> bool:
558
+ """Whether F-40's search runs over an FTS5 index or degrades to LIKE."""
559
+ return self._fts
560
+
561
+ @property
562
+ def schema_version(self) -> int:
563
+ return self._schema_version
564
+
565
+ @property
566
+ def supported_schema_version(self) -> int:
567
+ return SUPPORTED_SCHEMA_VERSION
568
+
569
+ @property
570
+ def audio_root(self) -> Path:
571
+ # Resolved late: ``paths.data_dir`` is cached, and a test that
572
+ # redirects it does so after this object might already exist.
573
+ return self._audio_root if self._audio_root is not None else audio_dir()
574
+
575
+ @property
576
+ def busy_timeout_s(self) -> float:
577
+ return self._busy_timeout_s
578
+
579
+ def close(self) -> None:
580
+ """Close every thread's connection.
581
+
582
+ Connections are created with ``check_same_thread=False`` purely so
583
+ that shutdown can close them from the thread that owns the Store;
584
+ they are still used by one thread each.
585
+ """
586
+ with self._lock:
587
+ self._closed = True
588
+ connections, self._connections = self._connections, []
589
+ for conn in connections:
590
+ try:
591
+ conn.close()
592
+ except sqlite3.Error:
593
+ continue
594
+
595
+ def __enter__(self) -> Store:
596
+ return self
597
+
598
+ def __exit__(self, *exc: object) -> None:
599
+ self.close()
600
+
601
+ # -- connections ----------------------------------------------------
602
+
603
+ def _conn(self) -> sqlite3.Connection:
604
+ if self._closed:
605
+ raise EchoActError(Code.DB_UNAVAILABLE, "The local database is closed.")
606
+ conn: sqlite3.Connection | None = getattr(self._local, "conn", None)
607
+ if conn is not None:
608
+ return conn
609
+ conn = self._open()
610
+ self._local.conn = conn
611
+ self._local.depth = 0
612
+ with self._lock:
613
+ self._connections.append(conn)
614
+ return conn
615
+
616
+ def _open(self) -> sqlite3.Connection:
617
+ try:
618
+ self._path.parent.mkdir(parents=True, exist_ok=True)
619
+ conn = sqlite3.connect(
620
+ self._path,
621
+ timeout=self._busy_timeout_s,
622
+ isolation_level=None,
623
+ check_same_thread=False,
624
+ )
625
+ except (sqlite3.Error, OSError) as exc:
626
+ raise translate_sqlite_error(exc) from exc
627
+ conn.row_factory = sqlite3.Row
628
+ try:
629
+ # A pragma cannot take a bound parameter; the value is an int
630
+ # this module computed, never anything a caller supplied.
631
+ conn.execute(f"PRAGMA busy_timeout = {int(self._busy_timeout_s * 1000)}")
632
+ conn.execute("PRAGMA journal_mode = WAL")
633
+ conn.execute("PRAGMA foreign_keys = ON")
634
+ # FULL, not NORMAL: N-14 asks for previously sound data to
635
+ # survive a save that fails midway, and A.2 expects the worker
636
+ # to be killable at any instant. Writes here are small and rare
637
+ # enough that the extra fsync costs nothing worth having.
638
+ conn.execute("PRAGMA synchronous = FULL")
639
+ except sqlite3.Error as exc:
640
+ conn.close()
641
+ raise translate_sqlite_error(exc) from exc
642
+ return conn
643
+
644
+ @contextmanager
645
+ def transaction(self) -> Iterator[sqlite3.Connection]:
646
+ """Run a unit of work atomically (N-14).
647
+
648
+ ``BEGIN IMMEDIATE`` rather than the default deferred begin: a
649
+ deferred transaction that has already read cannot always be upgraded
650
+ to a writer, and SQLite then returns ``SQLITE_BUSY`` *immediately*
651
+ rather than waiting out the busy timeout, so the bound this class
652
+ advertises would not apply to the very case it exists for.
653
+
654
+ Nesting is allowed and uses savepoints, so a caller can compose two
655
+ operations -- creating a job and claiming its re-request key, say --
656
+ into one all-or-nothing write.
657
+ """
658
+ conn = self._conn()
659
+ depth: int = getattr(self._local, "depth", 0)
660
+ savepoint = f"echoact_sp{depth}"
661
+ try:
662
+ if depth == 0:
663
+ conn.execute("BEGIN IMMEDIATE")
664
+ else:
665
+ conn.execute(f"SAVEPOINT {savepoint}")
666
+ except sqlite3.Error as exc:
667
+ raise translate_sqlite_error(exc, retry_after_s=self._busy_timeout_s) from exc
668
+ self._local.depth = depth + 1
669
+ try:
670
+ yield conn
671
+ except BaseException:
672
+ try:
673
+ if depth == 0:
674
+ conn.execute("ROLLBACK")
675
+ else:
676
+ conn.execute(f"ROLLBACK TO {savepoint}")
677
+ conn.execute(f"RELEASE {savepoint}")
678
+ except sqlite3.Error:
679
+ pass
680
+ raise
681
+ else:
682
+ try:
683
+ if depth == 0:
684
+ conn.execute("COMMIT")
685
+ else:
686
+ conn.execute(f"RELEASE {savepoint}")
687
+ except sqlite3.Error as exc:
688
+ try:
689
+ conn.execute("ROLLBACK")
690
+ except sqlite3.Error:
691
+ pass
692
+ raise translate_sqlite_error(exc, retry_after_s=self._busy_timeout_s) from exc
693
+ finally:
694
+ self._local.depth = depth
695
+
696
+ # ==================================================================
697
+ # Documents (F-38, F-40)
698
+ # ==================================================================
699
+
700
+ @_guard
701
+ def save_document(
702
+ self,
703
+ title: str,
704
+ body: str,
705
+ *,
706
+ document_id: str | None = None,
707
+ at: float | None = None,
708
+ ) -> Document:
709
+ """F-38's explicit save.
710
+
711
+ Checked against the retention ceiling first: N-16 forbids making room
712
+ by deleting something the user saved on purpose, so a save that does
713
+ not fit is refused instead.
714
+ """
715
+ moment = ids.now() if at is None else at
716
+ doc = Document(
717
+ document_id=document_id or ids.document_id(),
718
+ title=title,
719
+ body=body,
720
+ created_at=moment,
721
+ modified_at=moment,
722
+ version=1,
723
+ )
724
+ with self.transaction() as conn:
725
+ self.require_retention_capacity(text_bytes(title) + text_bytes(body))
726
+ conn.execute(
727
+ "INSERT INTO documents (document_id, title, body, created_at, modified_at,"
728
+ " version) VALUES (?, ?, ?, ?, ?, ?)",
729
+ (doc.document_id, doc.title, doc.body, moment, moment, doc.version),
730
+ )
731
+ return doc
732
+
733
+ @_guard
734
+ def get_document(self, document_id: str) -> Document:
735
+ row = self._conn().execute(
736
+ "SELECT * FROM documents WHERE document_id = ?", (document_id,)
737
+ ).fetchone()
738
+ if row is None:
739
+ raise EchoActError(Code.NOT_FOUND, "No such document.", detail={"id": document_id})
740
+ return _to_document(row)
741
+
742
+ @_guard
743
+ def list_documents(self, *, limit: int | None = None, offset: int = 0) -> Page[DocumentSummary]:
744
+ size, start = _clamp_page(limit, offset)
745
+ conn = self._conn()
746
+ total = int(conn.execute("SELECT count(*) FROM documents").fetchone()[0])
747
+ rows = conn.execute(
748
+ "SELECT document_id, title, created_at, modified_at, version,"
749
+ " length(body) AS body_codepoints FROM documents"
750
+ " ORDER BY modified_at DESC, document_id LIMIT ? OFFSET ?",
751
+ (size, start),
752
+ ).fetchall()
753
+ return Page(tuple(self._document_summary(r) for r in rows), total, size, start)
754
+
755
+ @staticmethod
756
+ def _document_summary(row: sqlite3.Row) -> DocumentSummary:
757
+ return DocumentSummary(
758
+ document_id=row["document_id"],
759
+ title=row["title"],
760
+ created_at=row["created_at"],
761
+ modified_at=row["modified_at"],
762
+ version=row["version"],
763
+ body_codepoints=row["body_codepoints"],
764
+ )
765
+
766
+ @_guard
767
+ def update_document(
768
+ self,
769
+ document_id: str,
770
+ *,
771
+ title: str | None = None,
772
+ body: str | None = None,
773
+ at: float | None = None,
774
+ ) -> Document:
775
+ """F-38's edit.
776
+
777
+ N-14 separates document edits from job snapshots: nothing here can
778
+ reach ``jobs.source_text``, and there is no foreign key that would
779
+ let a cascade do it either.
780
+ """
781
+ moment = ids.now() if at is None else at
782
+ with self.transaction() as conn:
783
+ current = self.get_document(document_id)
784
+ new_title = current.title if title is None else title
785
+ new_body = current.body if body is None else body
786
+ grew = (text_bytes(new_title) + text_bytes(new_body)) - (
787
+ text_bytes(current.title) + text_bytes(current.body)
788
+ )
789
+ if grew > 0:
790
+ self.require_retention_capacity(grew)
791
+ conn.execute(
792
+ "UPDATE documents SET title = ?, body = ?, modified_at = ?, version = version + 1"
793
+ " WHERE document_id = ?",
794
+ (new_title, new_body, moment, document_id),
795
+ )
796
+ return Document(
797
+ document_id=document_id,
798
+ title=new_title,
799
+ body=new_body,
800
+ created_at=current.created_at,
801
+ modified_at=moment,
802
+ version=current.version + 1,
803
+ )
804
+
805
+ @_guard
806
+ def delete_document(self, document_id: str) -> None:
807
+ """Delete one document and nothing else.
808
+
809
+ F-43 is explicit that deleting a document is never presented as
810
+ having deleted a separate job's source text -- and here it cannot,
811
+ because a job's snapshot is its own copy.
812
+ """
813
+ with self.transaction() as conn:
814
+ cur = conn.execute("DELETE FROM documents WHERE document_id = ?", (document_id,))
815
+ if cur.rowcount == 0:
816
+ raise EchoActError(Code.NOT_FOUND, "No such document.", detail={"id": document_id})
817
+
818
+ @_guard
819
+ def search_documents(
820
+ self, query: str, *, limit: int | None = None, offset: int = 0
821
+ ) -> Page[DocumentSummary]:
822
+ """F-40's search over titles and retained body text.
823
+
824
+ Runs on the FTS5 index where the build has one. Two cases fall back
825
+ to ``LIKE``: a build without FTS5, and a query shorter than a
826
+ trigram, which the index cannot match at all. Falling back keeps the
827
+ feature working rather than quietly returning nothing.
828
+ """
829
+ text = query.strip()
830
+ if not text:
831
+ return self.list_documents(limit=limit, offset=offset)
832
+ size, start = _clamp_page(limit, offset)
833
+ conn = self._conn()
834
+ if self._fts and len(text) >= _MIN_FTS_QUERY_CODEPOINTS:
835
+ phrase = _fts_phrase(text)
836
+ total = int(
837
+ conn.execute(
838
+ "SELECT count(*) FROM documents_fts WHERE documents_fts MATCH ?", (phrase,)
839
+ ).fetchone()[0]
840
+ )
841
+ rows = conn.execute(
842
+ "SELECT d.document_id, d.title, d.created_at, d.modified_at, d.version,"
843
+ " length(d.body) AS body_codepoints"
844
+ " FROM documents_fts JOIN documents d ON d.rowid = documents_fts.rowid"
845
+ " WHERE documents_fts MATCH ? ORDER BY rank LIMIT ? OFFSET ?",
846
+ (phrase, size, start),
847
+ ).fetchall()
848
+ else:
849
+ pattern = _like_pattern(text)
850
+ where = (
851
+ " WHERE title LIKE ? ESCAPE '\\' OR body LIKE ? ESCAPE '\\'"
852
+ )
853
+ total = int(
854
+ conn.execute(
855
+ "SELECT count(*) FROM documents" + where, (pattern, pattern)
856
+ ).fetchone()[0]
857
+ )
858
+ rows = conn.execute(
859
+ "SELECT document_id, title, created_at, modified_at, version,"
860
+ " length(body) AS body_codepoints FROM documents" + where
861
+ + " ORDER BY modified_at DESC, document_id LIMIT ? OFFSET ?",
862
+ (pattern, pattern, size, start),
863
+ ).fetchall()
864
+ return Page(tuple(self._document_summary(r) for r in rows), total, size, start)
865
+
866
+ # ==================================================================
867
+ # Jobs (F-39, F-40, F-41)
868
+ # ==================================================================
869
+
870
+ @_guard
871
+ def create_job(self, job: Job, *, at: float | None = None) -> Job:
872
+ """Record an accepted job and, if it already has them, its segments.
873
+
874
+ A retained job's snapshot is counted against the ceiling before the
875
+ row is written, so 4.1's "jobs requesting retention fail with a
876
+ storage error" happens at acceptance rather than at completion, when
877
+ the user has already waited.
878
+ """
879
+ moment = job.created_at if at is None else at
880
+ with self.transaction() as conn:
881
+ if job.retention is RetentionMode.RETAINED:
882
+ self.require_retention_capacity(text_bytes(job.source_text))
883
+ conn.execute(
884
+ "INSERT INTO jobs (job_id, kind, request_path, owner_client_id, client_label,"
885
+ " state, retention, source_text, model_id, settings_json, budget_json,"
886
+ " created_at, started_at, ended_at, error_code, error_message, idempotency_key,"
887
+ " generated_segments, total_segments)"
888
+ " VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
889
+ (
890
+ job.job_id,
891
+ job.kind.value,
892
+ job.request_path.value,
893
+ job.owner_client_id,
894
+ job.client_label,
895
+ job.state.value,
896
+ job.retention.value,
897
+ job.source_text,
898
+ job.settings.model_id,
899
+ json.dumps(job.settings.to_dict(), ensure_ascii=False),
900
+ None if job.budget is None else json.dumps(job.budget.to_dict()),
901
+ moment,
902
+ job.started_at,
903
+ job.ended_at,
904
+ job.error_code,
905
+ job.error_message,
906
+ job.idempotency_key,
907
+ job.generated_segments,
908
+ job.total_segments or len(job.segments),
909
+ ),
910
+ )
911
+ if job.segments:
912
+ self.insert_segments(job.job_id, job.segments)
913
+ return job
914
+
915
+ @_guard
916
+ def get_job(
917
+ self,
918
+ job_id: str,
919
+ *,
920
+ include_source_text: bool = True,
921
+ include_segments: bool = True,
922
+ ) -> Job:
923
+ """F-41's reopen: source text, settings, and the result if one exists."""
924
+ conn = self._conn()
925
+ columns = ", ".join(_JOB_COLUMNS)
926
+ row = conn.execute(
927
+ f"SELECT {columns} FROM jobs WHERE job_id = ?", (job_id,)
928
+ ).fetchone()
929
+ if row is None:
930
+ raise EchoActError(Code.NOT_FOUND, "No such job.", detail={"id": job_id})
931
+ source = ""
932
+ if include_source_text:
933
+ source = self.get_job_text(job_id) or ""
934
+ segments = self.list_segments(job_id) if include_segments else []
935
+ return self._to_job(row, source_text=source, segments=segments)
936
+
937
+ @_guard
938
+ def get_job_text(self, job_id: str) -> str | None:
939
+ """F-56's separate snapshot request.
940
+
941
+ ``None`` means the snapshot is not stored: either the job was one-off
942
+ or its retention window has closed and F-42 required the text to go.
943
+ The caller decides what that is on its surface; the store does not
944
+ pretend an empty string is the same thing.
945
+ """
946
+ row = self._conn().execute(
947
+ "SELECT source_text FROM jobs WHERE job_id = ?", (job_id,)
948
+ ).fetchone()
949
+ if row is None:
950
+ raise EchoActError(Code.NOT_FOUND, "No such job.", detail={"id": job_id})
951
+ return row["source_text"]
952
+
953
+ def _to_job(
954
+ self,
955
+ row: sqlite3.Row,
956
+ *,
957
+ source_text: str,
958
+ segments: Sequence[Segment],
959
+ result: Result | None = None,
960
+ ) -> Job:
961
+ return Job(
962
+ job_id=row["job_id"],
963
+ kind=JobKind(row["kind"]),
964
+ request_path=RequestPath(row["request_path"]),
965
+ owner_client_id=row["owner_client_id"],
966
+ state=JobState(row["state"]),
967
+ source_text=source_text,
968
+ settings=VoiceSettings.from_dict(json.loads(row["settings_json"])),
969
+ budget=None if row["budget_json"] is None else Budget.from_dict(
970
+ json.loads(row["budget_json"])
971
+ ),
972
+ retention=RetentionMode(row["retention"]),
973
+ created_at=row["created_at"],
974
+ started_at=row["started_at"],
975
+ ended_at=row["ended_at"],
976
+ error_code=row["error_code"],
977
+ error_message=row["error_message"],
978
+ client_label=row["client_label"],
979
+ idempotency_key=row["idempotency_key"],
980
+ segments=list(segments),
981
+ result=result if result is not None else self.get_result_for_job(row["job_id"]),
982
+ generated_segments=row["generated_segments"],
983
+ total_segments=row["total_segments"],
984
+ )
985
+
986
+ @_guard
987
+ def list_jobs(
988
+ self,
989
+ *,
990
+ created_from: float | None = None,
991
+ created_to: float | None = None,
992
+ model_id: str | None = None,
993
+ states: Iterable[JobState] | None = None,
994
+ owner_client_id: str | None = None,
995
+ retention: RetentionMode | None = None,
996
+ include_source_text: bool = False,
997
+ limit: int | None = None,
998
+ offset: int = 0,
999
+ ) -> Page[JobSummary]:
1000
+ """F-40's filtered, paged history.
1001
+
1002
+ Date, model, and state are all indexed columns, and the result's
1003
+ length comes from the joined ``results`` row -- no WAV is opened and
1004
+ no snapshot is read unless ``include_source_text`` asks for one,
1005
+ which is F-56's default made structural.
1006
+ """
1007
+ size, start = _clamp_page(limit, offset)
1008
+ where: list[str] = []
1009
+ args: list[Any] = []
1010
+ if created_from is not None:
1011
+ where.append("j.created_at >= ?")
1012
+ args.append(created_from)
1013
+ if created_to is not None:
1014
+ where.append("j.created_at <= ?")
1015
+ args.append(created_to)
1016
+ if model_id is not None:
1017
+ where.append("j.model_id = ?")
1018
+ args.append(model_id)
1019
+ if owner_client_id is not None:
1020
+ where.append("j.owner_client_id = ?")
1021
+ args.append(owner_client_id)
1022
+ if retention is not None:
1023
+ where.append("j.retention = ?")
1024
+ args.append(retention.value)
1025
+ state_values = tuple(JobState(s).value for s in states) if states is not None else ()
1026
+ if state_values:
1027
+ where.append(f"j.state IN ({','.join('?' * len(state_values))})")
1028
+ args.extend(state_values)
1029
+ clause = (" WHERE " + " AND ".join(where)) if where else ""
1030
+
1031
+ conn = self._conn()
1032
+ total = int(
1033
+ conn.execute(f"SELECT count(*) FROM jobs j{clause}", tuple(args)).fetchone()[0]
1034
+ )
1035
+ columns = ", ".join(f"j.{c}" for c in _JOB_COLUMNS)
1036
+ if include_source_text:
1037
+ columns += ", j.source_text"
1038
+ rows = conn.execute(
1039
+ f"SELECT {columns}, r.result_id AS r_id, r.frame_count AS r_frames,"
1040
+ " r.sample_rate AS r_rate, r.expires_at AS r_expires,"
1041
+ " r.integrity_state AS r_integrity"
1042
+ " FROM jobs j LEFT JOIN results r ON r.job_id = j.job_id"
1043
+ f"{clause} ORDER BY j.created_at DESC, j.job_id LIMIT ? OFFSET ?",
1044
+ (*args, size, start),
1045
+ ).fetchall()
1046
+ now = ids.now()
1047
+ items = tuple(
1048
+ self._job_summary(r, now=now, with_text=include_source_text) for r in rows
1049
+ )
1050
+ return Page(items, total, size, start)
1051
+
1052
+ @staticmethod
1053
+ def _job_summary(row: sqlite3.Row, *, now: float, with_text: bool) -> JobSummary:
1054
+ expires = row["r_expires"]
1055
+ integrity = row["r_integrity"]
1056
+ return JobSummary(
1057
+ job_id=row["job_id"],
1058
+ kind=JobKind(row["kind"]),
1059
+ request_path=RequestPath(row["request_path"]),
1060
+ owner_client_id=row["owner_client_id"],
1061
+ client_label=row["client_label"],
1062
+ state=JobState(row["state"]),
1063
+ retention=RetentionMode(row["retention"]),
1064
+ settings=VoiceSettings.from_dict(json.loads(row["settings_json"])),
1065
+ budget=None if row["budget_json"] is None else Budget.from_dict(
1066
+ json.loads(row["budget_json"])
1067
+ ),
1068
+ created_at=row["created_at"],
1069
+ started_at=row["started_at"],
1070
+ ended_at=row["ended_at"],
1071
+ error_code=row["error_code"],
1072
+ error_message=row["error_message"],
1073
+ generated_segments=row["generated_segments"],
1074
+ total_segments=row["total_segments"],
1075
+ has_result=row["r_id"] is not None,
1076
+ result_expired=expires is not None and expires <= now,
1077
+ result_integrity=None if integrity is None else ResultIntegrity(integrity),
1078
+ audio_duration_ms=_duration_ms(row["r_frames"], row["r_rate"]),
1079
+ source_text=row["source_text"] if with_text else None,
1080
+ )
1081
+
1082
+ @_guard
1083
+ def update_job_state(
1084
+ self,
1085
+ job_id: str,
1086
+ state: JobState,
1087
+ *,
1088
+ at: float | None = None,
1089
+ force: bool = False,
1090
+ ) -> JobState:
1091
+ """Move a job along Section 5.1's state machine.
1092
+
1093
+ The transition is checked against ``domain.can_transition`` rather
1094
+ than trusted, which is what stops a late worker reply from
1095
+ overwriting ``Complete`` with ``Canceled``: 5.1 keeps whichever
1096
+ terminal state was confirmed first, and ``Complete`` has no outgoing
1097
+ transitions at all. Repeating the state a job is already in is a
1098
+ no-op, so F-49's "repeated cancellation adds no further side effects"
1099
+ holds here too.
1100
+ """
1101
+ moment = ids.now() if at is None else at
1102
+ with self.transaction() as conn:
1103
+ row = conn.execute(
1104
+ "SELECT state, started_at FROM jobs WHERE job_id = ?", (job_id,)
1105
+ ).fetchone()
1106
+ if row is None:
1107
+ raise EchoActError(Code.NOT_FOUND, "No such job.", detail={"id": job_id})
1108
+ current = JobState(row["state"])
1109
+ if current is state:
1110
+ return current
1111
+ if not force:
1112
+ check_transition(current, state)
1113
+ started = row["started_at"]
1114
+ if started is None and state in (JobState.PREPARING_MODEL, JobState.GENERATING):
1115
+ started = moment
1116
+ ended = moment if state.is_terminal else None
1117
+ conn.execute(
1118
+ "UPDATE jobs SET state = ?, started_at = ?,"
1119
+ " ended_at = CASE WHEN ? IS NULL THEN ended_at ELSE ? END WHERE job_id = ?",
1120
+ (state.value, started, ended, ended, job_id),
1121
+ )
1122
+ return state
1123
+
1124
+ @_guard
1125
+ def record_job_error(
1126
+ self,
1127
+ job_id: str,
1128
+ code: Code | str,
1129
+ message: str | None = None,
1130
+ *,
1131
+ state: JobState = JobState.FAILED,
1132
+ at: float | None = None,
1133
+ ) -> JobState:
1134
+ """F-39 keeps the error code with the job; N-25 traces a problem by it.
1135
+
1136
+ The message is the code's own English text unless the caller has
1137
+ something more specific. Neither field ever carries body text --
1138
+ N-20 keeps that out of anything that is later exported.
1139
+ """
1140
+ as_code = Code(code) if not isinstance(code, Code) else code
1141
+ text = message if message is not None else EchoActError(as_code).message
1142
+ moment = ids.now() if at is None else at
1143
+ with self.transaction() as conn:
1144
+ final = self.update_job_state(job_id, state, at=moment)
1145
+ conn.execute(
1146
+ "UPDATE jobs SET error_code = ?, error_message = ? WHERE job_id = ?",
1147
+ (as_code.value, text, job_id),
1148
+ )
1149
+ return final
1150
+
1151
+ @_guard
1152
+ def set_job_budget(self, job_id: str, budget: Budget) -> None:
1153
+ """F-78 distinguishes the setting on screen from the one in force;
1154
+ this is the one in force, recorded on the job it applied to."""
1155
+ with self.transaction() as conn:
1156
+ cur = conn.execute(
1157
+ "UPDATE jobs SET budget_json = ? WHERE job_id = ?",
1158
+ (json.dumps(budget.to_dict()), job_id),
1159
+ )
1160
+ if cur.rowcount == 0:
1161
+ raise EchoActError(Code.NOT_FOUND, "No such job.", detail={"id": job_id})
1162
+
1163
+ @_guard
1164
+ def clear_job_source_text(self, job_id: str) -> None:
1165
+ """F-42: a one-off job's body text leaves permanent storage.
1166
+
1167
+ The job row stays, because 4.2 still requires the terminal job and
1168
+ its reason to be returned to a repeated re-request key.
1169
+ """
1170
+ with self.transaction() as conn:
1171
+ conn.execute("UPDATE jobs SET source_text = NULL WHERE job_id = ?", (job_id,))
1172
+
1173
+ @_guard
1174
+ def attach_result(self, result: Result, *, at: float | None = None) -> Result:
1175
+ """F-41's "if a result exists it can be played and saved", recorded.
1176
+
1177
+ A retained job's audio is checked against the ceiling here as well as
1178
+ at acceptance, because 4.1 lets the limit be reached at runtime and
1179
+ N-16 refuses the retention rather than deleting something to fit it.
1180
+ """
1181
+ moment = result.created_at if at is None else at
1182
+ result_id = result.result_id or ids.result_id()
1183
+ with self.transaction() as conn:
1184
+ row = conn.execute(
1185
+ "SELECT retention FROM jobs WHERE job_id = ?", (result.job_id,)
1186
+ ).fetchone()
1187
+ if row is None:
1188
+ raise EchoActError(Code.NOT_FOUND, "No such job.", detail={"id": result.job_id})
1189
+ if RetentionMode(row["retention"]) is RetentionMode.RETAINED:
1190
+ self.require_retention_capacity(result.byte_size)
1191
+ conn.execute(
1192
+ "INSERT INTO results (result_id, job_id, sample_rate, channels,"
1193
+ " sample_width_bits, frame_count, byte_size, digest, relative_path,"
1194
+ " created_at, expires_at, integrity_state, verified_at)"
1195
+ " VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, 'unverified', NULL)",
1196
+ (
1197
+ result_id,
1198
+ result.job_id,
1199
+ result.sample_rate,
1200
+ result.channels,
1201
+ result.sample_width_bits,
1202
+ result.frame_count,
1203
+ result.byte_size,
1204
+ result.digest,
1205
+ result.relative_path,
1206
+ moment,
1207
+ result.expires_at,
1208
+ ),
1209
+ )
1210
+ return replace(result, result_id=result_id, created_at=moment)
1211
+
1212
+ @_guard
1213
+ def preview_deletion(
1214
+ self,
1215
+ *,
1216
+ job_ids: Sequence[str] = (),
1217
+ document_ids: Sequence[str] = (),
1218
+ all_history: bool = False,
1219
+ ) -> DeletionScope:
1220
+ """F-43's scope preview, so a confirmation can state what goes.
1221
+
1222
+ Active jobs are reported rather than counted: 5.3 requires
1223
+ cancellation first and forbids deleting a running job in part.
1224
+ """
1225
+ conn = self._conn()
1226
+ if all_history:
1227
+ job_rows = conn.execute("SELECT job_id, state FROM jobs").fetchall()
1228
+ elif job_ids:
1229
+ marks = ",".join("?" * len(job_ids))
1230
+ job_rows = conn.execute(
1231
+ f"SELECT job_id, state FROM jobs WHERE job_id IN ({marks})", tuple(job_ids)
1232
+ ).fetchall()
1233
+ else:
1234
+ job_rows = []
1235
+ chosen = tuple(r["job_id"] for r in job_rows)
1236
+ blocked = tuple(r["job_id"] for r in job_rows if JobState(r["state"]).is_active)
1237
+
1238
+ segments = results = audio = text = documents = 0
1239
+ if chosen:
1240
+ marks = ",".join("?" * len(chosen))
1241
+ segments = int(
1242
+ conn.execute(
1243
+ f"SELECT count(*) FROM segments WHERE job_id IN ({marks})", chosen
1244
+ ).fetchone()[0]
1245
+ )
1246
+ row = conn.execute(
1247
+ f"SELECT count(*), coalesce(sum(byte_size), 0) FROM results"
1248
+ f" WHERE job_id IN ({marks})",
1249
+ chosen,
1250
+ ).fetchone()
1251
+ results, audio = int(row[0]), int(row[1])
1252
+ text = int(
1253
+ conn.execute(
1254
+ "SELECT coalesce(sum(length(CAST(source_text AS BLOB))), 0) FROM jobs"
1255
+ f" WHERE job_id IN ({marks})",
1256
+ chosen,
1257
+ ).fetchone()[0]
1258
+ )
1259
+ if document_ids:
1260
+ marks = ",".join("?" * len(document_ids))
1261
+ row = conn.execute(
1262
+ "SELECT count(*), coalesce(sum(length(CAST(title AS BLOB))"
1263
+ f" + length(CAST(body AS BLOB))), 0) FROM documents WHERE document_id IN ({marks})",
1264
+ tuple(document_ids),
1265
+ ).fetchone()
1266
+ documents, doc_bytes = int(row[0]), int(row[1])
1267
+ text += doc_bytes
1268
+ return DeletionScope(
1269
+ jobs=len(chosen),
1270
+ documents=documents,
1271
+ segments=segments,
1272
+ results=results,
1273
+ audio_bytes=audio,
1274
+ text_bytes=text,
1275
+ blocked_job_ids=blocked,
1276
+ )
1277
+
1278
+ @_guard
1279
+ def delete_job(self, job_id: str, *, force: bool = False) -> Deletion:
1280
+ """F-43's job deletion.
1281
+
1282
+ Refuses while the job is active: 5.3 requires cancellation first, and
1283
+ ``DELETE_BLOCKED_IN_USE`` is retryable so N-16's "deletion failures
1284
+ are shown as retryable and are not mistaken for completed deletions"
1285
+ holds at the surface without further translation.
1286
+ """
1287
+ with self.transaction() as conn:
1288
+ row = conn.execute(
1289
+ "SELECT state FROM jobs WHERE job_id = ?", (job_id,)
1290
+ ).fetchone()
1291
+ if row is None:
1292
+ raise EchoActError(Code.NOT_FOUND, "No such job.", detail={"id": job_id})
1293
+ if JobState(row["state"]).is_active and not force:
1294
+ raise EchoActError(
1295
+ Code.DELETE_BLOCKED_IN_USE,
1296
+ "Cancel this job before deleting it.",
1297
+ detail={"job_id": job_id, "state": row["state"]},
1298
+ retry_after_s=self._busy_timeout_s,
1299
+ )
1300
+ paths = self._audio_paths_for([job_id], conn)
1301
+ results = int(
1302
+ conn.execute(
1303
+ "SELECT count(*) FROM results WHERE job_id = ?", (job_id,)
1304
+ ).fetchone()[0]
1305
+ )
1306
+ conn.execute("DELETE FROM jobs WHERE job_id = ?", (job_id,))
1307
+ return Deletion(documents=0, jobs=1, results=results, audio_paths=paths)
1308
+
1309
+ @_guard
1310
+ def delete_all_history(self, *, force: bool = False) -> Deletion:
1311
+ """F-43's "delete all history". Documents are a separate scope and
1312
+ are untouched, which is the distinction F-43 insists on."""
1313
+ with self.transaction() as conn:
1314
+ active = [
1315
+ r["job_id"]
1316
+ for r in conn.execute(
1317
+ "SELECT job_id FROM jobs WHERE state IN"
1318
+ f" ({','.join('?' * len(_ACTIVE_STATE_VALUES))})",
1319
+ _ACTIVE_STATE_VALUES,
1320
+ ).fetchall()
1321
+ ]
1322
+ if active and not force:
1323
+ raise EchoActError(
1324
+ Code.DELETE_BLOCKED_IN_USE,
1325
+ "Cancel the running job before deleting history.",
1326
+ detail={"job_ids": active[:10]},
1327
+ retry_after_s=self._busy_timeout_s,
1328
+ )
1329
+ job_ids = [r["job_id"] for r in conn.execute("SELECT job_id FROM jobs").fetchall()]
1330
+ paths = self._audio_paths_for(job_ids, conn)
1331
+ results = int(conn.execute("SELECT count(*) FROM results").fetchone()[0])
1332
+ conn.execute("DELETE FROM jobs")
1333
+ return Deletion(documents=0, jobs=len(job_ids), results=results, audio_paths=paths)
1334
+
1335
+ @staticmethod
1336
+ def _audio_paths_for(job_ids: Sequence[str], conn: sqlite3.Connection) -> tuple[str, ...]:
1337
+ if not job_ids:
1338
+ return ()
1339
+ marks = ",".join("?" * len(job_ids))
1340
+ paths = [
1341
+ r[0]
1342
+ for r in conn.execute(
1343
+ f"SELECT relative_path FROM results WHERE job_id IN ({marks})", tuple(job_ids)
1344
+ )
1345
+ if r[0]
1346
+ ]
1347
+ paths.extend(
1348
+ r[0]
1349
+ for r in conn.execute(
1350
+ f"SELECT audio_path FROM segments WHERE job_id IN ({marks})"
1351
+ " AND audio_path IS NOT NULL",
1352
+ tuple(job_ids),
1353
+ )
1354
+ )
1355
+ return tuple(paths)
1356
+
1357
+ # ==================================================================
1358
+ # Segments
1359
+ # ==================================================================
1360
+
1361
+ @_guard
1362
+ def insert_segments(self, job_id: str, segments: Sequence[Segment]) -> tuple[Segment, ...]:
1363
+ """Write a job's segment table in one go.
1364
+
1365
+ Offsets go into columns named for what 4.2 standardises them as:
1366
+ code points, start inclusive, end exclusive. The schema enforces the
1367
+ ordering, so a UTF-16 index borrowed from Qt cannot quietly survive
1368
+ here even if it arrives.
1369
+ """
1370
+ stored: list[Segment] = []
1371
+ with self.transaction() as conn:
1372
+ for position, seg in enumerate(segments):
1373
+ index = seg.index if seg.index is not None else position
1374
+ segment_id = seg.segment_id or ids.segment_id()
1375
+ conn.execute(
1376
+ "INSERT INTO segments (segment_id, job_id, seq,"
1377
+ " source_start_codepoint_inclusive, source_end_codepoint_exclusive,"
1378
+ " spoken_text, language, audio_start_ms, audio_end_ms,"
1379
+ " trailing_silence_ms, audio_path, frame_count, ready)"
1380
+ " VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
1381
+ (
1382
+ segment_id,
1383
+ job_id,
1384
+ index,
1385
+ seg.source.start,
1386
+ seg.source.end,
1387
+ seg.spoken_text,
1388
+ seg.language,
1389
+ None if seg.time is None else seg.time.start_ms,
1390
+ None if seg.time is None else seg.time.end_ms,
1391
+ seg.trailing_silence_ms,
1392
+ seg.audio_path,
1393
+ seg.frame_count,
1394
+ int(seg.ready),
1395
+ ),
1396
+ )
1397
+ stored.append(
1398
+ Segment(
1399
+ index=index,
1400
+ source=seg.source,
1401
+ spoken_text=seg.spoken_text,
1402
+ language=seg.language,
1403
+ time=seg.time,
1404
+ trailing_silence_ms=seg.trailing_silence_ms,
1405
+ audio_path=seg.audio_path,
1406
+ frame_count=seg.frame_count,
1407
+ ready=seg.ready,
1408
+ segment_id=segment_id,
1409
+ )
1410
+ )
1411
+ self._refresh_segment_counts(job_id, conn)
1412
+ return tuple(stored)
1413
+
1414
+ @_guard
1415
+ def mark_segment_ready(
1416
+ self,
1417
+ job_id: str,
1418
+ index: int,
1419
+ *,
1420
+ time: TimeRange,
1421
+ audio_path: str | None = None,
1422
+ frame_count: int = 0,
1423
+ trailing_silence_ms: int | None = None,
1424
+ ) -> None:
1425
+ """A segment becomes playable (F-55, F-12).
1426
+
1427
+ The job's generated count is recomputed from the segment rows rather
1428
+ than incremented, so replaying the same completion -- which a
1429
+ reconnecting worker can do -- cannot inflate the progress F-11 shows.
1430
+ """
1431
+ with self.transaction() as conn:
1432
+ cur = conn.execute(
1433
+ "UPDATE segments SET ready = 1, audio_start_ms = ?, audio_end_ms = ?,"
1434
+ " audio_path = ?, frame_count = ?,"
1435
+ " trailing_silence_ms = CASE WHEN ? IS NULL THEN trailing_silence_ms ELSE ? END"
1436
+ " WHERE job_id = ? AND seq = ?",
1437
+ (
1438
+ time.start_ms,
1439
+ time.end_ms,
1440
+ audio_path,
1441
+ frame_count,
1442
+ trailing_silence_ms,
1443
+ trailing_silence_ms,
1444
+ job_id,
1445
+ index,
1446
+ ),
1447
+ )
1448
+ if cur.rowcount == 0:
1449
+ raise EchoActError(
1450
+ Code.NOT_FOUND, "No such segment.", detail={"job_id": job_id, "index": index}
1451
+ )
1452
+ self._refresh_segment_counts(job_id, conn)
1453
+
1454
+ @staticmethod
1455
+ def _refresh_segment_counts(job_id: str, conn: sqlite3.Connection) -> None:
1456
+ conn.execute(
1457
+ "UPDATE jobs SET"
1458
+ " total_segments = (SELECT count(*) FROM segments WHERE job_id = ?),"
1459
+ " generated_segments = (SELECT count(*) FROM segments WHERE job_id = ? AND ready = 1)"
1460
+ " WHERE job_id = ?",
1461
+ (job_id, job_id, job_id),
1462
+ )
1463
+
1464
+ @_guard
1465
+ def list_segments(self, job_id: str, *, ready_only: bool = False) -> list[Segment]:
1466
+ """F-55's segment list: order, source range, times, readiness. No
1467
+ audio is read -- an unready segment has no audio to read anyway, and
1468
+ F-55 forbids returning one as if it were finished."""
1469
+ columns = ", ".join(_SEGMENT_COLUMNS)
1470
+ clause = " AND ready = 1" if ready_only else ""
1471
+ rows = self._conn().execute(
1472
+ f"SELECT {columns} FROM segments WHERE job_id = ?{clause} ORDER BY seq", (job_id,)
1473
+ ).fetchall()
1474
+ return [_to_segment(r) for r in rows]
1475
+
1476
+ # ==================================================================
1477
+ # Results
1478
+ # ==================================================================
1479
+
1480
+ @_guard
1481
+ def get_result(self, result_id: str) -> Result:
1482
+ columns = ", ".join(_RESULT_COLUMNS)
1483
+ row = self._conn().execute(
1484
+ f"SELECT {columns} FROM results WHERE result_id = ?", (result_id,)
1485
+ ).fetchone()
1486
+ if row is None:
1487
+ raise EchoActError(Code.NOT_FOUND, "No such result.", detail={"id": result_id})
1488
+ return _to_result(row)
1489
+
1490
+ @_guard
1491
+ def get_result_for_job(self, job_id: str) -> Result | None:
1492
+ columns = ", ".join(_RESULT_COLUMNS)
1493
+ row = self._conn().execute(
1494
+ f"SELECT {columns} FROM results WHERE job_id = ?", (job_id,)
1495
+ ).fetchone()
1496
+ return None if row is None else _to_result(row)
1497
+
1498
+ @_guard
1499
+ def result_integrity(self, result_id: str) -> ResultIntegrity:
1500
+ row = self._conn().execute(
1501
+ "SELECT integrity_state FROM results WHERE result_id = ?", (result_id,)
1502
+ ).fetchone()
1503
+ if row is None:
1504
+ raise EchoActError(Code.NOT_FOUND, "No such result.", detail={"id": result_id})
1505
+ return ResultIntegrity(row["integrity_state"])
1506
+
1507
+ @_guard
1508
+ def expire_result(self, result_id: str, *, at: float | None = None) -> None:
1509
+ """4.1's one-off lifetime, applied.
1510
+
1511
+ The row survives its own expiry: 4.2 requires a repeated re-request
1512
+ key to return the existing job *and* the result's expired state, and
1513
+ a deleted row could only be reported as "never existed".
1514
+ """
1515
+ moment = ids.now() if at is None else at
1516
+ with self.transaction() as conn:
1517
+ cur = conn.execute(
1518
+ "UPDATE results SET expires_at = ? WHERE result_id = ?"
1519
+ " AND (expires_at IS NULL OR expires_at > ?)",
1520
+ (moment, result_id, moment),
1521
+ )
1522
+ if cur.rowcount == 0 and self._count(conn, "results", "result_id", result_id) == 0:
1523
+ raise EchoActError(Code.NOT_FOUND, "No such result.", detail={"id": result_id})
1524
+
1525
+ @_guard
1526
+ def due_results(self, *, at: float | None = None) -> tuple[Result, ...]:
1527
+ """Results whose lifetime has passed, for the caller to clean up.
1528
+
1529
+ Returned rather than deleted here: the files are the caller's to
1530
+ unlink, and F-73 excludes anything currently playing or generating
1531
+ from cleanup, which is knowledge this layer does not have.
1532
+ """
1533
+ moment = ids.now() if at is None else at
1534
+ columns = ", ".join(_RESULT_COLUMNS)
1535
+ rows = self._conn().execute(
1536
+ f"SELECT {columns} FROM results WHERE expires_at IS NOT NULL AND expires_at <= ?"
1537
+ " ORDER BY expires_at",
1538
+ (moment,),
1539
+ ).fetchall()
1540
+ return tuple(_to_result(r) for r in rows)
1541
+
1542
+ @_guard
1543
+ def delete_result(self, result_id: str) -> Deletion:
1544
+ """Remove a result row, leaving its file for the caller to unlink."""
1545
+ with self.transaction() as conn:
1546
+ row = conn.execute(
1547
+ "SELECT relative_path FROM results WHERE result_id = ?", (result_id,)
1548
+ ).fetchone()
1549
+ if row is None:
1550
+ raise EchoActError(Code.NOT_FOUND, "No such result.", detail={"id": result_id})
1551
+ conn.execute("DELETE FROM results WHERE result_id = ?", (result_id,))
1552
+ path = row["relative_path"]
1553
+ return Deletion(documents=0, jobs=0, results=1, audio_paths=(path,) if path else ())
1554
+
1555
+ def _resolve_audio(self, relative_path: str) -> Path | None:
1556
+ """Resolve a stored relative path inside the audio directory.
1557
+
1558
+ Anything that climbs out of the root resolves to ``None`` and is
1559
+ treated as a missing file. N-18 and N-27 both forbid a stored path
1560
+ from reaching outside the app's own tree, and a restored or tampered
1561
+ row is exactly where such a path would come from.
1562
+ """
1563
+ if not relative_path:
1564
+ return None
1565
+ root = self.audio_root
1566
+ try:
1567
+ candidate = (root / relative_path).resolve()
1568
+ candidate.relative_to(root.resolve())
1569
+ except (OSError, ValueError):
1570
+ return None
1571
+ return candidate
1572
+
1573
+ @_guard
1574
+ def verify_result(self, result_id: str, *, deep: bool = True, at: float | None = None) -> ResultIntegrity:
1575
+ """F-45's detection, for one result.
1576
+
1577
+ ``deep`` recomputes the SHA-256 in 4.2's integrity field. Shallow
1578
+ verification compares the recorded byte size, which is what catches
1579
+ the failure an abnormal termination actually produces -- a truncated
1580
+ file -- without reading gigabytes at start-up.
1581
+ """
1582
+ moment = ids.now() if at is None else at
1583
+ result = self.get_result(result_id)
1584
+ path = self._resolve_audio(result.relative_path)
1585
+ state = ResultIntegrity.OK
1586
+ if path is None or not path.exists():
1587
+ state = ResultIntegrity.MISSING
1588
+ else:
1589
+ try:
1590
+ if path.stat().st_size != result.byte_size:
1591
+ state = ResultIntegrity.CORRUPT
1592
+ elif deep:
1593
+ with path.open("rb") as fh:
1594
+ actual = hashlib.file_digest(fh, "sha256").hexdigest()
1595
+ if actual != result.digest:
1596
+ state = ResultIntegrity.CORRUPT
1597
+ except OSError:
1598
+ state = ResultIntegrity.MISSING
1599
+ with self.transaction() as conn:
1600
+ conn.execute(
1601
+ "UPDATE results SET integrity_state = ?, verified_at = ? WHERE result_id = ?",
1602
+ (state.value, moment, result_id),
1603
+ )
1604
+ return state
1605
+
1606
+ # ==================================================================
1607
+ # Re-request records (F-49, 4.1, 4.2)
1608
+ # ==================================================================
1609
+
1610
+ @_guard
1611
+ def lookup_idempotency(
1612
+ self, client_id: str, key: str, *, at: float | None = None
1613
+ ) -> IdempotencyRecord | None:
1614
+ """4.2's re-request lookup, scoped per client.
1615
+
1616
+ An expired entry reads as absent: 4.1 discloses that reusing a key
1617
+ after expiry may become a new request, and pretending otherwise would
1618
+ return a job the caller can no longer reason about.
1619
+ """
1620
+ moment = ids.now() if at is None else at
1621
+ row = self._conn().execute(
1622
+ "SELECT client_id, key, request_digest, job_id, created_at, expires_at"
1623
+ " FROM idempotency WHERE client_id = ? AND key = ? AND expires_at > ?",
1624
+ (client_id, key, moment),
1625
+ ).fetchone()
1626
+ return None if row is None else _to_idempotency(row)
1627
+
1628
+ @_guard
1629
+ def remember_idempotency(
1630
+ self,
1631
+ client_id: str,
1632
+ key: str,
1633
+ *,
1634
+ request_digest: str,
1635
+ job_id: str,
1636
+ at: float | None = None,
1637
+ ttl_s: float = IDEMPOTENCY_TTL_S,
1638
+ ) -> IdempotencyRecord:
1639
+ """Store the discriminator, the job, and an expiry -- and no text.
1640
+
1641
+ Kept in the database rather than in memory because 4.1 requires
1642
+ duplicate prevention to survive a normal restart for the same hour.
1643
+
1644
+ A live entry for the same key is never overwritten in silence: the
1645
+ same content returns the entry that already exists, and different
1646
+ content is F-49's conflict. Only an expired entry is replaced, which
1647
+ is 4.1's disclosed "reusing the same key after expiry may become a
1648
+ new request".
1649
+ """
1650
+ moment = ids.now() if at is None else at
1651
+ record = IdempotencyRecord(
1652
+ client_id=client_id,
1653
+ key=key,
1654
+ request_digest=request_digest,
1655
+ job_id=job_id,
1656
+ created_at=moment,
1657
+ expires_at=moment + ttl_s,
1658
+ )
1659
+ with self.transaction() as conn:
1660
+ existing = self.lookup_idempotency(client_id, key, at=moment)
1661
+ if existing is not None:
1662
+ if existing.request_digest != request_digest or existing.job_id != job_id:
1663
+ raise EchoActError(
1664
+ Code.IDEMPOTENCY_KEY_CONFLICT, detail={"job_id": existing.job_id}
1665
+ )
1666
+ return existing
1667
+ conn.execute(
1668
+ "INSERT INTO idempotency (client_id, key, request_digest, job_id, created_at,"
1669
+ " expires_at) VALUES (?, ?, ?, ?, ?, ?)"
1670
+ " ON CONFLICT (client_id, key) DO UPDATE SET"
1671
+ " request_digest = excluded.request_digest, job_id = excluded.job_id,"
1672
+ " created_at = excluded.created_at, expires_at = excluded.expires_at",
1673
+ (client_id, key, request_digest, job_id, moment, record.expires_at),
1674
+ )
1675
+ return record
1676
+
1677
+ @_guard
1678
+ def claim_job(
1679
+ self,
1680
+ job: Job,
1681
+ *,
1682
+ client_id: str,
1683
+ key: str,
1684
+ request_digest: str,
1685
+ at: float | None = None,
1686
+ ttl_s: float = IDEMPOTENCY_TTL_S,
1687
+ ) -> tuple[Job, bool]:
1688
+ """F-49's gate, as one atomic step.
1689
+
1690
+ Returns ``(job, created)``. The same key with the same content
1691
+ returns the existing job; different content is
1692
+ ``IDEMPOTENCY_KEY_CONFLICT``. Lookup and insert share one
1693
+ ``BEGIN IMMEDIATE`` transaction on purpose: with a single generation
1694
+ slot, two racing retries of the same request are the ordinary case,
1695
+ not the exotic one, and checking then inserting without a lock would
1696
+ let both through.
1697
+ """
1698
+ moment = ids.now() if at is None else at
1699
+ with self.transaction():
1700
+ existing = self.lookup_idempotency(client_id, key, at=moment)
1701
+ if existing is not None:
1702
+ if existing.request_digest != request_digest:
1703
+ raise EchoActError(
1704
+ Code.IDEMPOTENCY_KEY_CONFLICT,
1705
+ detail={"job_id": existing.job_id},
1706
+ )
1707
+ return self.get_job(existing.job_id), False
1708
+ self.create_job(job, at=moment)
1709
+ self.remember_idempotency(
1710
+ client_id,
1711
+ key,
1712
+ request_digest=request_digest,
1713
+ job_id=job.job_id,
1714
+ at=moment,
1715
+ ttl_s=ttl_s,
1716
+ )
1717
+ return job, True
1718
+
1719
+ @_guard
1720
+ def purge_expired_idempotency(self, *, at: float | None = None) -> int:
1721
+ moment = ids.now() if at is None else at
1722
+ with self.transaction() as conn:
1723
+ cur = conn.execute("DELETE FROM idempotency WHERE expires_at <= ?", (moment,))
1724
+ return cur.rowcount
1725
+
1726
+ # ==================================================================
1727
+ # Retention accounting (N-16, 4.1)
1728
+ # ==================================================================
1729
+
1730
+ @property
1731
+ def retention_limit_bytes(self) -> int:
1732
+ return self._retention_limit_bytes
1733
+
1734
+ def set_retention_limit(self, limit_bytes: int) -> None:
1735
+ """4.1: lowering the limit below what is already stored keeps the
1736
+ data and blocks new retention. Nothing is deleted here, ever."""
1737
+ self._retention_limit_bytes = int(limit_bytes)
1738
+
1739
+ @_guard
1740
+ def storage_usage(self, *, at: float | None = None) -> StorageUsage:
1741
+ """N-16's display: what the app manages against the ceiling.
1742
+
1743
+ Audio counts every result whose lifetime has not passed, one-off
1744
+ included, because the space is genuinely occupied while it lives.
1745
+ Source text counts only retained jobs: a one-off snapshot is
1746
+ temporary by definition and F-42 keeps it out of permanent history.
1747
+ """
1748
+ moment = ids.now() if at is None else at
1749
+ row = self._conn().execute(
1750
+ "SELECT"
1751
+ " (SELECT coalesce(sum(length(CAST(title AS BLOB))"
1752
+ " + length(CAST(body AS BLOB))), 0) FROM documents),"
1753
+ " (SELECT coalesce(sum(length(CAST(source_text AS BLOB))), 0) FROM jobs"
1754
+ " WHERE retention = ? AND source_text IS NOT NULL),"
1755
+ " (SELECT coalesce(sum(byte_size), 0) FROM results"
1756
+ " WHERE expires_at IS NULL OR expires_at > ?)",
1757
+ (RetentionMode.RETAINED.value, moment),
1758
+ ).fetchone()
1759
+ return StorageUsage(
1760
+ document_bytes=int(row[0]),
1761
+ job_text_bytes=int(row[1]),
1762
+ audio_bytes=int(row[2]),
1763
+ limit_bytes=self._retention_limit_bytes,
1764
+ )
1765
+
1766
+ def retention_fits(self, additional_bytes: int) -> bool:
1767
+ return self.storage_usage().fits(additional_bytes)
1768
+
1769
+ def require_retention_capacity(self, additional_bytes: int) -> None:
1770
+ """N-16, stated as code: the request fails, the data stays.
1771
+
1772
+ There is deliberately no eviction path here. "Explicitly saved
1773
+ documents and retained results are never silently deleted to free
1774
+ space" leaves exactly one behaviour when the ceiling is reached, and
1775
+ 4.1 confirms it: the retention request is reported as a failure.
1776
+ """
1777
+ if additional_bytes <= 0:
1778
+ return
1779
+ usage = self.storage_usage()
1780
+ if not usage.fits(additional_bytes):
1781
+ raise EchoActError(
1782
+ Code.RETENTION_LIMIT_REACHED,
1783
+ detail={
1784
+ "used_bytes": usage.total_bytes,
1785
+ "limit_bytes": usage.limit_bytes,
1786
+ "requested_bytes": additional_bytes,
1787
+ },
1788
+ )
1789
+
1790
+ # ==================================================================
1791
+ # Start-up reconciliation (F-45)
1792
+ # ==================================================================
1793
+
1794
+ @_guard
1795
+ def reconcile_on_start(self, *, at: float | None = None, deep: bool = False) -> ReconcileReport:
1796
+ """F-45. Call once, before anything else uses the store.
1797
+
1798
+ Three findings, and no action beyond recording them -- F-45 forbids
1799
+ regenerating or playing anything automatically, so nothing here
1800
+ starts a job, and the GUI decides what to offer.
1801
+
1802
+ * A job left non-terminal by an abnormal termination becomes
1803
+ Interrupted. It never becomes Complete: the audio it would claim
1804
+ to have does not exist. A job caught in Canceling becomes
1805
+ Canceled, because Section 5.1's transition table has no
1806
+ Canceling -> Interrupted edge and the user's cancellation is the
1807
+ outcome that actually happened.
1808
+ * A one-off result from a previous run is expired outright. 4.1
1809
+ gives it "one hour after a terminal state, or app exit, whichever
1810
+ comes first", and app exit has demonstrably come first.
1811
+ * A retained result whose file is gone or the wrong size is marked,
1812
+ so the GUI can say why playback is unavailable and offer delete and
1813
+ regenerate.
1814
+ """
1815
+ moment = ids.now() if at is None else at
1816
+ interrupted: list[str] = []
1817
+ canceled: list[str] = []
1818
+
1819
+ with self.transaction() as conn:
1820
+ revivable = tuple(
1821
+ v for v in _ACTIVE_STATE_VALUES if v != JobState.CANCELING.value
1822
+ )
1823
+ marks = ",".join("?" * len(revivable))
1824
+ interrupted = [
1825
+ r["job_id"]
1826
+ for r in conn.execute(
1827
+ f"SELECT job_id FROM jobs WHERE state IN ({marks})", revivable
1828
+ ).fetchall()
1829
+ ]
1830
+ canceled = [
1831
+ r["job_id"]
1832
+ for r in conn.execute(
1833
+ "SELECT job_id FROM jobs WHERE state = ?", (JobState.CANCELING.value,)
1834
+ ).fetchall()
1835
+ ]
1836
+ if interrupted:
1837
+ conn.execute(
1838
+ f"UPDATE jobs SET state = ?, ended_at = coalesce(ended_at, ?)"
1839
+ f" WHERE state IN ({marks})",
1840
+ (JobState.INTERRUPTED.value, moment, *revivable),
1841
+ )
1842
+ if canceled:
1843
+ conn.execute(
1844
+ "UPDATE jobs SET state = ?, ended_at = coalesce(ended_at, ?) WHERE state = ?",
1845
+ (JobState.CANCELED.value, moment, JobState.CANCELING.value),
1846
+ )
1847
+
1848
+ with self.transaction() as conn:
1849
+ expired = [
1850
+ r["result_id"]
1851
+ for r in conn.execute(
1852
+ "SELECT r.result_id FROM results r JOIN jobs j ON j.job_id = r.job_id"
1853
+ " WHERE j.retention = ? AND (r.expires_at IS NULL OR r.expires_at > ?)",
1854
+ (RetentionMode.ONE_OFF.value, moment),
1855
+ ).fetchall()
1856
+ ]
1857
+ if expired:
1858
+ marks = ",".join("?" * len(expired))
1859
+ conn.execute(
1860
+ f"UPDATE results SET expires_at = ? WHERE result_id IN ({marks})",
1861
+ (moment, *expired),
1862
+ )
1863
+
1864
+ retained = [
1865
+ r["result_id"]
1866
+ for r in self._conn().execute(
1867
+ "SELECT r.result_id FROM results r JOIN jobs j ON j.job_id = r.job_id"
1868
+ " WHERE j.retention = ?",
1869
+ (RetentionMode.RETAINED.value,),
1870
+ ).fetchall()
1871
+ ]
1872
+ missing: list[str] = []
1873
+ corrupt: list[str] = []
1874
+ for result_id in retained:
1875
+ state = self.verify_result(result_id, deep=deep, at=moment)
1876
+ if state is ResultIntegrity.MISSING:
1877
+ missing.append(result_id)
1878
+ elif state is ResultIntegrity.CORRUPT:
1879
+ corrupt.append(result_id)
1880
+
1881
+ return ReconcileReport(
1882
+ interrupted_job_ids=tuple(interrupted),
1883
+ canceled_job_ids=tuple(canceled),
1884
+ expired_result_ids=tuple(expired),
1885
+ missing_result_ids=tuple(missing),
1886
+ corrupt_result_ids=tuple(corrupt),
1887
+ )
1888
+
1889
+ # ==================================================================
1890
+ # Integration permissions (4.2, F-61, F-71)
1891
+ # ==================================================================
1892
+
1893
+ @_guard
1894
+ def upsert_client(
1895
+ self,
1896
+ client_id: str,
1897
+ label: str,
1898
+ capabilities: Iterable[Capability],
1899
+ *,
1900
+ active: bool = True,
1901
+ at: float | None = None,
1902
+ ) -> ClientRecord:
1903
+ moment = ids.now() if at is None else at
1904
+ caps = frozenset(Capability(c) for c in capabilities)
1905
+ payload = json.dumps(sorted(c.value for c in caps))
1906
+ with self.transaction() as conn:
1907
+ conn.execute(
1908
+ "INSERT INTO clients (client_id, label, capabilities, active, created_at)"
1909
+ " VALUES (?, ?, ?, ?, ?)"
1910
+ " ON CONFLICT (client_id) DO UPDATE SET label = excluded.label,"
1911
+ " capabilities = excluded.capabilities, active = excluded.active,"
1912
+ " revoked_at = CASE WHEN excluded.active = 1 THEN NULL ELSE clients.revoked_at END",
1913
+ (client_id, label, payload, int(active), moment),
1914
+ )
1915
+ return ClientRecord(
1916
+ client_id=client_id,
1917
+ label=label,
1918
+ capabilities=caps,
1919
+ active=active,
1920
+ created_at=moment,
1921
+ )
1922
+
1923
+ @_guard
1924
+ def get_client(self, client_id: str) -> ClientRecord | None:
1925
+ row = self._conn().execute(
1926
+ "SELECT * FROM clients WHERE client_id = ?", (client_id,)
1927
+ ).fetchone()
1928
+ return None if row is None else _to_client(row)
1929
+
1930
+ @_guard
1931
+ def list_clients(self) -> tuple[ClientRecord, ...]:
1932
+ rows = self._conn().execute("SELECT * FROM clients ORDER BY created_at").fetchall()
1933
+ return tuple(_to_client(r) for r in rows)
1934
+
1935
+ @_guard
1936
+ def revoke_client(self, client_id: str, *, at: float | None = None) -> None:
1937
+ """F-61's revocation, recorded rather than deleted.
1938
+
1939
+ The row stays so F-71 can still show what was granted and when it was
1940
+ withdrawn; deleting it would make an audit of a revoked client
1941
+ impossible. Cancelling that client's in-flight jobs is 5.3's
1942
+ separate obligation and belongs to the job engine.
1943
+ """
1944
+ moment = ids.now() if at is None else at
1945
+ with self.transaction() as conn:
1946
+ cur = conn.execute(
1947
+ "UPDATE clients SET active = 0, revoked_at = ? WHERE client_id = ?",
1948
+ (moment, client_id),
1949
+ )
1950
+ if cur.rowcount == 0:
1951
+ raise EchoActError(Code.NOT_FOUND, "No such client.", detail={"id": client_id})
1952
+
1953
+ @_guard
1954
+ def touch_client(self, client_id: str, *, at: float | None = None) -> None:
1955
+ """F-71 shows a client's last access time."""
1956
+ moment = ids.now() if at is None else at
1957
+ with self.transaction() as conn:
1958
+ conn.execute(
1959
+ "UPDATE clients SET last_seen_at = ? WHERE client_id = ?", (moment, client_id)
1960
+ )
1961
+
1962
+ # ==================================================================
1963
+ # Backup records (N-15, F-44 bookkeeping only)
1964
+ # ==================================================================
1965
+
1966
+ @_guard
1967
+ def list_backups(self, *, kind: str | None = None) -> tuple[BackupRecord, ...]:
1968
+ conn = self._conn()
1969
+ if kind is None:
1970
+ rows = conn.execute("SELECT * FROM backups ORDER BY created_at DESC").fetchall()
1971
+ else:
1972
+ rows = conn.execute(
1973
+ "SELECT * FROM backups WHERE kind = ? ORDER BY created_at DESC", (kind,)
1974
+ ).fetchall()
1975
+ return tuple(_to_backup(r) for r in rows)
1976
+
1977
+ @_guard
1978
+ def record_backup(
1979
+ self,
1980
+ *,
1981
+ location: str,
1982
+ kind: str = "manual",
1983
+ byte_size: int = 0,
1984
+ item_count: int = 0,
1985
+ verified: bool = False,
1986
+ note: str | None = None,
1987
+ app_version: str | None = None,
1988
+ at: float | None = None,
1989
+ ) -> BackupRecord:
1990
+ """Bookkeeping for a bundle another module produced.
1991
+
1992
+ The store records that a copy exists and at which schema version;
1993
+ producing, verifying, and restoring the bundle are F-44's and N-27's
1994
+ and live outside this package.
1995
+ """
1996
+ moment = ids.now() if at is None else at
1997
+ record = BackupRecord(
1998
+ backup_id=ids.backup_id(),
1999
+ kind=kind,
2000
+ created_at=moment,
2001
+ location=location,
2002
+ byte_size=byte_size,
2003
+ item_count=item_count,
2004
+ schema_version=self._schema_version,
2005
+ app_version=app_version or __version__,
2006
+ verified=verified,
2007
+ note=note,
2008
+ )
2009
+ with self.transaction() as conn:
2010
+ conn.execute(
2011
+ "INSERT INTO backups (backup_id, kind, created_at, location, byte_size,"
2012
+ " item_count, schema_version, app_version, verified, note)"
2013
+ " VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
2014
+ (
2015
+ record.backup_id,
2016
+ record.kind,
2017
+ record.created_at,
2018
+ record.location,
2019
+ record.byte_size,
2020
+ record.item_count,
2021
+ record.schema_version,
2022
+ record.app_version,
2023
+ int(record.verified),
2024
+ record.note,
2025
+ ),
2026
+ )
2027
+ return record
2028
+
2029
+ @_guard
2030
+ def delete_backup_record(self, backup_id: str) -> None:
2031
+ with self.transaction() as conn:
2032
+ cur = conn.execute("DELETE FROM backups WHERE backup_id = ?", (backup_id,))
2033
+ if cur.rowcount == 0:
2034
+ raise EchoActError(Code.NOT_FOUND, "No such backup.", detail={"id": backup_id})
2035
+
2036
+ # -- helpers --------------------------------------------------------
2037
+
2038
+ @staticmethod
2039
+ def _count(conn: sqlite3.Connection, table: str, column: str, value: Any) -> int:
2040
+ return int(
2041
+ conn.execute(f"SELECT count(*) FROM {table} WHERE {column} = ?", (value,)).fetchone()[0]
2042
+ )
2043
+
2044
+
2045
+ __all__ = [
2046
+ "DEFAULT_BUSY_TIMEOUT_S",
2047
+ "BackupRecord",
2048
+ "ClientRecord",
2049
+ "Deletion",
2050
+ "DeletionScope",
2051
+ "DocumentSummary",
2052
+ "IdempotencyRecord",
2053
+ "JobSummary",
2054
+ "Page",
2055
+ "ReconcileReport",
2056
+ "ResultIntegrity",
2057
+ "StorageUsage",
2058
+ "Store",
2059
+ "request_match_digest",
2060
+ "text_bytes",
2061
+ "translate_sqlite_error",
2062
+ ]