echoact 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- echoact/__init__.py +3 -0
- echoact/__main__.py +117 -0
- echoact/app.py +315 -0
- echoact/audio/__init__.py +0 -0
- echoact/audio/devices.py +192 -0
- echoact/audio/player.py +611 -0
- echoact/audio/wav.py +854 -0
- echoact/config/__init__.py +0 -0
- echoact/config/budget.py +370 -0
- echoact/config/settings.py +1244 -0
- echoact/db/__init__.py +0 -0
- echoact/db/backup.py +2429 -0
- echoact/db/migrations.py +434 -0
- echoact/db/schema.sql +214 -0
- echoact/db/store.py +2062 -0
- echoact/diagnostics.py +902 -0
- echoact/domain.py +487 -0
- echoact/engine/__init__.py +0 -0
- echoact/engine/container.py +843 -0
- echoact/engine/protocol.py +241 -0
- echoact/engine/runtime.py +324 -0
- echoact/engine/supervisor.py +961 -0
- echoact/engine/worker.py +659 -0
- echoact/errors.py +281 -0
- echoact/instance.py +172 -0
- echoact/jobs/__init__.py +0 -0
- echoact/jobs/engine.py +776 -0
- echoact/jobs/request.py +300 -0
- echoact/mcp/__init__.py +0 -0
- echoact/mcp/__main__.py +50 -0
- echoact/mcp/client.py +202 -0
- echoact/mcp/config.py +112 -0
- echoact/mcp/server.py +340 -0
- echoact/models/__init__.py +0 -0
- echoact/models/catalog.py +273 -0
- echoact/models/manifest.py +278 -0
- echoact/models/registry.py +1551 -0
- echoact/paths.py +93 -0
- echoact/policy.py +189 -0
- echoact/security/__init__.py +0 -0
- echoact/security/credentials.py +930 -0
- echoact/security/ratelimit.py +534 -0
- echoact/service/__init__.py +20 -0
- echoact/service/app.py +182 -0
- echoact/service/deps.py +563 -0
- echoact/service/errors.py +241 -0
- echoact/service/routes.py +1125 -0
- echoact/service/schemas.py +509 -0
- echoact/service/server.py +270 -0
- echoact/text/__init__.py +0 -0
- echoact/text/language.py +44 -0
- echoact/text/loader.py +577 -0
- echoact/text/normalize.py +924 -0
- echoact/text/segment.py +499 -0
- echoact/text/sniff.py +1202 -0
- echoact/ui/__init__.py +0 -0
- echoact/ui/bridge.py +50 -0
- echoact/ui/controls.py +360 -0
- echoact/ui/credential_dialog.py +131 -0
- echoact/ui/fonts.py +94 -0
- echoact/ui/i18n.py +260 -0
- echoact/ui/icons.py +440 -0
- echoact/ui/library.py +1642 -0
- echoact/ui/licence.py +162 -0
- echoact/ui/main_window.py +1202 -0
- echoact/ui/mcp_setup.py +494 -0
- echoact/ui/models_view.py +1142 -0
- echoact/ui/notifications.py +202 -0
- echoact/ui/reading.py +494 -0
- echoact/ui/settings_view.py +2258 -0
- echoact/ui/status_view.py +1193 -0
- echoact/ui/theme.py +579 -0
- echoact/util/__init__.py +0 -0
- echoact/util/ids.py +62 -0
- echoact/util/logging.py +127 -0
- echoact-0.1.0.dist-info/METADATA +162 -0
- echoact-0.1.0.dist-info/RECORD +80 -0
- echoact-0.1.0.dist-info/WHEEL +4 -0
- echoact-0.1.0.dist-info/entry_points.txt +3 -0
- echoact-0.1.0.dist-info/licenses/LICENSE +21 -0
echoact/db/backup.py
ADDED
|
@@ -0,0 +1,2429 @@
|
|
|
1
|
+
"""Backup and restore: F-44, N-15, N-27, F-74, and 4.1's restore limits.
|
|
2
|
+
|
|
3
|
+
A backup is one file the user chooses the location of. Inside it is a ZIP
|
|
4
|
+
holding a manifest, the library's rows as JSON lines, and the full-result WAV
|
|
5
|
+
files the user asked for. What is *not* inside it is the point of F-44 and
|
|
6
|
+
4.2: no model weights, no credentials, no client permissions. That exclusion
|
|
7
|
+
is a property of the writer rather than a note in a document -- this module
|
|
8
|
+
never walks the data directory, it emits exactly the members it names, and it
|
|
9
|
+
refuses outright if a stored audio path resolves anywhere but inside the audio
|
|
10
|
+
directory.
|
|
11
|
+
|
|
12
|
+
Three shapes of the problem, and the decision each one forced:
|
|
13
|
+
|
|
14
|
+
* **Consistency (N-27, 5.3).** The bundle is read through one deferred read
|
|
15
|
+
transaction held open for its whole life, so a document edited half way
|
|
16
|
+
through contributes the state it had when the backup started. A file copy
|
|
17
|
+
of a live write-ahead-log database would miss whatever is still in the log,
|
|
18
|
+
and re-reading per table would splice two points in time together.
|
|
19
|
+
* **Restore is the dangerous direction (N-27).** Everything about an archive
|
|
20
|
+
is hostile input: member names, declared sizes, item counts, digests. Every
|
|
21
|
+
one of them is checked *before* anything is applied, and a single bad member
|
|
22
|
+
rejects the whole archive rather than being skipped -- a partially applied
|
|
23
|
+
restore is exactly the outcome N-15 forbids. As a second layer, an archive
|
|
24
|
+
member name is never used as a destination path; restored audio is written
|
|
25
|
+
to a name this module mints.
|
|
26
|
+
* **Identity (F-44).** A restore adds items. Every row gets a freshly minted
|
|
27
|
+
identifier and the original is reported back in :class:`RestoreOutcome`, so
|
|
28
|
+
"which of these came from the backup, and what was it called before?" has an
|
|
29
|
+
answer. Reusing the original identifiers would either collide with live data
|
|
30
|
+
or silently overwrite it, and F-44 asks for neither.
|
|
31
|
+
|
|
32
|
+
The cancel token and the progress callback are the same shape
|
|
33
|
+
``echoact.models.registry`` uses for a download, and :class:`CancelToken` is
|
|
34
|
+
literally that class: a GUI that already owns one for model preparation should
|
|
35
|
+
not have to learn a second cancellation vocabulary.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
from __future__ import annotations
|
|
39
|
+
|
|
40
|
+
import hashlib
|
|
41
|
+
import json
|
|
42
|
+
import os
|
|
43
|
+
import re
|
|
44
|
+
import shutil
|
|
45
|
+
import sqlite3
|
|
46
|
+
import stat
|
|
47
|
+
import threading
|
|
48
|
+
import time
|
|
49
|
+
import zipfile
|
|
50
|
+
from collections.abc import Callable, Iterable, Iterator, Sequence
|
|
51
|
+
from contextlib import contextmanager, suppress
|
|
52
|
+
from dataclasses import dataclass, field, replace
|
|
53
|
+
from enum import StrEnum
|
|
54
|
+
from pathlib import Path, PurePosixPath
|
|
55
|
+
from typing import Any, Final
|
|
56
|
+
|
|
57
|
+
from .. import __version__
|
|
58
|
+
from ..domain import JobState, RetentionMode
|
|
59
|
+
from ..errors import Code, EchoActError
|
|
60
|
+
from ..models.registry import CancelToken
|
|
61
|
+
from ..paths import redact
|
|
62
|
+
from ..policy import (
|
|
63
|
+
BACKUP_KEEP_SCHEDULED,
|
|
64
|
+
BACKUP_RESTORE_MAX_BYTES,
|
|
65
|
+
BACKUP_RESTORE_MAX_ITEMS,
|
|
66
|
+
LOW_SPACE_WARNING_BYTES,
|
|
67
|
+
WORKER_RELEASE_DEADLINE_S,
|
|
68
|
+
)
|
|
69
|
+
from ..util import ids
|
|
70
|
+
from ..util.logging import get_logger
|
|
71
|
+
from .migrations import SUPPORTED_SCHEMA_VERSION, default_backup_dir, translate_sqlite_error
|
|
72
|
+
from .store import BackupRecord, Store
|
|
73
|
+
|
|
74
|
+
log = get_logger("db.backup")
|
|
75
|
+
|
|
76
|
+
__all__ = [
|
|
77
|
+
"BACKUP_FORMAT_VERSION",
|
|
78
|
+
"BACKUP_SUFFIX",
|
|
79
|
+
"DISCLOSURE",
|
|
80
|
+
"EXCLUDED_FROM_BACKUP",
|
|
81
|
+
"SCHEDULED_BACKUP_INTERVAL_S",
|
|
82
|
+
"BackupInspection",
|
|
83
|
+
"BackupOutcome",
|
|
84
|
+
"BackupPhase",
|
|
85
|
+
"BackupProgress",
|
|
86
|
+
"BackupScheduler",
|
|
87
|
+
"BackupSelection",
|
|
88
|
+
"CancelToken",
|
|
89
|
+
"ProgressCallback",
|
|
90
|
+
"RESTORE_GATE",
|
|
91
|
+
"RestoreGate",
|
|
92
|
+
"RestoreMode",
|
|
93
|
+
"RestoreOutcome",
|
|
94
|
+
"RestorePhase",
|
|
95
|
+
"RestoredItem",
|
|
96
|
+
"ScheduleDecision",
|
|
97
|
+
"ScheduleReason",
|
|
98
|
+
"create_backup",
|
|
99
|
+
"inspect_backup",
|
|
100
|
+
"last_scheduled_run",
|
|
101
|
+
"prune_scheduled_backups",
|
|
102
|
+
"restore_backup",
|
|
103
|
+
"scheduled_backup_name",
|
|
104
|
+
"verify_backup",
|
|
105
|
+
]
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
# ======================================================================
|
|
109
|
+
# The archive's own contract
|
|
110
|
+
# ======================================================================
|
|
111
|
+
|
|
112
|
+
#: The bundle layout. Bumped when a member's meaning changes; a reader that
|
|
113
|
+
#: does not recognise the number refuses rather than guessing (N-15).
|
|
114
|
+
BACKUP_FORMAT_VERSION: Final = 1
|
|
115
|
+
|
|
116
|
+
BACKUP_SUFFIX: Final = ".echoactbak"
|
|
117
|
+
|
|
118
|
+
MANIFEST_NAME: Final = "manifest.json"
|
|
119
|
+
DOCUMENTS_MEMBER: Final = "data/documents.jsonl"
|
|
120
|
+
JOBS_MEMBER: Final = "data/jobs.jsonl"
|
|
121
|
+
SEGMENTS_MEMBER: Final = "data/segments.jsonl"
|
|
122
|
+
RESULTS_MEMBER: Final = "data/results.jsonl"
|
|
123
|
+
AUDIO_PREFIX: Final = "audio/"
|
|
124
|
+
|
|
125
|
+
_TABLE_MEMBERS: Final = (DOCUMENTS_MEMBER, JOBS_MEMBER, SEGMENTS_MEMBER, RESULTS_MEMBER)
|
|
126
|
+
|
|
127
|
+
#: Which field of :class:`BackupCounts` each table member has to agree with.
|
|
128
|
+
#: 4.1's item ceiling is counted in these, so the mapping is what ties a
|
|
129
|
+
#: declared count to a member whose rows can be counted.
|
|
130
|
+
_COUNTED_MEMBERS: Final = {
|
|
131
|
+
DOCUMENTS_MEMBER: "documents",
|
|
132
|
+
JOBS_MEMBER: "jobs",
|
|
133
|
+
SEGMENTS_MEMBER: "segments",
|
|
134
|
+
RESULTS_MEMBER: "results",
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
#: F-44's disclosure, in English. The GUI localises it through
|
|
138
|
+
#: ``echoact.ui.i18n``; it is stated here as well because it is written into
|
|
139
|
+
#: the manifest, so a bundle discloses its own contents wherever it is opened.
|
|
140
|
+
DISCLOSURE: Final = (
|
|
141
|
+
"This backup contains the text of your saved documents, the text of your "
|
|
142
|
+
"retained job history, and the generated audio you selected. Keep it "
|
|
143
|
+
"somewhere you would keep the documents themselves. It is not encrypted."
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
#: What a bundle never contains, recorded in the manifest so the exclusion is
|
|
147
|
+
#: legible from the file itself (F-44, 4.2).
|
|
148
|
+
EXCLUDED_FROM_BACKUP: Final = (
|
|
149
|
+
"model_weights",
|
|
150
|
+
"credentials",
|
|
151
|
+
"client_permissions",
|
|
152
|
+
"idempotency_records",
|
|
153
|
+
"in_progress_results",
|
|
154
|
+
"one_off_results",
|
|
155
|
+
"segment_scratch_audio",
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
#: F-74's "once daily". ``echoact.policy`` fixes how many scheduled backups
|
|
159
|
+
#: are kept but not the period, so it is named here rather than spelled 86400
|
|
160
|
+
#: at the two places that need it.
|
|
161
|
+
SCHEDULED_BACKUP_INTERVAL_S: Final = 24 * 60 * 60
|
|
162
|
+
|
|
163
|
+
#: How long a failed scheduled attempt is left alone before another is tried.
|
|
164
|
+
#: Without it a one-minute tick would retry a failing backup sixty times an
|
|
165
|
+
#: hour; with it a transient failure still recovers well inside F-74's day.
|
|
166
|
+
SCHEDULED_BACKUP_RETRY_S: Final = 60 * 60
|
|
167
|
+
|
|
168
|
+
#: Read granularity. Not a policy limit: it is how often cancellation and
|
|
169
|
+
#: progress are observable while a large WAV is copied.
|
|
170
|
+
_CHUNK_BYTES: Final = 1 << 20
|
|
171
|
+
|
|
172
|
+
#: A manifest is a few hundred bytes per table. Anything larger is either a
|
|
173
|
+
#: bomb or not our manifest, and it is parsed before any limit is known.
|
|
174
|
+
_MAX_MANIFEST_BYTES: Final = 1 << 20
|
|
175
|
+
|
|
176
|
+
#: One JSON line holds at most one document body or one job snapshot, and F-03
|
|
177
|
+
#: caps either at 50,000 code points -- 150 KB of UTF-8 at the worst. The cap
|
|
178
|
+
#: is far above that and exists only so a line that never ends cannot be read
|
|
179
|
+
#: into memory forever.
|
|
180
|
+
_MAX_RECORD_BYTES: Final = 8 << 20
|
|
181
|
+
|
|
182
|
+
_SAFE_AUDIO_NAME: Final = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]{0,127}$")
|
|
183
|
+
_SAFE_ID: Final = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._-]{0,63}$")
|
|
184
|
+
_DRIVE_LETTER: Final = re.compile(r"^[A-Za-z]:")
|
|
185
|
+
|
|
186
|
+
_TERMINAL_STATE_VALUES: Final = tuple(
|
|
187
|
+
s.value for s in JobState if s.is_terminal
|
|
188
|
+
)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
# ======================================================================
|
|
192
|
+
# Progress and cancellation (the shape models.registry already uses)
|
|
193
|
+
# ======================================================================
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
class BackupPhase(StrEnum):
|
|
197
|
+
SNAPSHOT = "snapshot"
|
|
198
|
+
WRITING = "writing"
|
|
199
|
+
VERIFYING = "verifying"
|
|
200
|
+
COMPLETE = "complete"
|
|
201
|
+
CANCELLED = "cancelled"
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
class RestorePhase(StrEnum):
|
|
205
|
+
CHECKING = "checking"
|
|
206
|
+
PREPARING = "preparing"
|
|
207
|
+
APPLYING = "applying"
|
|
208
|
+
COMPLETE = "complete"
|
|
209
|
+
CANCELLED = "cancelled"
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
@dataclass(frozen=True, slots=True)
|
|
213
|
+
class BackupProgress:
|
|
214
|
+
"""One progress tick, for either direction.
|
|
215
|
+
|
|
216
|
+
5.3 requires progress state during a restore, and F-64's download report
|
|
217
|
+
already taught the GUI to read a phase, a label, and two running totals;
|
|
218
|
+
this is the same shape so the same widget can show it.
|
|
219
|
+
"""
|
|
220
|
+
|
|
221
|
+
phase: BackupPhase | RestorePhase
|
|
222
|
+
item: str
|
|
223
|
+
items_done: int
|
|
224
|
+
items_total: int
|
|
225
|
+
bytes_done: int
|
|
226
|
+
bytes_total: int
|
|
227
|
+
|
|
228
|
+
@property
|
|
229
|
+
def fraction(self) -> float:
|
|
230
|
+
if self.bytes_total > 0:
|
|
231
|
+
return min(1.0, self.bytes_done / self.bytes_total)
|
|
232
|
+
if self.items_total > 0:
|
|
233
|
+
return min(1.0, self.items_done / self.items_total)
|
|
234
|
+
return 1.0
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
ProgressCallback = Callable[[BackupProgress], None]
|
|
238
|
+
FreeSpace = Callable[[Path], int]
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _cancelled(token: CancelToken | None) -> bool:
|
|
242
|
+
return token is not None and token.cancelled
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
# ======================================================================
|
|
246
|
+
# 5.3: while a restore runs, new generation and edits are blocked
|
|
247
|
+
# ======================================================================
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
class RestoreGate:
|
|
251
|
+
"""The flag the rest of the app checks before it writes anything.
|
|
252
|
+
|
|
253
|
+
5.3 blocks new generation and edits for the duration of a restore. The
|
|
254
|
+
block lives here rather than in the job engine because the restore is what
|
|
255
|
+
knows when it starts and ends, and because the same answer has to serve the
|
|
256
|
+
engine, the GUI, and the REST surface. It is deliberately a plain object
|
|
257
|
+
with an explicit lifetime: a module-level boolean nobody clears on an
|
|
258
|
+
exception is how an app ends up permanently refusing to generate.
|
|
259
|
+
"""
|
|
260
|
+
|
|
261
|
+
__slots__ = ("_lock", "_active", "_progress")
|
|
262
|
+
|
|
263
|
+
def __init__(self) -> None:
|
|
264
|
+
self._lock = threading.Lock()
|
|
265
|
+
self._active = False
|
|
266
|
+
self._progress: BackupProgress | None = None
|
|
267
|
+
|
|
268
|
+
@property
|
|
269
|
+
def active(self) -> bool:
|
|
270
|
+
with self._lock:
|
|
271
|
+
return self._active
|
|
272
|
+
|
|
273
|
+
@property
|
|
274
|
+
def progress(self) -> BackupProgress | None:
|
|
275
|
+
"""5.3's "progress state is provided", readable from another thread."""
|
|
276
|
+
with self._lock:
|
|
277
|
+
return self._progress
|
|
278
|
+
|
|
279
|
+
@contextmanager
|
|
280
|
+
def hold(self) -> Iterator[None]:
|
|
281
|
+
with self._lock:
|
|
282
|
+
if self._active:
|
|
283
|
+
raise EchoActError(
|
|
284
|
+
Code.BUSY,
|
|
285
|
+
"A backup is already being restored.",
|
|
286
|
+
retry_after_s=WORKER_RELEASE_DEADLINE_S,
|
|
287
|
+
)
|
|
288
|
+
self._active = True
|
|
289
|
+
self._progress = None
|
|
290
|
+
try:
|
|
291
|
+
yield
|
|
292
|
+
finally:
|
|
293
|
+
with self._lock:
|
|
294
|
+
self._active = False
|
|
295
|
+
self._progress = None
|
|
296
|
+
|
|
297
|
+
def note(self, progress: BackupProgress) -> None:
|
|
298
|
+
with self._lock:
|
|
299
|
+
if self._active:
|
|
300
|
+
self._progress = progress
|
|
301
|
+
|
|
302
|
+
def require_idle(self, what: str = "This action") -> None:
|
|
303
|
+
"""Raise if a restore is running. Retryable: it will finish."""
|
|
304
|
+
if self.active:
|
|
305
|
+
raise EchoActError(
|
|
306
|
+
Code.BUSY,
|
|
307
|
+
f"{what} is not possible while a backup is being restored.",
|
|
308
|
+
retry_after_s=WORKER_RELEASE_DEADLINE_S,
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
#: The process-wide gate. One restore at a time, one flag for everyone.
|
|
313
|
+
RESTORE_GATE: Final = RestoreGate()
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
# ======================================================================
|
|
317
|
+
# What goes in, what came out
|
|
318
|
+
# ======================================================================
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
@dataclass(frozen=True, slots=True)
|
|
322
|
+
class BackupSelection:
|
|
323
|
+
"""F-44's "documents, history, and audio selected by the user".
|
|
324
|
+
|
|
325
|
+
``audio=False`` drops results *and* segments, not just the WAV files: a
|
|
326
|
+
segment without audio is a row F-45 would immediately report as a missing
|
|
327
|
+
result, so a bundle that kept them would restore into a library full of
|
|
328
|
+
findings the user never caused.
|
|
329
|
+
"""
|
|
330
|
+
|
|
331
|
+
documents: bool = True
|
|
332
|
+
history: bool = True
|
|
333
|
+
audio: bool = True
|
|
334
|
+
document_ids: tuple[str, ...] | None = None
|
|
335
|
+
job_ids: tuple[str, ...] | None = None
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
@dataclass(frozen=True, slots=True)
|
|
339
|
+
class BackupCounts:
|
|
340
|
+
documents: int = 0
|
|
341
|
+
jobs: int = 0
|
|
342
|
+
segments: int = 0
|
|
343
|
+
results: int = 0
|
|
344
|
+
audio_files: int = 0
|
|
345
|
+
|
|
346
|
+
@property
|
|
347
|
+
def items(self) -> int:
|
|
348
|
+
"""4.1 counts items after decompression; this is what a restore adds."""
|
|
349
|
+
return self.documents + self.jobs + self.segments + self.results + self.audio_files
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
@dataclass(frozen=True, slots=True)
|
|
353
|
+
class BackupOutcome:
|
|
354
|
+
path: Path
|
|
355
|
+
kind: str
|
|
356
|
+
created_at: float
|
|
357
|
+
counts: BackupCounts
|
|
358
|
+
byte_size: int
|
|
359
|
+
uncompressed_bytes: int
|
|
360
|
+
#: Results whose audio file was no longer on disk. There is nothing to
|
|
361
|
+
#: bundle, so neither the file nor the row is in the archive.
|
|
362
|
+
missing_results: tuple[str, ...]
|
|
363
|
+
#: Results whose file no longer matched 4.2's stored digest. The bytes on
|
|
364
|
+
#: disk are what the user has, so they are bundled and the archive records
|
|
365
|
+
#: their true digest; F-45's finding is reported rather than acted on here.
|
|
366
|
+
corrupt_results: tuple[str, ...]
|
|
367
|
+
cancelled: bool
|
|
368
|
+
record: BackupRecord | None
|
|
369
|
+
|
|
370
|
+
@property
|
|
371
|
+
def item_count(self) -> int:
|
|
372
|
+
return self.counts.items
|
|
373
|
+
|
|
374
|
+
@property
|
|
375
|
+
def contains_body_text(self) -> bool:
|
|
376
|
+
"""F-44's disclosure, as a fact about this file."""
|
|
377
|
+
return bool(self.counts.documents or self.counts.jobs)
|
|
378
|
+
|
|
379
|
+
@property
|
|
380
|
+
def contains_audio(self) -> bool:
|
|
381
|
+
return bool(self.counts.audio_files)
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
@dataclass(frozen=True, slots=True)
|
|
385
|
+
class BackupInspection:
|
|
386
|
+
"""What verification learned, before a single row is applied (N-27)."""
|
|
387
|
+
|
|
388
|
+
path: Path
|
|
389
|
+
format_version: int
|
|
390
|
+
app_version: str
|
|
391
|
+
schema_version: int
|
|
392
|
+
created_at: float
|
|
393
|
+
kind: str
|
|
394
|
+
counts: BackupCounts
|
|
395
|
+
uncompressed_bytes: int
|
|
396
|
+
compressed_bytes: int
|
|
397
|
+
retention_bytes: int
|
|
398
|
+
disclosure: str
|
|
399
|
+
deep: bool
|
|
400
|
+
|
|
401
|
+
@property
|
|
402
|
+
def item_count(self) -> int:
|
|
403
|
+
return self.counts.items
|
|
404
|
+
|
|
405
|
+
@property
|
|
406
|
+
def contains_body_text(self) -> bool:
|
|
407
|
+
return bool(self.counts.documents or self.counts.jobs)
|
|
408
|
+
|
|
409
|
+
@property
|
|
410
|
+
def contains_audio(self) -> bool:
|
|
411
|
+
return bool(self.counts.audio_files)
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
class RestoreMode(StrEnum):
|
|
415
|
+
"""F-44: restore adds new items *by default*."""
|
|
416
|
+
|
|
417
|
+
ADD = "add"
|
|
418
|
+
#: Everything the app manages is deleted first. N-15 governs this path:
|
|
419
|
+
#: the existing data is read and bundled into a recoverable backup, which
|
|
420
|
+
#: is what "validated before being overwritten" means in practice.
|
|
421
|
+
REPLACE = "replace"
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
@dataclass(frozen=True, slots=True)
|
|
425
|
+
class RestoredItem:
|
|
426
|
+
"""F-44's "original identifiers are distinguished from post-restore ones"."""
|
|
427
|
+
|
|
428
|
+
kind: str
|
|
429
|
+
original_id: str
|
|
430
|
+
restored_id: str
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
@dataclass(frozen=True, slots=True)
|
|
434
|
+
class RestoreOutcome:
|
|
435
|
+
source: Path
|
|
436
|
+
mode: RestoreMode
|
|
437
|
+
counts: BackupCounts
|
|
438
|
+
items: tuple[RestoredItem, ...]
|
|
439
|
+
owner_client_id: str
|
|
440
|
+
safety_backup: Path | None
|
|
441
|
+
cancelled: bool
|
|
442
|
+
|
|
443
|
+
def restored_id(self, original_id: str) -> str | None:
|
|
444
|
+
for item in self.items:
|
|
445
|
+
if item.original_id == original_id:
|
|
446
|
+
return item.restored_id
|
|
447
|
+
return None
|
|
448
|
+
|
|
449
|
+
def originals(self, kind: str) -> tuple[str, ...]:
|
|
450
|
+
return tuple(i.original_id for i in self.items if i.kind == kind)
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
# ======================================================================
|
|
454
|
+
# Small helpers
|
|
455
|
+
# ======================================================================
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
def _moment(at: float | None) -> float:
|
|
459
|
+
return ids.now() if at is None else at
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _disk_free(target: Path) -> int:
|
|
463
|
+
probe = target
|
|
464
|
+
while not probe.exists() and probe.parent != probe:
|
|
465
|
+
probe = probe.parent
|
|
466
|
+
try:
|
|
467
|
+
return int(shutil.disk_usage(probe).free)
|
|
468
|
+
except OSError:
|
|
469
|
+
# An unreadable device is not evidence of a full one; the write will
|
|
470
|
+
# report the truth. 4.1's warning is a guard, not the only check.
|
|
471
|
+
return 1 << 62
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def _require_space(target: Path, needed: int, free_space: FreeSpace) -> None:
|
|
475
|
+
"""4.1: below 1 GB free, backups are not started."""
|
|
476
|
+
free = free_space(target)
|
|
477
|
+
if free < needed + LOW_SPACE_WARNING_BYTES:
|
|
478
|
+
raise EchoActError(
|
|
479
|
+
Code.STORAGE_FULL,
|
|
480
|
+
"There is not enough free disk space for this backup.",
|
|
481
|
+
detail={"needed_bytes": needed, "free_bytes": free},
|
|
482
|
+
)
|
|
483
|
+
|
|
484
|
+
|
|
485
|
+
def _chunks[T](values: Sequence[T], size: int = 400) -> Iterator[Sequence[T]]:
|
|
486
|
+
for start in range(0, len(values), size):
|
|
487
|
+
yield values[start : start + size]
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def _json_line(payload: dict[str, Any]) -> bytes:
|
|
491
|
+
return (json.dumps(payload, ensure_ascii=False, separators=(",", ":")) + "\n").encode("utf-8")
|
|
492
|
+
|
|
493
|
+
|
|
494
|
+
def _as_object(value: Any, what: str) -> dict[str, Any]:
|
|
495
|
+
if not isinstance(value, dict):
|
|
496
|
+
raise EchoActError(Code.BACKUP_INVALID, f"The backup's {what} is not an object.")
|
|
497
|
+
return value
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def scheduled_backup_name(at: float, *, kind: str = "scheduled") -> str:
|
|
501
|
+
"""A sortable, collision-free file name for an automatic backup (F-74)."""
|
|
502
|
+
stamp = time.strftime("%Y%m%dT%H%M%S", time.gmtime(at))
|
|
503
|
+
return f"echoact-{kind}-{stamp}-{ids.backup_id()}{BACKUP_SUFFIX}"
|
|
504
|
+
|
|
505
|
+
|
|
506
|
+
# ======================================================================
|
|
507
|
+
# The point-in-time snapshot (N-27, 5.3)
|
|
508
|
+
# ======================================================================
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
class _Snapshot:
|
|
512
|
+
"""One read transaction, held open for the life of the bundle.
|
|
513
|
+
|
|
514
|
+
Opened on a connection of its own rather than borrowing the store's:
|
|
515
|
+
``Store`` hands out one connection per thread and other work on this thread
|
|
516
|
+
would join -- and eventually commit -- the transaction the snapshot depends
|
|
517
|
+
on. In write-ahead mode this reader sees the database as it was when the
|
|
518
|
+
transaction began and never blocks a writer, which is precisely 5.3's
|
|
519
|
+
"a document changed during backup contributes its state at the start".
|
|
520
|
+
"""
|
|
521
|
+
|
|
522
|
+
__slots__ = ("_conn", "_path")
|
|
523
|
+
|
|
524
|
+
def __init__(self, path: Path, *, busy_timeout_s: float) -> None:
|
|
525
|
+
self._path = path
|
|
526
|
+
try:
|
|
527
|
+
conn = sqlite3.connect(
|
|
528
|
+
path, timeout=busy_timeout_s, isolation_level=None, check_same_thread=False
|
|
529
|
+
)
|
|
530
|
+
conn.row_factory = sqlite3.Row
|
|
531
|
+
conn.execute(f"PRAGMA busy_timeout = {int(busy_timeout_s * 1000)}")
|
|
532
|
+
conn.execute("PRAGMA query_only = 1")
|
|
533
|
+
conn.execute("BEGIN DEFERRED")
|
|
534
|
+
# A deferred transaction takes its read snapshot at the first read,
|
|
535
|
+
# not at BEGIN, so the snapshot has to be pinned here rather than
|
|
536
|
+
# at whichever table happens to be queried first.
|
|
537
|
+
conn.execute("SELECT count(*) FROM sqlite_master").fetchone()
|
|
538
|
+
except sqlite3.Error as exc:
|
|
539
|
+
raise translate_sqlite_error(exc) from exc
|
|
540
|
+
self._conn = conn
|
|
541
|
+
|
|
542
|
+
def query(self, sql: str, params: Sequence[Any] = ()) -> list[sqlite3.Row]:
|
|
543
|
+
try:
|
|
544
|
+
return self._conn.execute(sql, tuple(params)).fetchall()
|
|
545
|
+
except sqlite3.Error as exc:
|
|
546
|
+
raise translate_sqlite_error(exc) from exc
|
|
547
|
+
|
|
548
|
+
def scalar(self, sql: str, params: Sequence[Any] = ()) -> Any:
|
|
549
|
+
rows = self.query(sql, params)
|
|
550
|
+
return rows[0][0] if rows else None
|
|
551
|
+
|
|
552
|
+
def close(self) -> None:
|
|
553
|
+
with suppress(sqlite3.Error):
|
|
554
|
+
self._conn.execute("ROLLBACK")
|
|
555
|
+
with suppress(sqlite3.Error):
|
|
556
|
+
self._conn.close()
|
|
557
|
+
|
|
558
|
+
def __enter__(self) -> _Snapshot:
|
|
559
|
+
return self
|
|
560
|
+
|
|
561
|
+
def __exit__(self, *exc: object) -> None:
|
|
562
|
+
self.close()
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
# ======================================================================
|
|
566
|
+
# Writing a bundle (F-44)
|
|
567
|
+
# ======================================================================
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
class _MemberWriter:
|
|
571
|
+
"""A zip member that hashes what it is given as it goes.
|
|
572
|
+
|
|
573
|
+
Always used as a context manager, and that is load-bearing rather than
|
|
574
|
+
tidiness. ``ZipFile.close`` refuses outright -- with a ``ValueError``,
|
|
575
|
+
raised *before* it releases the file object -- while a writing handle is
|
|
576
|
+
still open on it. So a read error part way through a WAV, or an
|
|
577
|
+
``EchoActError`` from the snapshot part way through a table, would unwind
|
|
578
|
+
through ``__exit__`` and come out as that ``ValueError`` instead: the real
|
|
579
|
+
cause replaced by a builtin no caller has code for (rule 3), the archive's
|
|
580
|
+
handle leaked, and -- on Windows, where an open file cannot be unlinked --
|
|
581
|
+
the half-written ``.part-`` file left in the directory the user chose,
|
|
582
|
+
which is exactly what :func:`create_backup` promises never to do.
|
|
583
|
+
"""
|
|
584
|
+
|
|
585
|
+
__slots__ = ("_raw", "_hash", "bytes_written")
|
|
586
|
+
|
|
587
|
+
def __init__(self, zf: zipfile.ZipFile, name: str, *, force_zip64: bool = False) -> None:
|
|
588
|
+
self._raw = zf.open(name, "w", force_zip64=force_zip64)
|
|
589
|
+
self._hash = hashlib.sha256()
|
|
590
|
+
self.bytes_written = 0
|
|
591
|
+
|
|
592
|
+
def write(self, data: bytes) -> None:
|
|
593
|
+
self._raw.write(data)
|
|
594
|
+
self._hash.update(data)
|
|
595
|
+
self.bytes_written += len(data)
|
|
596
|
+
|
|
597
|
+
def close(self) -> tuple[str, int]:
|
|
598
|
+
self._raw.close()
|
|
599
|
+
return self._hash.hexdigest(), self.bytes_written
|
|
600
|
+
|
|
601
|
+
def __enter__(self) -> _MemberWriter:
|
|
602
|
+
return self
|
|
603
|
+
|
|
604
|
+
def __exit__(self, exc_type: type[BaseException] | None, *_: object) -> None:
|
|
605
|
+
if exc_type is None:
|
|
606
|
+
return
|
|
607
|
+
# Closing a member whose write has just failed can fail in turn, and
|
|
608
|
+
# the failure already unwinding is the one worth reporting. The
|
|
609
|
+
# handle is released either way: zipfile clears its writing flag in a
|
|
610
|
+
# finally, so the archive can still be closed and the file unlinked.
|
|
611
|
+
with suppress(Exception):
|
|
612
|
+
self._raw.close()
|
|
613
|
+
|
|
614
|
+
|
|
615
|
+
def _resolve_inside(root: Path, relative_path: str) -> Path | None:
|
|
616
|
+
"""Resolve a stored relative path, or ``None`` if it leaves the root.
|
|
617
|
+
|
|
618
|
+
The same rule ``Store`` applies when it reads a result, repeated here
|
|
619
|
+
because this is the module that would otherwise copy the escaping file
|
|
620
|
+
into a bundle. N-27 and 4.2 together mean a stored path must never be
|
|
621
|
+
able to pull a credential or a model file into a backup.
|
|
622
|
+
"""
|
|
623
|
+
if not relative_path:
|
|
624
|
+
return None
|
|
625
|
+
try:
|
|
626
|
+
candidate = (root / relative_path).resolve()
|
|
627
|
+
candidate.relative_to(root.resolve())
|
|
628
|
+
except (OSError, ValueError):
|
|
629
|
+
return None
|
|
630
|
+
return candidate
|
|
631
|
+
|
|
632
|
+
|
|
633
|
+
def _own_limits(counts: BackupCounts, uncompressed: int) -> tuple[int, int]:
|
|
634
|
+
"""Ceilings for reading back a bundle this module has just written.
|
|
635
|
+
|
|
636
|
+
4.1 caps what a *restore* may decompress and apply. It says nothing about
|
|
637
|
+
what a library may hold, and F-44 and F-74 promise a backup of whatever it
|
|
638
|
+
does hold. Holding a freshly written bundle to the restore ceiling
|
|
639
|
+
enforces 4.1 on the side it does not govern while leaving the side it does
|
|
640
|
+
unguarded: about a hundred long retained jobs is 100,000 segments, and past
|
|
641
|
+
that line every manual backup fails, the scheduled backup fails on every
|
|
642
|
+
tick for ever, and the N-15 copy a replace takes of the live library cannot
|
|
643
|
+
be written either -- so the data is unbackupable exactly when there is most
|
|
644
|
+
of it. A file written from the live database seconds ago is not hostile
|
|
645
|
+
input; the self-check is looking for a bad write, so it is run against what
|
|
646
|
+
was actually written.
|
|
647
|
+
"""
|
|
648
|
+
# The archive holds one entry per audio file, one per table, and the
|
|
649
|
+
# manifest; ``items`` counts audio and result rows separately, so this is
|
|
650
|
+
# never the binding figure, but it is stated rather than assumed.
|
|
651
|
+
entries = counts.audio_files + len(_TABLE_MEMBERS) + 1
|
|
652
|
+
return (
|
|
653
|
+
max(BACKUP_RESTORE_MAX_BYTES, uncompressed + _MAX_MANIFEST_BYTES),
|
|
654
|
+
max(BACKUP_RESTORE_MAX_ITEMS, counts.items, entries),
|
|
655
|
+
)
|
|
656
|
+
|
|
657
|
+
|
|
658
|
+
def create_backup(
|
|
659
|
+
store: Store,
|
|
660
|
+
destination: str | os.PathLike[str],
|
|
661
|
+
*,
|
|
662
|
+
selection: BackupSelection | None = None,
|
|
663
|
+
kind: str = "manual",
|
|
664
|
+
at: float | None = None,
|
|
665
|
+
progress: ProgressCallback | None = None,
|
|
666
|
+
cancel: CancelToken | None = None,
|
|
667
|
+
free_space: FreeSpace = _disk_free,
|
|
668
|
+
overwrite: bool = False,
|
|
669
|
+
record: bool = True,
|
|
670
|
+
deep_verify: bool = True,
|
|
671
|
+
gate: RestoreGate | None = None,
|
|
672
|
+
) -> BackupOutcome:
|
|
673
|
+
"""Write one bundle from one point in time (F-44, N-27, 5.3).
|
|
674
|
+
|
|
675
|
+
The bundle is built at a temporary name beside the destination and moved
|
|
676
|
+
into place only once it has been verified, so a cancelled or failed backup
|
|
677
|
+
never leaves a plausible-looking file where the user asked for a good one --
|
|
678
|
+
and, for F-74, never leaves one that rotation would later count as sound.
|
|
679
|
+
"""
|
|
680
|
+
sel = selection or BackupSelection()
|
|
681
|
+
(gate or RESTORE_GATE).require_idle("Taking a backup")
|
|
682
|
+
moment = _moment(at)
|
|
683
|
+
target = Path(destination)
|
|
684
|
+
if target.exists() and not overwrite:
|
|
685
|
+
raise EchoActError(
|
|
686
|
+
Code.BACKUP_INVALID,
|
|
687
|
+
"A file already exists at that location.",
|
|
688
|
+
detail={"path": redact(target)},
|
|
689
|
+
)
|
|
690
|
+
|
|
691
|
+
with _Snapshot(store.path, busy_timeout_s=store.busy_timeout_s) as snap:
|
|
692
|
+
plan = _plan(snap, sel)
|
|
693
|
+
_require_space(target.parent, plan.audio_bytes + plan.text_bytes, free_space)
|
|
694
|
+
_emit(
|
|
695
|
+
progress,
|
|
696
|
+
BackupPhase.SNAPSHOT,
|
|
697
|
+
"snapshot",
|
|
698
|
+
0,
|
|
699
|
+
plan.counts.items,
|
|
700
|
+
0,
|
|
701
|
+
plan.audio_bytes,
|
|
702
|
+
)
|
|
703
|
+
temp = target.with_name(target.name + f".part-{ids.backup_id()}")
|
|
704
|
+
try:
|
|
705
|
+
counts, missing, corrupt, uncompressed = _write_archive(
|
|
706
|
+
snap, store, temp, plan, moment, kind, progress, cancel
|
|
707
|
+
)
|
|
708
|
+
except BaseException:
|
|
709
|
+
_discard(temp)
|
|
710
|
+
raise
|
|
711
|
+
if _cancelled(cancel):
|
|
712
|
+
_discard(temp)
|
|
713
|
+
return BackupOutcome(
|
|
714
|
+
path=target,
|
|
715
|
+
kind=kind,
|
|
716
|
+
created_at=moment,
|
|
717
|
+
counts=BackupCounts(),
|
|
718
|
+
byte_size=0,
|
|
719
|
+
uncompressed_bytes=0,
|
|
720
|
+
missing_results=missing,
|
|
721
|
+
corrupt_results=corrupt,
|
|
722
|
+
cancelled=True,
|
|
723
|
+
record=None,
|
|
724
|
+
)
|
|
725
|
+
|
|
726
|
+
_emit(progress, BackupPhase.VERIFYING, target.name, counts.items, counts.items, 0, 0)
|
|
727
|
+
self_max_bytes, self_max_items = _own_limits(counts, uncompressed)
|
|
728
|
+
try:
|
|
729
|
+
verify_backup(temp, deep=deep_verify, max_bytes=self_max_bytes, max_items=self_max_items)
|
|
730
|
+
except EchoActError:
|
|
731
|
+
_discard(temp)
|
|
732
|
+
raise
|
|
733
|
+
try:
|
|
734
|
+
os.replace(temp, target)
|
|
735
|
+
except OSError as exc:
|
|
736
|
+
_discard(temp)
|
|
737
|
+
raise EchoActError(
|
|
738
|
+
Code.STORAGE_FULL if getattr(exc, "errno", None) == 28 else Code.BACKUP_INVALID,
|
|
739
|
+
"The backup could not be moved into place.",
|
|
740
|
+
detail={"path": redact(target)},
|
|
741
|
+
cause=exc,
|
|
742
|
+
) from exc
|
|
743
|
+
|
|
744
|
+
size = target.stat().st_size
|
|
745
|
+
written: BackupRecord | None = None
|
|
746
|
+
if record:
|
|
747
|
+
written = store.record_backup(
|
|
748
|
+
location=str(target),
|
|
749
|
+
kind=kind,
|
|
750
|
+
byte_size=size,
|
|
751
|
+
item_count=counts.items,
|
|
752
|
+
verified=True,
|
|
753
|
+
note=json.dumps({"format": BACKUP_FORMAT_VERSION, "audio": counts.audio_files}),
|
|
754
|
+
at=moment,
|
|
755
|
+
)
|
|
756
|
+
log.info(
|
|
757
|
+
"backup written kind=%s items=%d bytes=%d path=%s",
|
|
758
|
+
kind,
|
|
759
|
+
counts.items,
|
|
760
|
+
size,
|
|
761
|
+
redact(target),
|
|
762
|
+
)
|
|
763
|
+
_emit(progress, BackupPhase.COMPLETE, target.name, counts.items, counts.items, size, size)
|
|
764
|
+
return BackupOutcome(
|
|
765
|
+
path=target,
|
|
766
|
+
kind=kind,
|
|
767
|
+
created_at=moment,
|
|
768
|
+
counts=counts,
|
|
769
|
+
byte_size=size,
|
|
770
|
+
uncompressed_bytes=uncompressed,
|
|
771
|
+
missing_results=missing,
|
|
772
|
+
corrupt_results=corrupt,
|
|
773
|
+
cancelled=False,
|
|
774
|
+
record=written,
|
|
775
|
+
)
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
@dataclass(frozen=True, slots=True)
|
|
779
|
+
class _Plan:
|
|
780
|
+
document_ids: tuple[str, ...]
|
|
781
|
+
job_ids: tuple[str, ...]
|
|
782
|
+
#: Empty when audio was not selected: a segment's meaning is the audio it
|
|
783
|
+
#: points at, so the two travel together or not at all.
|
|
784
|
+
segment_job_ids: tuple[str, ...]
|
|
785
|
+
result_rows: tuple[sqlite3.Row, ...]
|
|
786
|
+
counts: BackupCounts
|
|
787
|
+
audio_bytes: int
|
|
788
|
+
text_bytes: int
|
|
789
|
+
|
|
790
|
+
|
|
791
|
+
def _plan(snap: _Snapshot, sel: BackupSelection) -> _Plan:
|
|
792
|
+
"""Decide what is in the bundle, entirely from one snapshot.
|
|
793
|
+
|
|
794
|
+
Two exclusions are requirements rather than choices. 5.3 keeps the
|
|
795
|
+
temporary results of in-progress jobs out, and an in-progress job restored
|
|
796
|
+
without them would sit in a state 5.1 says can never be advanced -- so the
|
|
797
|
+
whole job is left out, not merely its result. 4.1 makes one-off jobs
|
|
798
|
+
temporary by definition, so "history" means retained history.
|
|
799
|
+
"""
|
|
800
|
+
documents: tuple[str, ...] = ()
|
|
801
|
+
if sel.documents:
|
|
802
|
+
rows = snap.query("SELECT document_id FROM documents ORDER BY created_at")
|
|
803
|
+
documents = tuple(r["document_id"] for r in rows)
|
|
804
|
+
if sel.document_ids is not None:
|
|
805
|
+
wanted = set(sel.document_ids)
|
|
806
|
+
documents = tuple(d for d in documents if d in wanted)
|
|
807
|
+
|
|
808
|
+
jobs: tuple[str, ...] = ()
|
|
809
|
+
if sel.history:
|
|
810
|
+
marks = ",".join("?" * len(_TERMINAL_STATE_VALUES))
|
|
811
|
+
rows = snap.query(
|
|
812
|
+
f"SELECT job_id FROM jobs WHERE retention = ? AND state IN ({marks})"
|
|
813
|
+
" ORDER BY created_at",
|
|
814
|
+
(RetentionMode.RETAINED.value, *_TERMINAL_STATE_VALUES),
|
|
815
|
+
)
|
|
816
|
+
jobs = tuple(r["job_id"] for r in rows)
|
|
817
|
+
if sel.job_ids is not None:
|
|
818
|
+
wanted = set(sel.job_ids)
|
|
819
|
+
jobs = tuple(j for j in jobs if j in wanted)
|
|
820
|
+
|
|
821
|
+
segments = 0
|
|
822
|
+
results: list[sqlite3.Row] = []
|
|
823
|
+
if jobs and sel.audio:
|
|
824
|
+
for chunk in _chunks(jobs):
|
|
825
|
+
marks = ",".join("?" * len(chunk))
|
|
826
|
+
segments += int(
|
|
827
|
+
snap.scalar(
|
|
828
|
+
f"SELECT count(*) FROM segments WHERE job_id IN ({marks})", tuple(chunk)
|
|
829
|
+
)
|
|
830
|
+
or 0
|
|
831
|
+
)
|
|
832
|
+
results.extend(
|
|
833
|
+
snap.query(
|
|
834
|
+
"SELECT result_id, job_id, sample_rate, channels, sample_width_bits,"
|
|
835
|
+
" frame_count, byte_size, digest, relative_path, created_at, expires_at"
|
|
836
|
+
f" FROM results WHERE job_id IN ({marks}) AND expires_at IS NULL",
|
|
837
|
+
tuple(chunk),
|
|
838
|
+
)
|
|
839
|
+
)
|
|
840
|
+
|
|
841
|
+
text_bytes = 0
|
|
842
|
+
if documents:
|
|
843
|
+
for chunk in _chunks(documents):
|
|
844
|
+
marks = ",".join("?" * len(chunk))
|
|
845
|
+
text_bytes += int(
|
|
846
|
+
snap.scalar(
|
|
847
|
+
"SELECT coalesce(sum(length(CAST(title AS BLOB))"
|
|
848
|
+
f" + length(CAST(body AS BLOB))), 0) FROM documents WHERE document_id IN ({marks})",
|
|
849
|
+
tuple(chunk),
|
|
850
|
+
)
|
|
851
|
+
or 0
|
|
852
|
+
)
|
|
853
|
+
if jobs:
|
|
854
|
+
for chunk in _chunks(jobs):
|
|
855
|
+
marks = ",".join("?" * len(chunk))
|
|
856
|
+
text_bytes += int(
|
|
857
|
+
snap.scalar(
|
|
858
|
+
"SELECT coalesce(sum(length(CAST(source_text AS BLOB))), 0) FROM jobs"
|
|
859
|
+
f" WHERE job_id IN ({marks}) AND source_text IS NOT NULL",
|
|
860
|
+
tuple(chunk),
|
|
861
|
+
)
|
|
862
|
+
or 0
|
|
863
|
+
)
|
|
864
|
+
|
|
865
|
+
audio_bytes = sum(int(r["byte_size"]) for r in results)
|
|
866
|
+
return _Plan(
|
|
867
|
+
document_ids=documents,
|
|
868
|
+
job_ids=jobs,
|
|
869
|
+
segment_job_ids=jobs if sel.audio else (),
|
|
870
|
+
result_rows=tuple(results),
|
|
871
|
+
counts=BackupCounts(
|
|
872
|
+
documents=len(documents),
|
|
873
|
+
jobs=len(jobs),
|
|
874
|
+
segments=segments,
|
|
875
|
+
results=len(results),
|
|
876
|
+
audio_files=len(results),
|
|
877
|
+
),
|
|
878
|
+
audio_bytes=audio_bytes,
|
|
879
|
+
text_bytes=text_bytes,
|
|
880
|
+
)
|
|
881
|
+
|
|
882
|
+
|
|
883
|
+
def _write_archive(
|
|
884
|
+
snap: _Snapshot,
|
|
885
|
+
store: Store,
|
|
886
|
+
temp: Path,
|
|
887
|
+
plan: _Plan,
|
|
888
|
+
moment: float,
|
|
889
|
+
kind: str,
|
|
890
|
+
progress: ProgressCallback | None,
|
|
891
|
+
cancel: CancelToken | None,
|
|
892
|
+
) -> tuple[BackupCounts, tuple[str, ...], tuple[str, ...], int]:
|
|
893
|
+
"""Members in dependency order: audio, then tables, then the manifest.
|
|
894
|
+
|
|
895
|
+
Audio goes first because a zip member cannot be withdrawn once written:
|
|
896
|
+
which results are in the bundle is only known after their files have been
|
|
897
|
+
copied, and the results table has to agree with that. The manifest goes
|
|
898
|
+
last because it carries the digests of everything before it.
|
|
899
|
+
|
|
900
|
+
Each result row is exported with the digest of the bytes that went into
|
|
901
|
+
the archive rather than the one the database held. Normally they are the
|
|
902
|
+
same. When they are not, the file is what the user actually has, and an
|
|
903
|
+
archive whose recorded digest disagreed with its own contents could never
|
|
904
|
+
be restored -- so the bundle stays internally consistent and the
|
|
905
|
+
disagreement is reported instead.
|
|
906
|
+
"""
|
|
907
|
+
members: dict[str, dict[str, Any]] = {}
|
|
908
|
+
missing: list[str] = []
|
|
909
|
+
corrupt: list[str] = []
|
|
910
|
+
audio_root = store.audio_root
|
|
911
|
+
done_bytes = 0
|
|
912
|
+
done_items = 0
|
|
913
|
+
total_items = plan.counts.items
|
|
914
|
+
good_results: list[tuple[sqlite3.Row, str, int]] = []
|
|
915
|
+
uncompressed = 0
|
|
916
|
+
|
|
917
|
+
temp.parent.mkdir(parents=True, exist_ok=True)
|
|
918
|
+
try:
|
|
919
|
+
with zipfile.ZipFile(temp, "w", compression=zipfile.ZIP_DEFLATED, allowZip64=True) as zf:
|
|
920
|
+
for row in plan.result_rows:
|
|
921
|
+
if _cancelled(cancel):
|
|
922
|
+
return BackupCounts(), tuple(missing), tuple(corrupt), 0
|
|
923
|
+
result_id = str(row["result_id"])
|
|
924
|
+
if not _SAFE_ID.match(result_id):
|
|
925
|
+
raise EchoActError(
|
|
926
|
+
Code.BACKUP_INVALID,
|
|
927
|
+
"A stored result identifier is not a safe file name.",
|
|
928
|
+
detail={"result_id": result_id[:32]},
|
|
929
|
+
)
|
|
930
|
+
source = _resolve_inside(audio_root, str(row["relative_path"] or ""))
|
|
931
|
+
if source is None:
|
|
932
|
+
# A row pointing outside the audio directory is how a
|
|
933
|
+
# credential or a model file would end up in a bundle
|
|
934
|
+
# (N-27, 4.2). It is a corrupt database, not a missing
|
|
935
|
+
# file, and the safe answer is to write nothing at all.
|
|
936
|
+
raise EchoActError(
|
|
937
|
+
Code.BACKUP_INVALID,
|
|
938
|
+
"A stored audio path points outside the audio directory; "
|
|
939
|
+
"no backup was written.",
|
|
940
|
+
detail={"result_id": result_id},
|
|
941
|
+
)
|
|
942
|
+
name = f"{AUDIO_PREFIX}{result_id}.wav"
|
|
943
|
+
digest, size = _copy_into(zf, name, source, cancel)
|
|
944
|
+
if _cancelled(cancel):
|
|
945
|
+
return BackupCounts(), tuple(missing), tuple(corrupt), 0
|
|
946
|
+
if digest is None:
|
|
947
|
+
missing.append(result_id)
|
|
948
|
+
continue
|
|
949
|
+
if digest != str(row["digest"]):
|
|
950
|
+
corrupt.append(result_id)
|
|
951
|
+
members[name] = {"sha256": digest, "bytes": size}
|
|
952
|
+
good_results.append((row, digest, size))
|
|
953
|
+
uncompressed += size
|
|
954
|
+
done_bytes += size
|
|
955
|
+
done_items += 1
|
|
956
|
+
_emit(
|
|
957
|
+
progress,
|
|
958
|
+
BackupPhase.WRITING,
|
|
959
|
+
name,
|
|
960
|
+
done_items,
|
|
961
|
+
total_items,
|
|
962
|
+
done_bytes,
|
|
963
|
+
plan.audio_bytes,
|
|
964
|
+
)
|
|
965
|
+
|
|
966
|
+
for name, rows in (
|
|
967
|
+
(DOCUMENTS_MEMBER, _document_records(snap, plan.document_ids)),
|
|
968
|
+
(JOBS_MEMBER, _job_records(snap, plan.job_ids)),
|
|
969
|
+
(SEGMENTS_MEMBER, _segment_records(snap, plan.segment_job_ids)),
|
|
970
|
+
(RESULTS_MEMBER, _result_records(good_results)),
|
|
971
|
+
):
|
|
972
|
+
with _MemberWriter(zf, name) as writer:
|
|
973
|
+
count = 0
|
|
974
|
+
for payload in rows:
|
|
975
|
+
writer.write(_json_line(payload))
|
|
976
|
+
count += 1
|
|
977
|
+
digest, size = writer.close()
|
|
978
|
+
members[name] = {"sha256": digest, "bytes": size, "records": count}
|
|
979
|
+
uncompressed += size
|
|
980
|
+
done_items += count
|
|
981
|
+
_emit(
|
|
982
|
+
progress,
|
|
983
|
+
BackupPhase.WRITING,
|
|
984
|
+
name,
|
|
985
|
+
min(done_items, total_items),
|
|
986
|
+
total_items,
|
|
987
|
+
done_bytes,
|
|
988
|
+
plan.audio_bytes,
|
|
989
|
+
)
|
|
990
|
+
|
|
991
|
+
counts = BackupCounts(
|
|
992
|
+
documents=int(members[DOCUMENTS_MEMBER]["records"]),
|
|
993
|
+
jobs=int(members[JOBS_MEMBER]["records"]),
|
|
994
|
+
segments=int(members[SEGMENTS_MEMBER]["records"]),
|
|
995
|
+
results=int(members[RESULTS_MEMBER]["records"]),
|
|
996
|
+
audio_files=len(good_results),
|
|
997
|
+
)
|
|
998
|
+
manifest = {
|
|
999
|
+
"format": BACKUP_FORMAT_VERSION,
|
|
1000
|
+
"app_version": __version__,
|
|
1001
|
+
"schema_version": store.schema_version,
|
|
1002
|
+
"created_at": moment,
|
|
1003
|
+
"kind": kind,
|
|
1004
|
+
"counts": {
|
|
1005
|
+
"documents": counts.documents,
|
|
1006
|
+
"jobs": counts.jobs,
|
|
1007
|
+
"segments": counts.segments,
|
|
1008
|
+
"results": counts.results,
|
|
1009
|
+
"audio_files": counts.audio_files,
|
|
1010
|
+
},
|
|
1011
|
+
"item_count": counts.items,
|
|
1012
|
+
"uncompressed_bytes": uncompressed,
|
|
1013
|
+
"retention_bytes": plan.text_bytes + sum(size for _row, _d, size in good_results),
|
|
1014
|
+
"members": members,
|
|
1015
|
+
"disclosure": DISCLOSURE,
|
|
1016
|
+
"excluded": list(EXCLUDED_FROM_BACKUP),
|
|
1017
|
+
}
|
|
1018
|
+
with _MemberWriter(zf, MANIFEST_NAME) as writer:
|
|
1019
|
+
writer.write(json.dumps(manifest, ensure_ascii=False, indent=1).encode("utf-8"))
|
|
1020
|
+
writer.close()
|
|
1021
|
+
except OSError as exc:
|
|
1022
|
+
raise EchoActError(
|
|
1023
|
+
Code.STORAGE_FULL if getattr(exc, "errno", None) == 28 else Code.BACKUP_INVALID,
|
|
1024
|
+
"The backup could not be written.",
|
|
1025
|
+
detail={"path": redact(temp)},
|
|
1026
|
+
cause=exc,
|
|
1027
|
+
) from exc
|
|
1028
|
+
return counts, tuple(missing), tuple(corrupt), uncompressed
|
|
1029
|
+
|
|
1030
|
+
|
|
1031
|
+
def _copy_into(
|
|
1032
|
+
zf: zipfile.ZipFile, name: str, source: Path, cancel: CancelToken | None
|
|
1033
|
+
) -> tuple[str | None, int]:
|
|
1034
|
+
"""Stream one WAV in, hashing it. ``None`` means the file is gone."""
|
|
1035
|
+
try:
|
|
1036
|
+
handle = source.open("rb")
|
|
1037
|
+
except FileNotFoundError:
|
|
1038
|
+
return None, 0
|
|
1039
|
+
except OSError as exc:
|
|
1040
|
+
raise EchoActError(
|
|
1041
|
+
Code.BACKUP_INVALID,
|
|
1042
|
+
"An audio file could not be read for the backup.",
|
|
1043
|
+
detail={"path": redact(source)},
|
|
1044
|
+
cause=exc,
|
|
1045
|
+
) from exc
|
|
1046
|
+
with handle, _MemberWriter(zf, name, force_zip64=True) as writer:
|
|
1047
|
+
while True:
|
|
1048
|
+
if _cancelled(cancel):
|
|
1049
|
+
writer.close()
|
|
1050
|
+
return None, 0
|
|
1051
|
+
chunk = handle.read(_CHUNK_BYTES)
|
|
1052
|
+
if not chunk:
|
|
1053
|
+
break
|
|
1054
|
+
writer.write(chunk)
|
|
1055
|
+
return writer.close()
|
|
1056
|
+
|
|
1057
|
+
|
|
1058
|
+
def _document_records(snap: _Snapshot, ids_: Sequence[str]) -> Iterator[dict[str, Any]]:
|
|
1059
|
+
for chunk in _chunks(list(ids_)):
|
|
1060
|
+
marks = ",".join("?" * len(chunk))
|
|
1061
|
+
for row in snap.query(
|
|
1062
|
+
"SELECT document_id, title, body, created_at, modified_at, version"
|
|
1063
|
+
f" FROM documents WHERE document_id IN ({marks}) ORDER BY created_at",
|
|
1064
|
+
tuple(chunk),
|
|
1065
|
+
):
|
|
1066
|
+
yield dict(row)
|
|
1067
|
+
|
|
1068
|
+
|
|
1069
|
+
def _job_records(snap: _Snapshot, ids_: Sequence[str]) -> Iterator[dict[str, Any]]:
|
|
1070
|
+
for chunk in _chunks(list(ids_)):
|
|
1071
|
+
marks = ",".join("?" * len(chunk))
|
|
1072
|
+
for row in snap.query(
|
|
1073
|
+
"SELECT job_id, kind, request_path, state, retention, source_text, model_id,"
|
|
1074
|
+
" settings_json, budget_json, created_at, started_at, ended_at, error_code,"
|
|
1075
|
+
" error_message, generated_segments, total_segments"
|
|
1076
|
+
f" FROM jobs WHERE job_id IN ({marks}) ORDER BY created_at",
|
|
1077
|
+
tuple(chunk),
|
|
1078
|
+
):
|
|
1079
|
+
# owner_client_id, client_label and idempotency_key are absent by
|
|
1080
|
+
# design: F-44 gives restored data to the GUI owner and restores no
|
|
1081
|
+
# external client's permissions, and 4.2 keeps re-request records
|
|
1082
|
+
# per client with their own expiry.
|
|
1083
|
+
yield dict(row)
|
|
1084
|
+
|
|
1085
|
+
|
|
1086
|
+
def _segment_records(snap: _Snapshot, ids_: Sequence[str]) -> Iterator[dict[str, Any]]:
|
|
1087
|
+
for chunk in _chunks(list(ids_)):
|
|
1088
|
+
marks = ",".join("?" * len(chunk))
|
|
1089
|
+
for row in snap.query(
|
|
1090
|
+
"SELECT segment_id, job_id, seq, source_start_codepoint_inclusive,"
|
|
1091
|
+
" source_end_codepoint_exclusive, spoken_text, language, audio_start_ms,"
|
|
1092
|
+
" audio_end_ms, trailing_silence_ms, frame_count, ready"
|
|
1093
|
+
f" FROM segments WHERE job_id IN ({marks}) ORDER BY job_id, seq",
|
|
1094
|
+
tuple(chunk),
|
|
1095
|
+
):
|
|
1096
|
+
# audio_path is not exported: per-segment audio lives in the
|
|
1097
|
+
# scratch tree that N-02 clears on relaunch, so it is never part
|
|
1098
|
+
# of a durable bundle. The full result WAV carries the audio.
|
|
1099
|
+
yield dict(row)
|
|
1100
|
+
|
|
1101
|
+
|
|
1102
|
+
def _result_records(
|
|
1103
|
+
rows: Iterable[tuple[sqlite3.Row, str, int]],
|
|
1104
|
+
) -> Iterator[dict[str, Any]]:
|
|
1105
|
+
for row, digest, size in rows:
|
|
1106
|
+
payload = dict(row)
|
|
1107
|
+
# The path inside the bundle is derived, never stored: a restore picks
|
|
1108
|
+
# its own destination and must not be steered by a recorded path.
|
|
1109
|
+
payload.pop("relative_path", None)
|
|
1110
|
+
payload["digest"] = digest
|
|
1111
|
+
payload["byte_size"] = size
|
|
1112
|
+
payload["audio_member"] = f"{AUDIO_PREFIX}{row['result_id']}.wav"
|
|
1113
|
+
yield payload
|
|
1114
|
+
|
|
1115
|
+
|
|
1116
|
+
def _emit(
|
|
1117
|
+
cb: ProgressCallback | None,
|
|
1118
|
+
phase: BackupPhase | RestorePhase,
|
|
1119
|
+
item: str,
|
|
1120
|
+
items_done: int,
|
|
1121
|
+
items_total: int,
|
|
1122
|
+
bytes_done: int,
|
|
1123
|
+
bytes_total: int,
|
|
1124
|
+
) -> BackupProgress:
|
|
1125
|
+
tick = BackupProgress(
|
|
1126
|
+
phase=phase,
|
|
1127
|
+
item=item,
|
|
1128
|
+
items_done=items_done,
|
|
1129
|
+
items_total=items_total,
|
|
1130
|
+
bytes_done=bytes_done,
|
|
1131
|
+
bytes_total=bytes_total,
|
|
1132
|
+
)
|
|
1133
|
+
if cb is not None:
|
|
1134
|
+
cb(tick)
|
|
1135
|
+
return tick
|
|
1136
|
+
|
|
1137
|
+
|
|
1138
|
+
def _discard(path: Path) -> None:
|
|
1139
|
+
with suppress(OSError):
|
|
1140
|
+
path.unlink()
|
|
1141
|
+
|
|
1142
|
+
|
|
1143
|
+
# ======================================================================
|
|
1144
|
+
# Reading a bundle: every check happens before anything is applied (N-27)
|
|
1145
|
+
# ======================================================================
|
|
1146
|
+
|
|
1147
|
+
|
|
1148
|
+
def _reject(message: str, **detail: Any) -> EchoActError:
|
|
1149
|
+
return EchoActError(Code.BACKUP_INVALID, message, detail=detail)
|
|
1150
|
+
|
|
1151
|
+
|
|
1152
|
+
def _check_member_name(name: str) -> None:
|
|
1153
|
+
"""Refuse any name that could address a file outside the restore target.
|
|
1154
|
+
|
|
1155
|
+
Each clause is a real attack and not a restatement of the next: a relative
|
|
1156
|
+
escape, an absolute POSIX path, a Windows drive-qualified path, a UNC
|
|
1157
|
+
share, a backslash separator that only Windows honours, and a name that
|
|
1158
|
+
normalises out of the tree. The allow-list at the end would reject all of
|
|
1159
|
+
them on its own; they are checked separately so that a bad archive is
|
|
1160
|
+
reported for the reason it is bad, and so that widening the allow-list
|
|
1161
|
+
later cannot quietly widen the escape surface too.
|
|
1162
|
+
"""
|
|
1163
|
+
if not name or "\x00" in name:
|
|
1164
|
+
raise _reject("The backup contains an unnamed entry.")
|
|
1165
|
+
if "\\" in name:
|
|
1166
|
+
raise _reject("The backup contains a Windows path separator.", entry=name[:80])
|
|
1167
|
+
if name.startswith("/"):
|
|
1168
|
+
raise _reject("The backup contains an absolute path.", entry=name[:80])
|
|
1169
|
+
if _DRIVE_LETTER.match(name):
|
|
1170
|
+
raise _reject("The backup contains a drive-qualified path.", entry=name[:80])
|
|
1171
|
+
parts = name.split("/")
|
|
1172
|
+
if any(part in ("", ".", "..") for part in parts):
|
|
1173
|
+
raise _reject("The backup contains a relative path escape.", entry=name[:80])
|
|
1174
|
+
if PurePosixPath(name).as_posix() != name:
|
|
1175
|
+
raise _reject("The backup contains a path that does not normalise.", entry=name[:80])
|
|
1176
|
+
if name == MANIFEST_NAME or name in _TABLE_MEMBERS:
|
|
1177
|
+
return
|
|
1178
|
+
if name.startswith(AUDIO_PREFIX) and _SAFE_AUDIO_NAME.match(name[len(AUDIO_PREFIX) :]):
|
|
1179
|
+
return
|
|
1180
|
+
raise _reject("The backup contains an unexpected entry.", entry=name[:80])
|
|
1181
|
+
|
|
1182
|
+
|
|
1183
|
+
def _check_member_kind(info: zipfile.ZipInfo) -> None:
|
|
1184
|
+
"""Refuse anything that is not a plain file.
|
|
1185
|
+
|
|
1186
|
+
A zip entry carries a unix mode in the high half of ``external_attr``; a
|
|
1187
|
+
symlink entry is the classic way to make an extractor write through to a
|
|
1188
|
+
path the archive never names. Directory entries are refused too: this
|
|
1189
|
+
format writes none, and creating one is not something a restore needs.
|
|
1190
|
+
|
|
1191
|
+
Only the file-type bits are consulted. An entry written on a system that
|
|
1192
|
+
records permissions but no type -- which includes Python's own writer, so
|
|
1193
|
+
it includes every archive this module produces -- leaves them zero, and
|
|
1194
|
+
reading that as "not a regular file" would reject our own bundles.
|
|
1195
|
+
"""
|
|
1196
|
+
if info.is_dir():
|
|
1197
|
+
raise _reject("The backup contains a directory entry.", entry=info.filename[:80])
|
|
1198
|
+
kind_bits = (info.external_attr >> 16) & 0o170000
|
|
1199
|
+
if kind_bits and kind_bits != stat.S_IFREG:
|
|
1200
|
+
kind = "symbolic link" if kind_bits == stat.S_IFLNK else "special file"
|
|
1201
|
+
raise _reject(f"The backup contains a {kind}.", entry=info.filename[:80])
|
|
1202
|
+
|
|
1203
|
+
|
|
1204
|
+
@contextmanager
|
|
1205
|
+
def _member_stream(zf: zipfile.ZipFile, info: zipfile.ZipInfo) -> Iterator[Any]:
|
|
1206
|
+
"""Open one member, or reject the archive -- with no third outcome.
|
|
1207
|
+
|
|
1208
|
+
``zipfile`` answers a hostile header with a builtin, not with
|
|
1209
|
+
``BadZipFile``: ``RuntimeError`` for an entry whose encryption flag is
|
|
1210
|
+
set, ``NotImplementedError`` for a compression method this build does not
|
|
1211
|
+
have, and whatever the decompressor for a *supported* method raises for a
|
|
1212
|
+
stream that is not one. Catching those by name is catching the ones we
|
|
1213
|
+
happened to think of, and every one of them escaping as a builtin means a
|
|
1214
|
+
caller that handles :class:`EchoActError` -- the chooser, the service --
|
|
1215
|
+
sees an unhandled crash instead of "this backup is corrupt" (rule 3,
|
|
1216
|
+
N-27). Every way an archive can fail to decode means one thing here, so
|
|
1217
|
+
the whole decode is funnelled through this one place and answered with one
|
|
1218
|
+
code. ``EchoActError`` is let through: our own rejections are already the
|
|
1219
|
+
answer, and re-wrapping one would lose the code that says why.
|
|
1220
|
+
"""
|
|
1221
|
+
try:
|
|
1222
|
+
raw = zf.open(info, "r")
|
|
1223
|
+
except EchoActError:
|
|
1224
|
+
raise
|
|
1225
|
+
except Exception as exc:
|
|
1226
|
+
raise _reject(
|
|
1227
|
+
"The backup contains an entry that cannot be read.", entry=info.filename[:80]
|
|
1228
|
+
) from exc
|
|
1229
|
+
try:
|
|
1230
|
+
yield raw
|
|
1231
|
+
finally:
|
|
1232
|
+
with suppress(Exception):
|
|
1233
|
+
raw.close()
|
|
1234
|
+
|
|
1235
|
+
|
|
1236
|
+
def _bounded_read(zf: zipfile.ZipFile, info: zipfile.ZipInfo, limit: int) -> Iterator[bytes]:
|
|
1237
|
+
"""Stream a member, refusing to keep going past ``limit``.
|
|
1238
|
+
|
|
1239
|
+
The declared size in the header is not evidence. 4.1 caps what a restore
|
|
1240
|
+
may decompress, so the cap is enforced against bytes actually produced.
|
|
1241
|
+
"""
|
|
1242
|
+
read = 0
|
|
1243
|
+
with _member_stream(zf, info) as raw:
|
|
1244
|
+
while True:
|
|
1245
|
+
try:
|
|
1246
|
+
chunk = raw.read(_CHUNK_BYTES)
|
|
1247
|
+
except EchoActError:
|
|
1248
|
+
raise
|
|
1249
|
+
except Exception as exc:
|
|
1250
|
+
raise _reject(
|
|
1251
|
+
"The backup contains an entry that cannot be read.", entry=info.filename[:80]
|
|
1252
|
+
) from exc
|
|
1253
|
+
if not chunk:
|
|
1254
|
+
return
|
|
1255
|
+
read += len(chunk)
|
|
1256
|
+
if read > limit:
|
|
1257
|
+
raise EchoActError(
|
|
1258
|
+
Code.BACKUP_TOO_LARGE,
|
|
1259
|
+
"The backup decompresses to more than the restore limit allows.",
|
|
1260
|
+
detail={"limit_bytes": limit, "entry": info.filename[:80]},
|
|
1261
|
+
)
|
|
1262
|
+
yield chunk
|
|
1263
|
+
|
|
1264
|
+
|
|
1265
|
+
def _member_digest(zf: zipfile.ZipFile, info: zipfile.ZipInfo, limit: int) -> tuple[str, int, int]:
|
|
1266
|
+
"""The member's digest, its true size, and how many lines it really holds.
|
|
1267
|
+
|
|
1268
|
+
The line count is what finally binds a table's declared row count to its
|
|
1269
|
+
contents. It counts newlines instead of parsing, and rounds an
|
|
1270
|
+
unterminated final line up, so it can only ever over-count against
|
|
1271
|
+
:func:`_iter_records` -- and an over-count is a rejected archive, never a
|
|
1272
|
+
row applied past 4.1's ceiling.
|
|
1273
|
+
"""
|
|
1274
|
+
h = hashlib.sha256()
|
|
1275
|
+
size = 0
|
|
1276
|
+
lines = 0
|
|
1277
|
+
terminated = True
|
|
1278
|
+
for chunk in _bounded_read(zf, info, limit):
|
|
1279
|
+
h.update(chunk)
|
|
1280
|
+
size += len(chunk)
|
|
1281
|
+
lines += chunk.count(b"\n")
|
|
1282
|
+
terminated = chunk.endswith(b"\n")
|
|
1283
|
+
if size and not terminated:
|
|
1284
|
+
lines += 1
|
|
1285
|
+
return h.hexdigest(), size, lines
|
|
1286
|
+
|
|
1287
|
+
|
|
1288
|
+
def _iter_records(
|
|
1289
|
+
zf: zipfile.ZipFile, info: zipfile.ZipInfo, *, limit: int
|
|
1290
|
+
) -> Iterator[dict[str, Any]]:
|
|
1291
|
+
buffer = b""
|
|
1292
|
+
for chunk in _bounded_read(zf, info, limit):
|
|
1293
|
+
buffer += chunk
|
|
1294
|
+
while True:
|
|
1295
|
+
newline = buffer.find(b"\n")
|
|
1296
|
+
if newline < 0:
|
|
1297
|
+
break
|
|
1298
|
+
line, buffer = buffer[:newline], buffer[newline + 1 :]
|
|
1299
|
+
if line.strip():
|
|
1300
|
+
yield _decode_record(line, info.filename)
|
|
1301
|
+
if len(buffer) > _MAX_RECORD_BYTES:
|
|
1302
|
+
raise _reject("The backup contains an unterminated record.", entry=info.filename[:80])
|
|
1303
|
+
if buffer.strip():
|
|
1304
|
+
yield _decode_record(buffer, info.filename)
|
|
1305
|
+
|
|
1306
|
+
|
|
1307
|
+
def _decode_record(line: bytes, member: str) -> dict[str, Any]:
|
|
1308
|
+
try:
|
|
1309
|
+
return _as_object(json.loads(line.decode("utf-8")), f"{member} record")
|
|
1310
|
+
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
|
|
1311
|
+
raise _reject("The backup contains a record that is not valid JSON.", entry=member) from exc
|
|
1312
|
+
|
|
1313
|
+
|
|
1314
|
+
def inspect_backup(
|
|
1315
|
+
path: str | os.PathLike[str],
|
|
1316
|
+
*,
|
|
1317
|
+
max_bytes: int = BACKUP_RESTORE_MAX_BYTES,
|
|
1318
|
+
max_items: int = BACKUP_RESTORE_MAX_ITEMS,
|
|
1319
|
+
) -> BackupInspection:
|
|
1320
|
+
"""Structure, compatibility, and 4.1's limits -- without reading a member.
|
|
1321
|
+
|
|
1322
|
+
Separate from :func:`verify_backup` so a chooser dialog can describe a file
|
|
1323
|
+
(F-44's disclosure, its size, its age) without paying to hash it.
|
|
1324
|
+
"""
|
|
1325
|
+
return _inspect(Path(path), max_bytes=max_bytes, max_items=max_items, deep=False, digest=False)
|
|
1326
|
+
|
|
1327
|
+
|
|
1328
|
+
def verify_backup(
|
|
1329
|
+
path: str | os.PathLike[str],
|
|
1330
|
+
*,
|
|
1331
|
+
max_bytes: int = BACKUP_RESTORE_MAX_BYTES,
|
|
1332
|
+
max_items: int = BACKUP_RESTORE_MAX_ITEMS,
|
|
1333
|
+
deep: bool = False,
|
|
1334
|
+
cancel: CancelToken | None = None,
|
|
1335
|
+
) -> BackupInspection:
|
|
1336
|
+
"""N-27's "integrity and compatibility are verified before restore".
|
|
1337
|
+
|
|
1338
|
+
Shallow verification reads and hashes the table members; ``deep`` also
|
|
1339
|
+
reads every audio member and checks it against 4.2's stored digest, which
|
|
1340
|
+
is what F-74 means by a *sound* backup before it rotates older ones away.
|
|
1341
|
+
"""
|
|
1342
|
+
return _inspect(
|
|
1343
|
+
Path(path), max_bytes=max_bytes, max_items=max_items, deep=deep, digest=True, cancel=cancel
|
|
1344
|
+
)
|
|
1345
|
+
|
|
1346
|
+
|
|
1347
|
+
def _inspect(
|
|
1348
|
+
path: Path,
|
|
1349
|
+
*,
|
|
1350
|
+
max_bytes: int,
|
|
1351
|
+
max_items: int,
|
|
1352
|
+
deep: bool,
|
|
1353
|
+
digest: bool,
|
|
1354
|
+
cancel: CancelToken | None = None,
|
|
1355
|
+
) -> BackupInspection:
|
|
1356
|
+
if not path.is_file():
|
|
1357
|
+
raise EchoActError(
|
|
1358
|
+
Code.FILE_NOT_FOUND, "That backup file does not exist.", detail={"path": redact(path)}
|
|
1359
|
+
)
|
|
1360
|
+
compressed = path.stat().st_size
|
|
1361
|
+
try:
|
|
1362
|
+
with zipfile.ZipFile(path, "r") as zf:
|
|
1363
|
+
infos = zf.infolist()
|
|
1364
|
+
names = [i.filename for i in infos]
|
|
1365
|
+
if len(set(names)) != len(names):
|
|
1366
|
+
raise _reject("The backup names the same entry twice.")
|
|
1367
|
+
if len(infos) > max_items:
|
|
1368
|
+
raise EchoActError(
|
|
1369
|
+
Code.BACKUP_TOO_LARGE,
|
|
1370
|
+
"The backup holds more entries than a restore may apply.",
|
|
1371
|
+
detail={"entries": len(infos), "limit": max_items},
|
|
1372
|
+
)
|
|
1373
|
+
declared = 0
|
|
1374
|
+
for info in infos:
|
|
1375
|
+
_check_member_name(info.filename)
|
|
1376
|
+
_check_member_kind(info)
|
|
1377
|
+
declared += max(0, int(info.file_size))
|
|
1378
|
+
if declared > max_bytes:
|
|
1379
|
+
raise EchoActError(
|
|
1380
|
+
Code.BACKUP_TOO_LARGE,
|
|
1381
|
+
"The backup decompresses to more than the restore limit allows.",
|
|
1382
|
+
detail={"uncompressed_bytes": declared, "limit_bytes": max_bytes},
|
|
1383
|
+
)
|
|
1384
|
+
manifest = _read_manifest(zf, names)
|
|
1385
|
+
inspection = _check_manifest(
|
|
1386
|
+
path, manifest, zf, compressed, max_bytes=max_bytes, max_items=max_items
|
|
1387
|
+
)
|
|
1388
|
+
if digest:
|
|
1389
|
+
_verify_members(zf, manifest, max_bytes=max_bytes, deep=deep, cancel=cancel)
|
|
1390
|
+
except EchoActError:
|
|
1391
|
+
# Already the answer, and carrying the code that says why.
|
|
1392
|
+
raise
|
|
1393
|
+
except zipfile.BadZipFile as exc:
|
|
1394
|
+
raise _reject("The backup is not a readable archive.", path=redact(path)) from exc
|
|
1395
|
+
except Exception as exc:
|
|
1396
|
+
# Opening the directory of a crafted archive fails in as many ways as
|
|
1397
|
+
# reading a member does; the reasoning in _member_stream applies here.
|
|
1398
|
+
raise _reject("The backup could not be read.", path=redact(path)) from exc
|
|
1399
|
+
return replace(inspection, deep=deep and digest)
|
|
1400
|
+
|
|
1401
|
+
|
|
1402
|
+
def _read_manifest(zf: zipfile.ZipFile, names: Sequence[str]) -> dict[str, Any]:
|
|
1403
|
+
if MANIFEST_NAME not in names:
|
|
1404
|
+
raise _reject("The backup has no manifest.")
|
|
1405
|
+
info = zf.getinfo(MANIFEST_NAME)
|
|
1406
|
+
if info.file_size > _MAX_MANIFEST_BYTES:
|
|
1407
|
+
raise _reject("The backup's manifest is implausibly large.")
|
|
1408
|
+
raw = b"".join(_bounded_read(zf, info, _MAX_MANIFEST_BYTES))
|
|
1409
|
+
try:
|
|
1410
|
+
return _as_object(json.loads(raw.decode("utf-8")), "manifest")
|
|
1411
|
+
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
|
|
1412
|
+
raise _reject("The backup's manifest is not valid JSON.") from exc
|
|
1413
|
+
|
|
1414
|
+
|
|
1415
|
+
def _check_manifest(
|
|
1416
|
+
path: Path,
|
|
1417
|
+
manifest: dict[str, Any],
|
|
1418
|
+
zf: zipfile.ZipFile,
|
|
1419
|
+
compressed: int,
|
|
1420
|
+
*,
|
|
1421
|
+
max_bytes: int,
|
|
1422
|
+
max_items: int,
|
|
1423
|
+
) -> BackupInspection:
|
|
1424
|
+
fmt = manifest.get("format")
|
|
1425
|
+
if not isinstance(fmt, int) or fmt != BACKUP_FORMAT_VERSION:
|
|
1426
|
+
raise EchoActError(
|
|
1427
|
+
Code.BACKUP_INCOMPATIBLE,
|
|
1428
|
+
"That backup was written in a format this version does not read.",
|
|
1429
|
+
detail={"found": fmt, "supported": BACKUP_FORMAT_VERSION},
|
|
1430
|
+
)
|
|
1431
|
+
schema = manifest.get("schema_version")
|
|
1432
|
+
if not isinstance(schema, int) or schema > SUPPORTED_SCHEMA_VERSION:
|
|
1433
|
+
# N-15: an older build must not touch a newer format at all.
|
|
1434
|
+
raise EchoActError(
|
|
1435
|
+
Code.BACKUP_INCOMPATIBLE,
|
|
1436
|
+
"That backup holds a newer data format than this version understands.",
|
|
1437
|
+
detail={"found": schema, "supported": SUPPORTED_SCHEMA_VERSION},
|
|
1438
|
+
)
|
|
1439
|
+
counts_raw = _as_object(manifest.get("counts", {}), "counts")
|
|
1440
|
+
counts = BackupCounts(
|
|
1441
|
+
documents=_non_negative(counts_raw.get("documents", 0), "documents"),
|
|
1442
|
+
jobs=_non_negative(counts_raw.get("jobs", 0), "jobs"),
|
|
1443
|
+
segments=_non_negative(counts_raw.get("segments", 0), "segments"),
|
|
1444
|
+
results=_non_negative(counts_raw.get("results", 0), "results"),
|
|
1445
|
+
audio_files=_non_negative(counts_raw.get("audio_files", 0), "audio_files"),
|
|
1446
|
+
)
|
|
1447
|
+
declared_items = _non_negative(manifest.get("item_count", counts.items), "item_count")
|
|
1448
|
+
if max(declared_items, counts.items) > max_items:
|
|
1449
|
+
raise EchoActError(
|
|
1450
|
+
Code.BACKUP_TOO_LARGE,
|
|
1451
|
+
"The backup holds more items than a restore may apply.",
|
|
1452
|
+
detail={"items": max(declared_items, counts.items), "limit": max_items},
|
|
1453
|
+
)
|
|
1454
|
+
uncompressed = _non_negative(manifest.get("uncompressed_bytes", 0), "uncompressed_bytes")
|
|
1455
|
+
retention = _non_negative(manifest.get("retention_bytes", 0), "retention_bytes")
|
|
1456
|
+
if uncompressed > max_bytes:
|
|
1457
|
+
raise EchoActError(
|
|
1458
|
+
Code.BACKUP_TOO_LARGE,
|
|
1459
|
+
"The backup decompresses to more than the restore limit allows.",
|
|
1460
|
+
detail={"uncompressed_bytes": uncompressed, "limit_bytes": max_bytes},
|
|
1461
|
+
)
|
|
1462
|
+
members = _as_object(manifest.get("members", {}), "member list")
|
|
1463
|
+
present = {i.filename for i in zf.infolist()} - {MANIFEST_NAME}
|
|
1464
|
+
listed = set(members)
|
|
1465
|
+
if present != listed:
|
|
1466
|
+
raise _reject(
|
|
1467
|
+
"The backup's manifest does not match its contents.",
|
|
1468
|
+
unlisted=sorted(present - listed)[:5],
|
|
1469
|
+
missing=sorted(listed - present)[:5],
|
|
1470
|
+
)
|
|
1471
|
+
audio_members = {n for n in present if n.startswith(AUDIO_PREFIX)}
|
|
1472
|
+
if len(audio_members) != counts.audio_files:
|
|
1473
|
+
raise _reject(
|
|
1474
|
+
"The backup declares a different number of audio files than it holds.",
|
|
1475
|
+
declared=counts.audio_files,
|
|
1476
|
+
found=len(audio_members),
|
|
1477
|
+
)
|
|
1478
|
+
_check_declarations(zf, members, counts, uncompressed, retention)
|
|
1479
|
+
return BackupInspection(
|
|
1480
|
+
path=path,
|
|
1481
|
+
format_version=fmt,
|
|
1482
|
+
app_version=str(manifest.get("app_version", "")),
|
|
1483
|
+
schema_version=schema,
|
|
1484
|
+
created_at=float(manifest.get("created_at", 0.0) or 0.0),
|
|
1485
|
+
kind=str(manifest.get("kind", "manual")),
|
|
1486
|
+
counts=counts,
|
|
1487
|
+
uncompressed_bytes=uncompressed,
|
|
1488
|
+
compressed_bytes=compressed,
|
|
1489
|
+
retention_bytes=retention,
|
|
1490
|
+
disclosure=str(manifest.get("disclosure", DISCLOSURE)),
|
|
1491
|
+
deep=False,
|
|
1492
|
+
)
|
|
1493
|
+
|
|
1494
|
+
|
|
1495
|
+
def _check_declarations(
|
|
1496
|
+
zf: zipfile.ZipFile,
|
|
1497
|
+
members: dict[str, Any],
|
|
1498
|
+
counts: BackupCounts,
|
|
1499
|
+
uncompressed: int,
|
|
1500
|
+
retention: int,
|
|
1501
|
+
) -> None:
|
|
1502
|
+
"""Tie every figure 4.1 caps to something the archive cannot simply assert.
|
|
1503
|
+
|
|
1504
|
+
The manifest is the one part of a bundle a forger writes freely, and until
|
|
1505
|
+
this existed it was the *only* source for the item count, the decompressed
|
|
1506
|
+
size and the retention cost: an archive holding a quarter of a million
|
|
1507
|
+
documents could declare one item and one byte, pass verification, and be
|
|
1508
|
+
applied in full -- with the free-space and retention guards run against the
|
|
1509
|
+
lie and 4.1's 100,000-item ceiling never touching a row that was actually
|
|
1510
|
+
inserted. Only the audio file count was bound to reality.
|
|
1511
|
+
|
|
1512
|
+
So each per-member figure has to match the size the archive's own directory
|
|
1513
|
+
records, which is also the hard ceiling on how much ``zipfile`` will ever
|
|
1514
|
+
hand back for that entry; each aggregate has to be at least the sum of its
|
|
1515
|
+
parts; and each table's declared row count has to match the ``records``
|
|
1516
|
+
figure that :func:`_verify_members` then holds to the lines that really
|
|
1517
|
+
come out. Understatement is what is refused, in both directions: a
|
|
1518
|
+
manifest may not make a bundle look smaller than it is.
|
|
1519
|
+
"""
|
|
1520
|
+
total = 0
|
|
1521
|
+
audio_bytes = 0
|
|
1522
|
+
for name, raw_meta in members.items():
|
|
1523
|
+
meta = _as_object(raw_meta, "member entry")
|
|
1524
|
+
size = _non_negative(meta.get("bytes"), f"recorded size for {name[:80]}")
|
|
1525
|
+
if size != zf.getinfo(name).file_size:
|
|
1526
|
+
raise _reject("A backup entry is not the size the manifest records.", entry=name[:80])
|
|
1527
|
+
total += size
|
|
1528
|
+
if name.startswith(AUDIO_PREFIX):
|
|
1529
|
+
audio_bytes += size
|
|
1530
|
+
field_name = _COUNTED_MEMBERS.get(name)
|
|
1531
|
+
if field_name is None:
|
|
1532
|
+
continue
|
|
1533
|
+
rows = _non_negative(meta.get("records", 0), f"record count for {field_name}")
|
|
1534
|
+
if rows != getattr(counts, field_name):
|
|
1535
|
+
raise _reject(
|
|
1536
|
+
"The backup declares a different number of rows than its member holds.",
|
|
1537
|
+
entry=name[:80],
|
|
1538
|
+
declared=getattr(counts, field_name),
|
|
1539
|
+
listed=rows,
|
|
1540
|
+
)
|
|
1541
|
+
for name, field_name in _COUNTED_MEMBERS.items():
|
|
1542
|
+
if name not in members and getattr(counts, field_name):
|
|
1543
|
+
raise _reject("The backup declares rows in a member it does not contain.", entry=name)
|
|
1544
|
+
if uncompressed < total:
|
|
1545
|
+
raise _reject(
|
|
1546
|
+
"The backup understates what it decompresses to.", declared=uncompressed, found=total
|
|
1547
|
+
)
|
|
1548
|
+
if retention < audio_bytes:
|
|
1549
|
+
# Audio is stored verbatim, so its retention cost is exactly these
|
|
1550
|
+
# bytes; the text on top of it is what the manifest may still add.
|
|
1551
|
+
raise _reject(
|
|
1552
|
+
"The backup understates what restoring it would occupy.",
|
|
1553
|
+
declared=retention,
|
|
1554
|
+
found=audio_bytes,
|
|
1555
|
+
)
|
|
1556
|
+
|
|
1557
|
+
|
|
1558
|
+
def _non_negative(value: Any, what: str) -> int:
|
|
1559
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
|
1560
|
+
raise _reject(f"The backup's {what} is not a count.", value=str(value)[:40])
|
|
1561
|
+
return value
|
|
1562
|
+
|
|
1563
|
+
|
|
1564
|
+
def _verify_members(
|
|
1565
|
+
zf: zipfile.ZipFile,
|
|
1566
|
+
manifest: dict[str, Any],
|
|
1567
|
+
*,
|
|
1568
|
+
max_bytes: int,
|
|
1569
|
+
deep: bool,
|
|
1570
|
+
cancel: CancelToken | None,
|
|
1571
|
+
) -> None:
|
|
1572
|
+
members = _as_object(manifest.get("members", {}), "member list")
|
|
1573
|
+
for name, raw_meta in members.items():
|
|
1574
|
+
if _cancelled(cancel):
|
|
1575
|
+
return
|
|
1576
|
+
if name.startswith(AUDIO_PREFIX) and not deep:
|
|
1577
|
+
continue
|
|
1578
|
+
meta = _as_object(raw_meta, "member entry")
|
|
1579
|
+
found, size, lines = _member_digest(zf, zf.getinfo(name), max_bytes)
|
|
1580
|
+
if found != meta.get("sha256"):
|
|
1581
|
+
raise _reject("The backup's contents do not match its manifest.", entry=name[:80])
|
|
1582
|
+
recorded = meta.get("bytes")
|
|
1583
|
+
if isinstance(recorded, int) and recorded != size:
|
|
1584
|
+
raise _reject("A backup entry is not the size the manifest records.", entry=name[:80])
|
|
1585
|
+
if name in _COUNTED_MEMBERS and lines != meta.get("records"):
|
|
1586
|
+
# A digest only says the member is the one the manifest's author
|
|
1587
|
+
# meant to ship. This says the manifest's row count -- which
|
|
1588
|
+
# _check_manifest has already tied to the counts 4.1 caps -- is
|
|
1589
|
+
# the number of rows a restore would actually insert.
|
|
1590
|
+
raise _reject(
|
|
1591
|
+
"A backup entry does not hold the number of rows the manifest records.",
|
|
1592
|
+
entry=name[:80],
|
|
1593
|
+
recorded=meta.get("records"),
|
|
1594
|
+
found=lines,
|
|
1595
|
+
)
|
|
1596
|
+
|
|
1597
|
+
|
|
1598
|
+
# ======================================================================
|
|
1599
|
+
# Restore (F-44, N-15, N-27, 5.3)
|
|
1600
|
+
# ======================================================================
|
|
1601
|
+
|
|
1602
|
+
|
|
1603
|
+
class _Cancelled(Exception):
|
|
1604
|
+
"""Raised inside the apply transaction so SQLite rolls it back.
|
|
1605
|
+
|
|
1606
|
+
Returning a "cancelled" outcome from inside ``store.transaction()`` would
|
|
1607
|
+
leave the context manager to commit everything written so far, which is the
|
|
1608
|
+
half-applied restore N-15 forbids. Cancellation has to unwind.
|
|
1609
|
+
"""
|
|
1610
|
+
|
|
1611
|
+
|
|
1612
|
+
@dataclass(slots=True)
|
|
1613
|
+
class _Applied:
|
|
1614
|
+
"""Files written outside the transaction, so a rollback can undo them."""
|
|
1615
|
+
|
|
1616
|
+
paths: list[Path] = field(default_factory=list)
|
|
1617
|
+
|
|
1618
|
+
def undo(self) -> None:
|
|
1619
|
+
for path in self.paths:
|
|
1620
|
+
_discard(path)
|
|
1621
|
+
self.paths.clear()
|
|
1622
|
+
|
|
1623
|
+
|
|
1624
|
+
@dataclass(slots=True)
|
|
1625
|
+
class _ItemBudget:
|
|
1626
|
+
"""4.1's item ceiling, counted against rows that are actually inserted.
|
|
1627
|
+
|
|
1628
|
+
Verification refuses an over-large archive before a row is applied, and
|
|
1629
|
+
that is where this is normally decided. This counts what really goes in,
|
|
1630
|
+
so the ceiling holds against a member whose contents a future check fails
|
|
1631
|
+
to bind to the manifest -- the transaction has not committed while it is
|
|
1632
|
+
counting, so exceeding it is still a refusal and not a partial restore.
|
|
1633
|
+
"""
|
|
1634
|
+
|
|
1635
|
+
limit: int
|
|
1636
|
+
used: int = 0
|
|
1637
|
+
|
|
1638
|
+
def take(self, n: int = 1) -> None:
|
|
1639
|
+
self.used += n
|
|
1640
|
+
if self.used > self.limit:
|
|
1641
|
+
raise EchoActError(
|
|
1642
|
+
Code.BACKUP_TOO_LARGE,
|
|
1643
|
+
"The backup holds more items than a restore may apply.",
|
|
1644
|
+
detail={"items": self.used, "limit": self.limit},
|
|
1645
|
+
)
|
|
1646
|
+
|
|
1647
|
+
|
|
1648
|
+
def restore_backup(
|
|
1649
|
+
store: Store,
|
|
1650
|
+
source: str | os.PathLike[str],
|
|
1651
|
+
*,
|
|
1652
|
+
owner_client_id: str,
|
|
1653
|
+
mode: RestoreMode = RestoreMode.ADD,
|
|
1654
|
+
max_bytes: int = BACKUP_RESTORE_MAX_BYTES,
|
|
1655
|
+
max_items: int = BACKUP_RESTORE_MAX_ITEMS,
|
|
1656
|
+
at: float | None = None,
|
|
1657
|
+
progress: ProgressCallback | None = None,
|
|
1658
|
+
cancel: CancelToken | None = None,
|
|
1659
|
+
free_space: FreeSpace = _disk_free,
|
|
1660
|
+
gate: RestoreGate | None = None,
|
|
1661
|
+
safety_backup: bool = True,
|
|
1662
|
+
) -> RestoreOutcome:
|
|
1663
|
+
"""Apply a bundle, or apply none of it.
|
|
1664
|
+
|
|
1665
|
+
The order is the requirement: verify (N-27), then check what applying would
|
|
1666
|
+
cost against 4.1's decompressed-size, item-count, retention and free-space
|
|
1667
|
+
limits, then -- only for :attr:`RestoreMode.REPLACE` -- secure a recoverable
|
|
1668
|
+
copy of the data about to be overwritten (N-15), and only then write. Rows
|
|
1669
|
+
go in inside one transaction and audio files are tracked so that a failure
|
|
1670
|
+
or a cancellation anywhere leaves the library exactly as it was.
|
|
1671
|
+
|
|
1672
|
+
Two things stand between REPLACE and the empty library that would be the
|
|
1673
|
+
worst possible answer to "restore from a corrupt database". Verification
|
|
1674
|
+
is always deep, whatever it costs to hash the audio a second time: a
|
|
1675
|
+
shallow check leaves audio unread until ``_apply`` extracts it, which on
|
|
1676
|
+
this path is *after* ``_clear_library`` has committed, so a rotted bundle
|
|
1677
|
+
would be discovered only once the library it was to replace no longer
|
|
1678
|
+
existed. And because the clear commits outside the transaction and some
|
|
1679
|
+
failures -- a bad record, a failing disk -- can only be met while applying,
|
|
1680
|
+
:func:`_put_back` restores the copy taken moments before, and the error
|
|
1681
|
+
says what became of the data either way.
|
|
1682
|
+
"""
|
|
1683
|
+
the_gate = gate or RESTORE_GATE
|
|
1684
|
+
path = Path(source)
|
|
1685
|
+
moment = _moment(at)
|
|
1686
|
+
|
|
1687
|
+
with the_gate.hold():
|
|
1688
|
+
tick = _emit(progress, RestorePhase.CHECKING, path.name, 0, 0, 0, 0)
|
|
1689
|
+
the_gate.note(tick)
|
|
1690
|
+
inspection = verify_backup(
|
|
1691
|
+
path, max_bytes=max_bytes, max_items=max_items, deep=True, cancel=cancel
|
|
1692
|
+
)
|
|
1693
|
+
if _cancelled(cancel):
|
|
1694
|
+
return _cancelled_restore(path, mode, owner_client_id, None)
|
|
1695
|
+
|
|
1696
|
+
_require_space(store.audio_root, inspection.uncompressed_bytes, free_space)
|
|
1697
|
+
usage = store.storage_usage(at=moment)
|
|
1698
|
+
# What survives the restore, which for REPLACE is nothing: everything
|
|
1699
|
+
# this figure counts is deleted before a row of the bundle lands.
|
|
1700
|
+
# Adding the incoming bundle to it would refuse a restore because of
|
|
1701
|
+
# data that will not be there, so a library over half its allowance
|
|
1702
|
+
# could not be recovered from its own backup -- the one case REPLACE
|
|
1703
|
+
# exists for (Section 9, F-44).
|
|
1704
|
+
kept = 0 if mode is RestoreMode.REPLACE else usage.total_bytes
|
|
1705
|
+
if kept + inspection.retention_bytes > usage.limit_bytes:
|
|
1706
|
+
raise EchoActError(
|
|
1707
|
+
Code.RETENTION_LIMIT_REACHED,
|
|
1708
|
+
"Restoring this backup would exceed the retention limit.",
|
|
1709
|
+
detail={
|
|
1710
|
+
"used_bytes": kept,
|
|
1711
|
+
"limit_bytes": usage.limit_bytes,
|
|
1712
|
+
"requested_bytes": inspection.retention_bytes,
|
|
1713
|
+
},
|
|
1714
|
+
)
|
|
1715
|
+
|
|
1716
|
+
rescue: BackupOutcome | None = None
|
|
1717
|
+
cleared = False
|
|
1718
|
+
|
|
1719
|
+
def past_recall() -> None:
|
|
1720
|
+
"""``_clear_library`` has deleted something; see :func:`_put_back`."""
|
|
1721
|
+
nonlocal cleared
|
|
1722
|
+
cleared = True
|
|
1723
|
+
|
|
1724
|
+
written = _Applied()
|
|
1725
|
+
try:
|
|
1726
|
+
if mode is RestoreMode.REPLACE:
|
|
1727
|
+
tick = _emit(progress, RestorePhase.PREPARING, "existing data", 0, 0, 0, 0)
|
|
1728
|
+
the_gate.note(tick)
|
|
1729
|
+
rescue = _secure_existing(store, moment, free_space) if safety_backup else None
|
|
1730
|
+
_clear_library(store, past_recall)
|
|
1731
|
+
return _apply(
|
|
1732
|
+
store,
|
|
1733
|
+
path,
|
|
1734
|
+
inspection,
|
|
1735
|
+
owner_client_id=owner_client_id,
|
|
1736
|
+
mode=mode,
|
|
1737
|
+
moment=moment,
|
|
1738
|
+
progress=progress,
|
|
1739
|
+
cancel=cancel,
|
|
1740
|
+
gate=the_gate,
|
|
1741
|
+
written=written,
|
|
1742
|
+
safety=_rescue_path(rescue),
|
|
1743
|
+
max_bytes=max_bytes,
|
|
1744
|
+
max_items=max_items,
|
|
1745
|
+
)
|
|
1746
|
+
except _Cancelled:
|
|
1747
|
+
written.undo()
|
|
1748
|
+
if cleared:
|
|
1749
|
+
_put_back(store, rescue, owner_client_id, moment, the_gate)
|
|
1750
|
+
_emit(progress, RestorePhase.CANCELLED, path.name, 0, inspection.item_count, 0, 0)
|
|
1751
|
+
return _cancelled_restore(path, mode, owner_client_id, _rescue_path(rescue))
|
|
1752
|
+
except EchoActError as exc:
|
|
1753
|
+
written.undo()
|
|
1754
|
+
if not cleared:
|
|
1755
|
+
raise
|
|
1756
|
+
raise _replace_failed(
|
|
1757
|
+
exc,
|
|
1758
|
+
_rescue_path(rescue),
|
|
1759
|
+
_put_back(store, rescue, owner_client_id, moment, the_gate),
|
|
1760
|
+
) from exc
|
|
1761
|
+
except BaseException:
|
|
1762
|
+
written.undo()
|
|
1763
|
+
if cleared:
|
|
1764
|
+
_put_back(store, rescue, owner_client_id, moment, the_gate)
|
|
1765
|
+
raise
|
|
1766
|
+
|
|
1767
|
+
|
|
1768
|
+
def _rescue_path(rescue: BackupOutcome | None) -> Path | None:
|
|
1769
|
+
return None if rescue is None else rescue.path
|
|
1770
|
+
|
|
1771
|
+
|
|
1772
|
+
def _cancelled_restore(
|
|
1773
|
+
path: Path, mode: RestoreMode, owner: str, safety: Path | None
|
|
1774
|
+
) -> RestoreOutcome:
|
|
1775
|
+
return RestoreOutcome(
|
|
1776
|
+
source=path,
|
|
1777
|
+
mode=mode,
|
|
1778
|
+
counts=BackupCounts(),
|
|
1779
|
+
items=(),
|
|
1780
|
+
owner_client_id=owner,
|
|
1781
|
+
safety_backup=safety,
|
|
1782
|
+
cancelled=True,
|
|
1783
|
+
)
|
|
1784
|
+
|
|
1785
|
+
|
|
1786
|
+
def _secure_existing(store: Store, moment: float, free_space: FreeSpace) -> BackupOutcome:
|
|
1787
|
+
"""N-15: the data about to be overwritten is read, bundled, and verified.
|
|
1788
|
+
|
|
1789
|
+
Producing a full bundle rather than copying the database file is what makes
|
|
1790
|
+
this a *validation*: every row is read through one snapshot and every audio
|
|
1791
|
+
file is hashed against 4.2's stored digest on the way in, so a corrupt
|
|
1792
|
+
library is discovered before it is destroyed rather than after.
|
|
1793
|
+
|
|
1794
|
+
The whole outcome is returned, not just the path, because it is also the
|
|
1795
|
+
input to :func:`_put_back`: how many rows and bytes it holds is what lets
|
|
1796
|
+
that read the file back without holding a bundle of the user's own library
|
|
1797
|
+
to 4.1's restore ceiling.
|
|
1798
|
+
"""
|
|
1799
|
+
directory = default_backup_dir()
|
|
1800
|
+
destination = directory / scheduled_backup_name(moment, kind="pre-restore")
|
|
1801
|
+
return create_backup(
|
|
1802
|
+
store,
|
|
1803
|
+
destination,
|
|
1804
|
+
kind="pre_migration",
|
|
1805
|
+
at=moment,
|
|
1806
|
+
free_space=free_space,
|
|
1807
|
+
gate=_NULL_GATE,
|
|
1808
|
+
deep_verify=True,
|
|
1809
|
+
)
|
|
1810
|
+
|
|
1811
|
+
|
|
1812
|
+
def _put_back(
|
|
1813
|
+
store: Store,
|
|
1814
|
+
rescue: BackupOutcome | None,
|
|
1815
|
+
owner_client_id: str,
|
|
1816
|
+
moment: float,
|
|
1817
|
+
gate: RestoreGate,
|
|
1818
|
+
) -> bool:
|
|
1819
|
+
"""Undo the one step of a REPLACE that the transaction cannot (N-15).
|
|
1820
|
+
|
|
1821
|
+
``_clear_library`` commits. A row deletion and an unlinked WAV are not
|
|
1822
|
+
enrolled in the apply transaction and are not rolled back with it, so
|
|
1823
|
+
without this, *any* failure after that point -- a rotted audio member, a
|
|
1824
|
+
write error, a record the archive should never have contained -- leaves an
|
|
1825
|
+
empty library, which is precisely the outcome F-44, N-15 and Section 9 all
|
|
1826
|
+
forbid. The bundle taken from the live library seconds earlier is applied
|
|
1827
|
+
back into what the clear left, which is normally nothing.
|
|
1828
|
+
|
|
1829
|
+
Identifiers are minted afresh, as they are for any restore (F-44), and if
|
|
1830
|
+
the clear itself failed part way then whatever survived it is restored
|
|
1831
|
+
alongside itself: a duplicate the user can delete beats a job they cannot
|
|
1832
|
+
get back. The whole of it is best effort by
|
|
1833
|
+
construction: it runs while another failure is unwinding and must never
|
|
1834
|
+
replace it, so what it managed is reported to the caller instead, which
|
|
1835
|
+
tells the user -- along with where the copy still on disk is.
|
|
1836
|
+
"""
|
|
1837
|
+
if rescue is None or not rescue.path.is_file():
|
|
1838
|
+
return False
|
|
1839
|
+
own_bytes, own_items = _own_limits(rescue.counts, rescue.uncompressed_bytes)
|
|
1840
|
+
recovered = _Applied()
|
|
1841
|
+
try:
|
|
1842
|
+
inspection = verify_backup(rescue.path, max_bytes=own_bytes, max_items=own_items, deep=True)
|
|
1843
|
+
_apply(
|
|
1844
|
+
store,
|
|
1845
|
+
rescue.path,
|
|
1846
|
+
inspection,
|
|
1847
|
+
owner_client_id=owner_client_id,
|
|
1848
|
+
mode=RestoreMode.ADD,
|
|
1849
|
+
moment=moment,
|
|
1850
|
+
progress=None,
|
|
1851
|
+
cancel=None,
|
|
1852
|
+
gate=gate,
|
|
1853
|
+
written=recovered,
|
|
1854
|
+
safety=None,
|
|
1855
|
+
max_bytes=own_bytes,
|
|
1856
|
+
max_items=own_items,
|
|
1857
|
+
)
|
|
1858
|
+
except Exception:
|
|
1859
|
+
recovered.undo()
|
|
1860
|
+
log.error("restore rollback failed source=%s", redact(rescue.path))
|
|
1861
|
+
return False
|
|
1862
|
+
log.info(
|
|
1863
|
+
"restore rolled back documents=%d jobs=%d source=%s",
|
|
1864
|
+
rescue.counts.documents,
|
|
1865
|
+
rescue.counts.jobs,
|
|
1866
|
+
redact(rescue.path),
|
|
1867
|
+
)
|
|
1868
|
+
return True
|
|
1869
|
+
|
|
1870
|
+
|
|
1871
|
+
def _replace_failed(original: EchoActError, safety: Path | None, put_back: bool) -> EchoActError:
|
|
1872
|
+
"""Say what became of the existing data, not only why the restore failed.
|
|
1873
|
+
|
|
1874
|
+
Section 9 ends at "corrupted data is never overwritten automatically", and
|
|
1875
|
+
a user left with an empty library and an error about a digest has no way to
|
|
1876
|
+
know a copy of everything is sitting in the data directory. The original
|
|
1877
|
+
code is kept -- it is still why the restore was refused, and F-57 makes the
|
|
1878
|
+
code the contract -- while the message and detail gain the fate of what was
|
|
1879
|
+
there before and the path of the copy that still holds it.
|
|
1880
|
+
"""
|
|
1881
|
+
if put_back:
|
|
1882
|
+
note = "The data that was there has been put back."
|
|
1883
|
+
fate = "restored"
|
|
1884
|
+
elif safety is not None:
|
|
1885
|
+
note = "The data that was there is in the copy taken before the restore started."
|
|
1886
|
+
fate = "in_safety_backup"
|
|
1887
|
+
else:
|
|
1888
|
+
note = "No copy of the data that was there was taken."
|
|
1889
|
+
fate = "lost"
|
|
1890
|
+
detail = dict(original.detail)
|
|
1891
|
+
detail["existing_data"] = fate
|
|
1892
|
+
if safety is not None:
|
|
1893
|
+
detail["safety_backup"] = redact(safety)
|
|
1894
|
+
return EchoActError(
|
|
1895
|
+
original.code,
|
|
1896
|
+
f"{original.message} {note}",
|
|
1897
|
+
detail=detail,
|
|
1898
|
+
retry_after_s=original.retry_after_s,
|
|
1899
|
+
cause=original,
|
|
1900
|
+
)
|
|
1901
|
+
|
|
1902
|
+
|
|
1903
|
+
def _clear_library(store: Store, committed: Callable[[], None]) -> None:
|
|
1904
|
+
"""Delete what a REPLACE restore is about to supersede.
|
|
1905
|
+
|
|
1906
|
+
``force`` is never passed: 5.3 requires a running job to be cancelled
|
|
1907
|
+
first, and ``delete_all_history`` refusing is exactly that rule -- and it
|
|
1908
|
+
refuses *before* it deletes anything, so a refusal costs the user nothing
|
|
1909
|
+
and needs no undoing.
|
|
1910
|
+
|
|
1911
|
+
``committed`` is called at the moment that stops being true. Everything
|
|
1912
|
+
after it is outside the caller's apply transaction and outside any other,
|
|
1913
|
+
so a failure part way through here is as unrecoverable on its own as a
|
|
1914
|
+
failure during the apply: the caller has to put the data back itself
|
|
1915
|
+
(N-15), and this is how it is told that it must.
|
|
1916
|
+
"""
|
|
1917
|
+
deletion = store.delete_all_history()
|
|
1918
|
+
committed()
|
|
1919
|
+
for relative in deletion.audio_paths:
|
|
1920
|
+
resolved = _resolve_inside(store.audio_root, relative)
|
|
1921
|
+
if resolved is not None:
|
|
1922
|
+
_discard(resolved)
|
|
1923
|
+
while True:
|
|
1924
|
+
page = store.list_documents(limit=100)
|
|
1925
|
+
if not page.items:
|
|
1926
|
+
return
|
|
1927
|
+
for summary in page.items:
|
|
1928
|
+
store.delete_document(summary.document_id)
|
|
1929
|
+
|
|
1930
|
+
|
|
1931
|
+
def _apply(
|
|
1932
|
+
store: Store,
|
|
1933
|
+
path: Path,
|
|
1934
|
+
inspection: BackupInspection,
|
|
1935
|
+
*,
|
|
1936
|
+
owner_client_id: str,
|
|
1937
|
+
mode: RestoreMode,
|
|
1938
|
+
moment: float,
|
|
1939
|
+
progress: ProgressCallback | None,
|
|
1940
|
+
cancel: CancelToken | None,
|
|
1941
|
+
gate: RestoreGate,
|
|
1942
|
+
written: _Applied,
|
|
1943
|
+
safety: Path | None,
|
|
1944
|
+
max_bytes: int,
|
|
1945
|
+
max_items: int,
|
|
1946
|
+
) -> RestoreOutcome:
|
|
1947
|
+
items: list[RestoredItem] = []
|
|
1948
|
+
job_map: dict[str, str] = {}
|
|
1949
|
+
counts = {"documents": 0, "jobs": 0, "segments": 0, "results": 0, "audio": 0}
|
|
1950
|
+
total = inspection.item_count
|
|
1951
|
+
budget = _ItemBudget(max_items)
|
|
1952
|
+
done = 0
|
|
1953
|
+
audio_root = store.audio_root
|
|
1954
|
+
restored_dir = audio_root / "restored"
|
|
1955
|
+
|
|
1956
|
+
with zipfile.ZipFile(path, "r") as zf, store.transaction() as conn:
|
|
1957
|
+
names = {i.filename for i in zf.infolist()}
|
|
1958
|
+
|
|
1959
|
+
for record in _records(zf, names, DOCUMENTS_MEMBER, max_bytes):
|
|
1960
|
+
if _cancelled(cancel):
|
|
1961
|
+
raise _Cancelled
|
|
1962
|
+
original = _text(record, "document_id")
|
|
1963
|
+
new_id = ids.document_id()
|
|
1964
|
+
conn.execute(
|
|
1965
|
+
"INSERT INTO documents (document_id, title, body, created_at, modified_at,"
|
|
1966
|
+
" version) VALUES (?, ?, ?, ?, ?, ?)",
|
|
1967
|
+
(
|
|
1968
|
+
new_id,
|
|
1969
|
+
_text(record, "title"),
|
|
1970
|
+
_text(record, "body"),
|
|
1971
|
+
_number(record, "created_at", moment),
|
|
1972
|
+
_number(record, "modified_at", moment),
|
|
1973
|
+
max(1, int(record.get("version") or 1)),
|
|
1974
|
+
),
|
|
1975
|
+
)
|
|
1976
|
+
items.append(RestoredItem("document", original, new_id))
|
|
1977
|
+
counts["documents"] += 1
|
|
1978
|
+
budget.take()
|
|
1979
|
+
done += 1
|
|
1980
|
+
gate.note(_emit(progress, RestorePhase.APPLYING, "documents", done, total, 0, 0))
|
|
1981
|
+
|
|
1982
|
+
for record in _records(zf, names, JOBS_MEMBER, max_bytes):
|
|
1983
|
+
if _cancelled(cancel):
|
|
1984
|
+
raise _Cancelled
|
|
1985
|
+
original = _text(record, "job_id")
|
|
1986
|
+
new_id = ids.job_id()
|
|
1987
|
+
job_map[original] = new_id
|
|
1988
|
+
conn.execute(
|
|
1989
|
+
"INSERT INTO jobs (job_id, kind, request_path, owner_client_id, client_label,"
|
|
1990
|
+
" state, retention, source_text, model_id, settings_json, budget_json,"
|
|
1991
|
+
" created_at, started_at, ended_at, error_code, error_message, idempotency_key,"
|
|
1992
|
+
" generated_segments, total_segments)"
|
|
1993
|
+
" VALUES (?, ?, ?, ?, NULL, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, ?, ?)",
|
|
1994
|
+
(
|
|
1995
|
+
new_id,
|
|
1996
|
+
_text(record, "kind"),
|
|
1997
|
+
_text(record, "request_path"),
|
|
1998
|
+
# F-44: restored data belongs to the GUI owner, and no
|
|
1999
|
+
# external client's access is restored with it.
|
|
2000
|
+
owner_client_id,
|
|
2001
|
+
_state(record),
|
|
2002
|
+
RetentionMode.RETAINED.value,
|
|
2003
|
+
record.get("source_text"),
|
|
2004
|
+
_text(record, "model_id"),
|
|
2005
|
+
_text(record, "settings_json"),
|
|
2006
|
+
record.get("budget_json"),
|
|
2007
|
+
_number(record, "created_at", moment),
|
|
2008
|
+
record.get("started_at"),
|
|
2009
|
+
record.get("ended_at"),
|
|
2010
|
+
record.get("error_code"),
|
|
2011
|
+
record.get("error_message"),
|
|
2012
|
+
max(0, int(record.get("generated_segments") or 0)),
|
|
2013
|
+
max(0, int(record.get("total_segments") or 0)),
|
|
2014
|
+
),
|
|
2015
|
+
)
|
|
2016
|
+
items.append(RestoredItem("job", original, new_id))
|
|
2017
|
+
counts["jobs"] += 1
|
|
2018
|
+
budget.take()
|
|
2019
|
+
done += 1
|
|
2020
|
+
gate.note(_emit(progress, RestorePhase.APPLYING, "history", done, total, 0, 0))
|
|
2021
|
+
|
|
2022
|
+
for record in _records(zf, names, SEGMENTS_MEMBER, max_bytes):
|
|
2023
|
+
if _cancelled(cancel):
|
|
2024
|
+
raise _Cancelled
|
|
2025
|
+
job_id = job_map.get(_text(record, "job_id"))
|
|
2026
|
+
if job_id is None:
|
|
2027
|
+
raise _reject("The backup has a segment with no job.")
|
|
2028
|
+
conn.execute(
|
|
2029
|
+
"INSERT INTO segments (segment_id, job_id, seq,"
|
|
2030
|
+
" source_start_codepoint_inclusive, source_end_codepoint_exclusive,"
|
|
2031
|
+
" spoken_text, language, audio_start_ms, audio_end_ms, trailing_silence_ms,"
|
|
2032
|
+
" audio_path, frame_count, ready)"
|
|
2033
|
+
" VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, ?, ?)",
|
|
2034
|
+
(
|
|
2035
|
+
ids.segment_id(),
|
|
2036
|
+
job_id,
|
|
2037
|
+
int(record.get("seq") or 0),
|
|
2038
|
+
_span(record, "source_start_codepoint_inclusive"),
|
|
2039
|
+
_span(record, "source_end_codepoint_exclusive"),
|
|
2040
|
+
_text(record, "spoken_text", allow_empty=True),
|
|
2041
|
+
_text(record, "language", allow_empty=True),
|
|
2042
|
+
record.get("audio_start_ms"),
|
|
2043
|
+
record.get("audio_end_ms"),
|
|
2044
|
+
max(0, int(record.get("trailing_silence_ms") or 0)),
|
|
2045
|
+
max(0, int(record.get("frame_count") or 0)),
|
|
2046
|
+
1 if record.get("ready") else 0,
|
|
2047
|
+
),
|
|
2048
|
+
)
|
|
2049
|
+
counts["segments"] += 1
|
|
2050
|
+
budget.take()
|
|
2051
|
+
done += 1
|
|
2052
|
+
gate.note(_emit(progress, RestorePhase.APPLYING, "segments", done, total, 0, 0))
|
|
2053
|
+
|
|
2054
|
+
for record in _records(zf, names, RESULTS_MEMBER, max_bytes):
|
|
2055
|
+
if _cancelled(cancel):
|
|
2056
|
+
raise _Cancelled
|
|
2057
|
+
job_id = job_map.get(_text(record, "job_id"))
|
|
2058
|
+
if job_id is None:
|
|
2059
|
+
raise _reject("The backup has a result with no job.")
|
|
2060
|
+
new_id = ids.result_id()
|
|
2061
|
+
expected = _text(record, "digest")
|
|
2062
|
+
member = _text(record, "audio_member")
|
|
2063
|
+
# The member name is validated a second time on the way out of the
|
|
2064
|
+
# record, not only on the way in from the archive: this string came
|
|
2065
|
+
# from a JSON line, which the name checks never saw.
|
|
2066
|
+
_check_member_name(member)
|
|
2067
|
+
if member not in names:
|
|
2068
|
+
raise _reject("The backup names audio it does not contain.", entry=member[:80])
|
|
2069
|
+
destination = restored_dir / f"{new_id}.wav"
|
|
2070
|
+
written.paths.append(destination)
|
|
2071
|
+
size = _extract_audio(zf, member, destination, expected, cancel, max_bytes)
|
|
2072
|
+
relative = destination.relative_to(audio_root).as_posix()
|
|
2073
|
+
conn.execute(
|
|
2074
|
+
"INSERT INTO results (result_id, job_id, sample_rate, channels,"
|
|
2075
|
+
" sample_width_bits, frame_count, byte_size, digest, relative_path,"
|
|
2076
|
+
" created_at, expires_at, integrity_state, verified_at)"
|
|
2077
|
+
" VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, NULL, 'ok', ?)",
|
|
2078
|
+
(
|
|
2079
|
+
new_id,
|
|
2080
|
+
job_id,
|
|
2081
|
+
int(record.get("sample_rate") or 0),
|
|
2082
|
+
int(record.get("channels") or 0),
|
|
2083
|
+
int(record.get("sample_width_bits") or 0),
|
|
2084
|
+
max(0, int(record.get("frame_count") or 0)),
|
|
2085
|
+
size,
|
|
2086
|
+
expected,
|
|
2087
|
+
relative,
|
|
2088
|
+
_number(record, "created_at", moment),
|
|
2089
|
+
moment,
|
|
2090
|
+
),
|
|
2091
|
+
)
|
|
2092
|
+
counts["results"] += 1
|
|
2093
|
+
counts["audio"] += 1
|
|
2094
|
+
# A result is two of 4.1's items: the row and the file.
|
|
2095
|
+
budget.take(2)
|
|
2096
|
+
done += 1
|
|
2097
|
+
gate.note(_emit(progress, RestorePhase.APPLYING, member, done, total, size, size))
|
|
2098
|
+
|
|
2099
|
+
usage = store.storage_usage(at=moment)
|
|
2100
|
+
if usage.total_bytes > usage.limit_bytes:
|
|
2101
|
+
# Checked again with the rows in place: the pre-check trusted the
|
|
2102
|
+
# manifest, and 4.1 refuses rather than deleting to make room.
|
|
2103
|
+
raise EchoActError(
|
|
2104
|
+
Code.RETENTION_LIMIT_REACHED,
|
|
2105
|
+
"Restoring this backup would exceed the retention limit.",
|
|
2106
|
+
detail={"used_bytes": usage.total_bytes, "limit_bytes": usage.limit_bytes},
|
|
2107
|
+
)
|
|
2108
|
+
|
|
2109
|
+
log.info(
|
|
2110
|
+
"restore applied mode=%s documents=%d jobs=%d results=%d source=%s",
|
|
2111
|
+
mode.value,
|
|
2112
|
+
counts["documents"],
|
|
2113
|
+
counts["jobs"],
|
|
2114
|
+
counts["results"],
|
|
2115
|
+
redact(path),
|
|
2116
|
+
)
|
|
2117
|
+
final = BackupCounts(
|
|
2118
|
+
documents=counts["documents"],
|
|
2119
|
+
jobs=counts["jobs"],
|
|
2120
|
+
segments=counts["segments"],
|
|
2121
|
+
results=counts["results"],
|
|
2122
|
+
audio_files=counts["audio"],
|
|
2123
|
+
)
|
|
2124
|
+
_emit(progress, RestorePhase.COMPLETE, path.name, total, total, 0, 0)
|
|
2125
|
+
return RestoreOutcome(
|
|
2126
|
+
source=path,
|
|
2127
|
+
mode=mode,
|
|
2128
|
+
counts=final,
|
|
2129
|
+
items=tuple(items),
|
|
2130
|
+
owner_client_id=owner_client_id,
|
|
2131
|
+
safety_backup=safety,
|
|
2132
|
+
cancelled=False,
|
|
2133
|
+
)
|
|
2134
|
+
|
|
2135
|
+
|
|
2136
|
+
def _records(
|
|
2137
|
+
zf: zipfile.ZipFile, names: set[str], member: str, max_bytes: int
|
|
2138
|
+
) -> Iterator[dict[str, Any]]:
|
|
2139
|
+
if member not in names:
|
|
2140
|
+
return
|
|
2141
|
+
yield from _iter_records(zf, zf.getinfo(member), limit=max_bytes)
|
|
2142
|
+
|
|
2143
|
+
|
|
2144
|
+
def _extract_audio(
|
|
2145
|
+
zf: zipfile.ZipFile,
|
|
2146
|
+
member: str,
|
|
2147
|
+
destination: Path,
|
|
2148
|
+
expected_digest: str,
|
|
2149
|
+
cancel: CancelToken | None,
|
|
2150
|
+
max_bytes: int,
|
|
2151
|
+
) -> int:
|
|
2152
|
+
"""Write one WAV to a name this module chose, and refuse a mismatch.
|
|
2153
|
+
|
|
2154
|
+
The destination is never derived from the archive; ``member`` only says
|
|
2155
|
+
which bytes to read. That is the second half of N-27's path defence: even
|
|
2156
|
+
if a name check were wrong, there is nowhere for a crafted name to land.
|
|
2157
|
+
"""
|
|
2158
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
2159
|
+
h = hashlib.sha256()
|
|
2160
|
+
size = 0
|
|
2161
|
+
info = zf.getinfo(member)
|
|
2162
|
+
try:
|
|
2163
|
+
with destination.open("wb") as out:
|
|
2164
|
+
for chunk in _bounded_read(zf, info, max_bytes):
|
|
2165
|
+
if _cancelled(cancel):
|
|
2166
|
+
raise _Cancelled
|
|
2167
|
+
out.write(chunk)
|
|
2168
|
+
h.update(chunk)
|
|
2169
|
+
size += len(chunk)
|
|
2170
|
+
except OSError as exc:
|
|
2171
|
+
raise EchoActError(
|
|
2172
|
+
Code.STORAGE_FULL if getattr(exc, "errno", None) == 28 else Code.BACKUP_INVALID,
|
|
2173
|
+
"Restored audio could not be written.",
|
|
2174
|
+
detail={"path": redact(destination)},
|
|
2175
|
+
cause=exc,
|
|
2176
|
+
) from exc
|
|
2177
|
+
if h.hexdigest() != expected_digest:
|
|
2178
|
+
raise _reject("Restored audio does not match its recorded digest.", entry=member[:80])
|
|
2179
|
+
return size
|
|
2180
|
+
|
|
2181
|
+
|
|
2182
|
+
def _text(record: dict[str, Any], key: str, *, allow_empty: bool = False) -> str:
|
|
2183
|
+
value = record.get(key)
|
|
2184
|
+
if not isinstance(value, str) or (not value and not allow_empty):
|
|
2185
|
+
raise _reject(f"The backup has a record with no {key}.")
|
|
2186
|
+
return value
|
|
2187
|
+
|
|
2188
|
+
|
|
2189
|
+
def _number(record: dict[str, Any], key: str, fallback: float) -> float:
|
|
2190
|
+
value = record.get(key)
|
|
2191
|
+
if isinstance(value, bool) or not isinstance(value, int | float):
|
|
2192
|
+
return fallback
|
|
2193
|
+
return float(value)
|
|
2194
|
+
|
|
2195
|
+
|
|
2196
|
+
def _span(record: dict[str, Any], key: str) -> int:
|
|
2197
|
+
value = record.get(key)
|
|
2198
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
|
2199
|
+
raise _reject(f"The backup has a segment with a bad {key}.")
|
|
2200
|
+
return value
|
|
2201
|
+
|
|
2202
|
+
|
|
2203
|
+
def _state(record: dict[str, Any]) -> str:
|
|
2204
|
+
"""Only a terminal state may be restored.
|
|
2205
|
+
|
|
2206
|
+
5.1 says a job is never run again, so a restored job in ``generating``
|
|
2207
|
+
would be a row nothing could ever advance. The bundle never contains one;
|
|
2208
|
+
an archive that claims otherwise is rejected rather than corrected.
|
|
2209
|
+
"""
|
|
2210
|
+
value = _text(record, "state")
|
|
2211
|
+
if value not in _TERMINAL_STATE_VALUES:
|
|
2212
|
+
raise _reject("The backup contains a job that never finished.", state=value[:32])
|
|
2213
|
+
return value
|
|
2214
|
+
|
|
2215
|
+
|
|
2216
|
+
class _NullGate(RestoreGate):
|
|
2217
|
+
"""A gate that is never busy, for the backup a restore takes of itself."""
|
|
2218
|
+
|
|
2219
|
+
def require_idle(self, what: str = "This action") -> None:
|
|
2220
|
+
return
|
|
2221
|
+
|
|
2222
|
+
|
|
2223
|
+
_NULL_GATE: Final = _NullGate()
|
|
2224
|
+
|
|
2225
|
+
|
|
2226
|
+
# ======================================================================
|
|
2227
|
+
# F-74: the scheduled backup
|
|
2228
|
+
# ======================================================================
|
|
2229
|
+
|
|
2230
|
+
|
|
2231
|
+
class ScheduleReason(StrEnum):
|
|
2232
|
+
DISABLED = "disabled"
|
|
2233
|
+
NO_LOCATION = "no_location"
|
|
2234
|
+
NOT_DUE = "not_due"
|
|
2235
|
+
FIRST_RUN = "first_run"
|
|
2236
|
+
DUE = "due"
|
|
2237
|
+
#: The app was not running when the schedule came round. F-74 makes this
|
|
2238
|
+
#: up once, which falls out of dating the next run from the last one that
|
|
2239
|
+
#: happened rather than from the schedule that was missed.
|
|
2240
|
+
MISSED = "missed"
|
|
2241
|
+
DEFERRED_GENERATING = "deferred_generating"
|
|
2242
|
+
DEFERRED_RESTORING = "deferred_restoring"
|
|
2243
|
+
DEFERRED_AFTER_FAILURE = "deferred_after_failure"
|
|
2244
|
+
|
|
2245
|
+
|
|
2246
|
+
@dataclass(frozen=True, slots=True)
|
|
2247
|
+
class ScheduleDecision:
|
|
2248
|
+
run: bool
|
|
2249
|
+
reason: ScheduleReason
|
|
2250
|
+
due_at: float | None
|
|
2251
|
+
|
|
2252
|
+
def __bool__(self) -> bool:
|
|
2253
|
+
return self.run
|
|
2254
|
+
|
|
2255
|
+
|
|
2256
|
+
def last_scheduled_run(store: Store) -> float | None:
|
|
2257
|
+
"""When the last *sound* scheduled backup was taken (F-74).
|
|
2258
|
+
|
|
2259
|
+
An unverified record is an attempt, not a backup; counting one as a run
|
|
2260
|
+
would let a day pass with nothing recoverable on disk.
|
|
2261
|
+
"""
|
|
2262
|
+
for record in store.list_backups(kind="scheduled"):
|
|
2263
|
+
if record.verified:
|
|
2264
|
+
return record.created_at
|
|
2265
|
+
return None
|
|
2266
|
+
|
|
2267
|
+
|
|
2268
|
+
class BackupScheduler:
|
|
2269
|
+
"""F-74's once-daily backup, as a plain object with an explicit clock.
|
|
2270
|
+
|
|
2271
|
+
Nothing here sleeps, waits, or reads a wall clock of its own: the caller
|
|
2272
|
+
ticks it with ``now``. That is what makes "deferred during generation",
|
|
2273
|
+
"made up only once", and "seven kept" testable without a day passing.
|
|
2274
|
+
"""
|
|
2275
|
+
|
|
2276
|
+
__slots__ = ("interval_s", "keep", "retry_after_s", "run_on_first_enable", "_last_failure")
|
|
2277
|
+
|
|
2278
|
+
def __init__(
|
|
2279
|
+
self,
|
|
2280
|
+
*,
|
|
2281
|
+
interval_s: float = SCHEDULED_BACKUP_INTERVAL_S,
|
|
2282
|
+
keep: int = BACKUP_KEEP_SCHEDULED,
|
|
2283
|
+
retry_after_s: float = SCHEDULED_BACKUP_RETRY_S,
|
|
2284
|
+
run_on_first_enable: bool = True,
|
|
2285
|
+
) -> None:
|
|
2286
|
+
self.interval_s = float(interval_s)
|
|
2287
|
+
self.keep = int(keep)
|
|
2288
|
+
self.retry_after_s = float(retry_after_s)
|
|
2289
|
+
#: With this off, a machine that is never on for a whole day would
|
|
2290
|
+
#: never get a scheduled backup at all, because F-74 runs the schedule
|
|
2291
|
+
#: only while the app is running.
|
|
2292
|
+
self.run_on_first_enable = run_on_first_enable
|
|
2293
|
+
self._last_failure: float | None = None
|
|
2294
|
+
|
|
2295
|
+
def decide(
|
|
2296
|
+
self,
|
|
2297
|
+
now: float,
|
|
2298
|
+
*,
|
|
2299
|
+
enabled: bool,
|
|
2300
|
+
location: str | os.PathLike[str] | None,
|
|
2301
|
+
last_run: float | None,
|
|
2302
|
+
generating: bool = False,
|
|
2303
|
+
restoring: bool = False,
|
|
2304
|
+
) -> ScheduleDecision:
|
|
2305
|
+
"""Whether to take a scheduled backup at ``now``, and why not if not."""
|
|
2306
|
+
if not enabled:
|
|
2307
|
+
return ScheduleDecision(False, ScheduleReason.DISABLED, None)
|
|
2308
|
+
if location is None or not str(location):
|
|
2309
|
+
return ScheduleDecision(False, ScheduleReason.NO_LOCATION, None)
|
|
2310
|
+
due_at = None if last_run is None else last_run + self.interval_s
|
|
2311
|
+
if last_run is None:
|
|
2312
|
+
reason = ScheduleReason.FIRST_RUN
|
|
2313
|
+
if not self.run_on_first_enable:
|
|
2314
|
+
return ScheduleDecision(False, ScheduleReason.NOT_DUE, now + self.interval_s)
|
|
2315
|
+
elif now < due_at:
|
|
2316
|
+
return ScheduleDecision(False, ScheduleReason.NOT_DUE, due_at)
|
|
2317
|
+
elif now >= last_run + 2 * self.interval_s:
|
|
2318
|
+
reason = ScheduleReason.MISSED
|
|
2319
|
+
else:
|
|
2320
|
+
reason = ScheduleReason.DUE
|
|
2321
|
+
if generating:
|
|
2322
|
+
# F-74 defers rather than cancels: the decision stays due, so the
|
|
2323
|
+
# next tick after generation ends runs it.
|
|
2324
|
+
return ScheduleDecision(False, ScheduleReason.DEFERRED_GENERATING, due_at)
|
|
2325
|
+
if restoring:
|
|
2326
|
+
return ScheduleDecision(False, ScheduleReason.DEFERRED_RESTORING, due_at)
|
|
2327
|
+
if self._last_failure is not None and now < self._last_failure + self.retry_after_s:
|
|
2328
|
+
return ScheduleDecision(
|
|
2329
|
+
False, ScheduleReason.DEFERRED_AFTER_FAILURE, self._last_failure + self.retry_after_s
|
|
2330
|
+
)
|
|
2331
|
+
return ScheduleDecision(True, reason, due_at)
|
|
2332
|
+
|
|
2333
|
+
def run_due(
|
|
2334
|
+
self,
|
|
2335
|
+
store: Store,
|
|
2336
|
+
now: float,
|
|
2337
|
+
*,
|
|
2338
|
+
enabled: bool,
|
|
2339
|
+
location: str | os.PathLike[str] | None,
|
|
2340
|
+
generating: bool = False,
|
|
2341
|
+
restoring: bool = False,
|
|
2342
|
+
selection: BackupSelection | None = None,
|
|
2343
|
+
progress: ProgressCallback | None = None,
|
|
2344
|
+
cancel: CancelToken | None = None,
|
|
2345
|
+
free_space: FreeSpace = _disk_free,
|
|
2346
|
+
gate: RestoreGate | None = None,
|
|
2347
|
+
) -> BackupOutcome | None:
|
|
2348
|
+
"""Take the backup if it is due, then rotate -- in that order.
|
|
2349
|
+
|
|
2350
|
+
4.1 is explicit that older scheduled backups are cleaned up only after
|
|
2351
|
+
a new one is verified, so rotation happens here, after
|
|
2352
|
+
:func:`create_backup` has verified and recorded the new file, and never
|
|
2353
|
+
on a failed or cancelled attempt.
|
|
2354
|
+
"""
|
|
2355
|
+
the_gate = gate or RESTORE_GATE
|
|
2356
|
+
decision = self.decide(
|
|
2357
|
+
now,
|
|
2358
|
+
enabled=enabled,
|
|
2359
|
+
location=location,
|
|
2360
|
+
last_run=last_scheduled_run(store),
|
|
2361
|
+
generating=generating,
|
|
2362
|
+
restoring=restoring or the_gate.active,
|
|
2363
|
+
)
|
|
2364
|
+
if not decision.run:
|
|
2365
|
+
return None
|
|
2366
|
+
directory = Path(str(location))
|
|
2367
|
+
destination = directory / scheduled_backup_name(now)
|
|
2368
|
+
try:
|
|
2369
|
+
outcome = create_backup(
|
|
2370
|
+
store,
|
|
2371
|
+
destination,
|
|
2372
|
+
selection=selection,
|
|
2373
|
+
kind="scheduled",
|
|
2374
|
+
at=now,
|
|
2375
|
+
progress=progress,
|
|
2376
|
+
cancel=cancel,
|
|
2377
|
+
free_space=free_space,
|
|
2378
|
+
gate=the_gate,
|
|
2379
|
+
)
|
|
2380
|
+
except Exception:
|
|
2381
|
+
# Every failed attempt backs off, not only the ones that arrive as
|
|
2382
|
+
# an EchoActError. A backup that fails some other way fails the
|
|
2383
|
+
# same way on the next tick, and retrying it sixty times an hour
|
|
2384
|
+
# is what SCHEDULED_BACKUP_RETRY_S exists to prevent.
|
|
2385
|
+
self._last_failure = now
|
|
2386
|
+
raise
|
|
2387
|
+
if outcome.cancelled:
|
|
2388
|
+
self._last_failure = now
|
|
2389
|
+
return outcome
|
|
2390
|
+
self._last_failure = None
|
|
2391
|
+
prune_scheduled_backups(
|
|
2392
|
+
store,
|
|
2393
|
+
keep=self.keep,
|
|
2394
|
+
protect=() if outcome.record is None else (outcome.record.backup_id,),
|
|
2395
|
+
)
|
|
2396
|
+
return outcome
|
|
2397
|
+
|
|
2398
|
+
|
|
2399
|
+
def prune_scheduled_backups(
|
|
2400
|
+
store: Store,
|
|
2401
|
+
*,
|
|
2402
|
+
keep: int = BACKUP_KEEP_SCHEDULED,
|
|
2403
|
+
protect: Sequence[str] = (),
|
|
2404
|
+
) -> tuple[str, ...]:
|
|
2405
|
+
"""Keep the ``keep`` most recent sound scheduled backups; drop the rest.
|
|
2406
|
+
|
|
2407
|
+
Only records of kind ``scheduled`` are ever considered. 4.1 says manual
|
|
2408
|
+
backups are never deleted automatically, and the pre-migration copies N-15
|
|
2409
|
+
takes are not this feature's to reclaim either -- so both are invisible
|
|
2410
|
+
here by construction rather than by a filter that could be edited away.
|
|
2411
|
+
"""
|
|
2412
|
+
protected = set(protect)
|
|
2413
|
+
records = sorted(store.list_backups(kind="scheduled"), key=lambda r: r.created_at, reverse=True)
|
|
2414
|
+
kept = 0
|
|
2415
|
+
removed: list[str] = []
|
|
2416
|
+
for record in records:
|
|
2417
|
+
# The backup that triggered this rotation is one of the kept seven,
|
|
2418
|
+
# not an eighth alongside them.
|
|
2419
|
+
if record.backup_id in protected or (record.verified and kept < keep):
|
|
2420
|
+
kept += 1
|
|
2421
|
+
continue
|
|
2422
|
+
location = Path(record.location)
|
|
2423
|
+
if location.is_file() and location.name.endswith(BACKUP_SUFFIX):
|
|
2424
|
+
_discard(location)
|
|
2425
|
+
store.delete_backup_record(record.backup_id)
|
|
2426
|
+
removed.append(record.backup_id)
|
|
2427
|
+
if removed:
|
|
2428
|
+
log.info("scheduled backups pruned kept=%d removed=%d", kept, len(removed))
|
|
2429
|
+
return tuple(removed)
|