weightsdb 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- weightsdb/__about__.py +1 -0
- weightsdb/__init__.py +71 -0
- weightsdb/backup.py +610 -0
- weightsdb/engine.py +229 -0
- weightsdb/errors.py +83 -0
- weightsdb/health.py +318 -0
- weightsdb/migrations.py +328 -0
- weightsdb/py.typed +0 -0
- weightsdb/redaction.py +26 -0
- weightsdb/session.py +115 -0
- weightsdb/testing.py +131 -0
- weightsdb/types.py +187 -0
- weightsdb-0.2.0.dist-info/METADATA +96 -0
- weightsdb-0.2.0.dist-info/RECORD +16 -0
- weightsdb-0.2.0.dist-info/WHEEL +4 -0
- weightsdb-0.2.0.dist-info/licenses/LICENSE +201 -0
weightsdb/engine.py
ADDED
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""weightsdb.engine — dialect-correct engine construction.
|
|
2
|
+
|
|
3
|
+
Database standards §2: SQLite gets ``foreign_keys=ON``, ``journal_mode=WAL``, ``busy_timeout``,
|
|
4
|
+
``synchronous=NORMAL``, applied per connection so a pool reconnect never silently loses them;
|
|
5
|
+
PostgreSQL gets ``statement_timeout``, ``lock_timeout`` and ``application_name``. Only these two
|
|
6
|
+
dialects are supported (§2) — a third requires an ADR, not a code change here.
|
|
7
|
+
|
|
8
|
+
Moved from FreeWeight's ``infrastructure.db.engine``, which was written "as if it were WeightsDB's
|
|
9
|
+
own" module for exactly this move (ADR-0011). Generalized in one respect the application version
|
|
10
|
+
did not need: SQLite lock contention beyond ``busy_timeout`` now raises the typed
|
|
11
|
+
:class:`~weightsdb.errors.StorageBusy` (spec §13) instead of a raw
|
|
12
|
+
``sqlalchemy.exc.OperationalError`` — FreeWeight had one consumer and no test for that translation;
|
|
13
|
+
WeightsDB's contract requires it.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import sqlite3
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from sqlalchemy import create_engine, event
|
|
23
|
+
from sqlalchemy.engine import Engine, make_url
|
|
24
|
+
from sqlalchemy.exc import OperationalError as SAOperationalError
|
|
25
|
+
|
|
26
|
+
from weightsdb.errors import DatabaseError, StorageBusy
|
|
27
|
+
|
|
28
|
+
__all__ = ["READ_ONLY_EXECUTION_OPTION", "create_engine_for"]
|
|
29
|
+
|
|
30
|
+
READ_ONLY_EXECUTION_OPTION = "weightsdb_read_only"
|
|
31
|
+
"""Execution option marking a transaction as read-only.
|
|
32
|
+
|
|
33
|
+
Set it on a connection (:meth:`Connection.execution_options`) — as
|
|
34
|
+
:func:`~weightsdb.session.transaction` does — to select a deferred ``BEGIN`` instead of
|
|
35
|
+
``BEGIN IMMEDIATE`` on SQLite, enforced with ``PRAGMA query_only``; on PostgreSQL it is inert, since
|
|
36
|
+
ordinary MVCC already lets readers and writers proceed without blocking each other.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
_SUPPORTED_DIALECTS = frozenset({"sqlite", "postgresql"})
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def create_engine_for(
|
|
43
|
+
url: str,
|
|
44
|
+
*,
|
|
45
|
+
echo: bool = False,
|
|
46
|
+
pool_size: int | None = None,
|
|
47
|
+
statement_timeout_ms: int | None = None,
|
|
48
|
+
sqlite_busy_timeout_ms: int = 5000,
|
|
49
|
+
application_name: str | None = None,
|
|
50
|
+
) -> Engine:
|
|
51
|
+
"""Build an engine with dialect-correct settings applied to every connection.
|
|
52
|
+
|
|
53
|
+
Every setting below is applied by an event listener on the returned engine's connection pool,
|
|
54
|
+
not once at construction time — the pool can and does open new DBAPI connections after the
|
|
55
|
+
first (on recycle, after a disposal, after a dropped network connection to PostgreSQL), and a
|
|
56
|
+
setting that only took effect on the first connection would silently stop applying on the
|
|
57
|
+
second (the specific failure mode spec §7's "pragmas lost after a pool reconnect" names).
|
|
58
|
+
|
|
59
|
+
SQLite additionally gets ``BEGIN IMMEDIATE`` transaction semantics: pysqlite's own implicit
|
|
60
|
+
transaction handling is disabled (``isolation_level = None`` on the raw connection) and
|
|
61
|
+
replaced with an explicit ``BEGIN IMMEDIATE`` issued by SQLAlchemy's ``"begin"`` hook, so lock
|
|
62
|
+
contention under SQLite's single-writer model fails fast — at the start of a transaction,
|
|
63
|
+
within ``sqlite_busy_timeout_ms`` — rather than silently at commit time, which is when
|
|
64
|
+
pysqlite's default deferred-transaction behaviour would otherwise surface it.
|
|
65
|
+
|
|
66
|
+
That matters because the deferred failure is not merely late, it is **unrecoverable**: a
|
|
67
|
+
transaction that reads, then tries to write after another connection has committed, gets
|
|
68
|
+
``SQLITE_BUSY_SNAPSHOT``, which ``busy_timeout`` does not apply to. It fails instantly, no
|
|
69
|
+
amount of waiting helps, and the only escape is to roll back and redo work already done.
|
|
70
|
+
``BEGIN IMMEDIATE`` converts that into an ordinary retryable wait at the start, before any work
|
|
71
|
+
exists to lose.
|
|
72
|
+
|
|
73
|
+
The cost is that it declares *every* transaction a writer, and WAL's whole point is that
|
|
74
|
+
readers never contend — so a transaction that only reads must say so, via
|
|
75
|
+
:data:`READ_ONLY_EXECUTION_OPTION` (:func:`~weightsdb.session.transaction`), and gets a
|
|
76
|
+
deferred ``BEGIN`` instead. Without that, two concurrent readers would queue behind each other
|
|
77
|
+
on the single write lock for up to ``sqlite_busy_timeout_ms`` and then fail.
|
|
78
|
+
|
|
79
|
+
Args:
|
|
80
|
+
url: A ``sqlite:///`` or ``postgresql(+driver)://`` URL. Only these two dialects are
|
|
81
|
+
supported (database standards §2).
|
|
82
|
+
echo: Log every emitted SQL statement. Development use only.
|
|
83
|
+
pool_size: PostgreSQL connection pool size. Ignored for SQLite, whose pool classes do not
|
|
84
|
+
take this argument.
|
|
85
|
+
statement_timeout_ms: PostgreSQL ``statement_timeout`` (and, identically, ``lock_timeout``,
|
|
86
|
+
since a statement that cannot even acquire its lock should not wait longer than the
|
|
87
|
+
statement itself is allowed to run). ``None`` leaves both at the server default.
|
|
88
|
+
Ignored for SQLite, which has no equivalent server-side setting —
|
|
89
|
+
``sqlite_busy_timeout_ms`` is SQLite's analogue.
|
|
90
|
+
sqlite_busy_timeout_ms: SQLite's ``PRAGMA busy_timeout``. Ignored for PostgreSQL.
|
|
91
|
+
application_name: PostgreSQL ``application_name``, surfaced in ``pg_stat_activity``.
|
|
92
|
+
Ignored for SQLite.
|
|
93
|
+
|
|
94
|
+
Returns:
|
|
95
|
+
A configured :class:`~sqlalchemy.Engine`. Construction is cheap — no connection is opened
|
|
96
|
+
until first use — so this comfortably meets the ≤ 50 ms creation budget (spec §15).
|
|
97
|
+
|
|
98
|
+
Raises:
|
|
99
|
+
DatabaseError: ``url``'s dialect is neither ``sqlite`` nor ``postgresql``.
|
|
100
|
+
"""
|
|
101
|
+
dialect = make_url(url).get_backend_name()
|
|
102
|
+
if dialect not in _SUPPORTED_DIALECTS:
|
|
103
|
+
raise DatabaseError(
|
|
104
|
+
f"Unsupported dialect {dialect!r}; only sqlite and postgresql are supported "
|
|
105
|
+
"(database standards §2). Adding a third dialect requires an ADR.",
|
|
106
|
+
details={"dialect": dialect},
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
engine_kwargs: dict[str, Any] = {"echo": echo}
|
|
110
|
+
if dialect == "postgresql" and pool_size is not None:
|
|
111
|
+
engine_kwargs["pool_size"] = pool_size
|
|
112
|
+
|
|
113
|
+
engine = create_engine(url, **engine_kwargs)
|
|
114
|
+
|
|
115
|
+
if dialect == "sqlite":
|
|
116
|
+
_ensure_sqlite_directory_exists(url)
|
|
117
|
+
_configure_sqlite(engine, busy_timeout_ms=sqlite_busy_timeout_ms)
|
|
118
|
+
else:
|
|
119
|
+
_configure_postgresql(
|
|
120
|
+
engine, statement_timeout_ms=statement_timeout_ms, application_name=application_name
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
return engine
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _ensure_sqlite_directory_exists(url: str) -> None:
|
|
127
|
+
"""Create the SQLite file's parent directory, so a fresh install has somewhere to write.
|
|
128
|
+
|
|
129
|
+
A no-op for ``:memory:`` and for an already-existing directory.
|
|
130
|
+
"""
|
|
131
|
+
# `URL.database`, not a hand-rolled parse: the sqlite dialect owns the rule for how many
|
|
132
|
+
# leading slashes separate "sqlite://" from an absolute path, and getting it wrong turns
|
|
133
|
+
# "sqlite:////tmp/x" into the relative "tmp/x" against whatever the cwd happens to be.
|
|
134
|
+
database = make_url(url).database
|
|
135
|
+
if not database or database == ":memory:":
|
|
136
|
+
return
|
|
137
|
+
Path(database).parent.mkdir(parents=True, exist_ok=True)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _raise_if_busy(exc: SAOperationalError, *, busy_timeout_ms: int) -> None:
|
|
141
|
+
"""Translate a SQLite busy/snapshot-busy failure into :class:`StorageBusy`.
|
|
142
|
+
|
|
143
|
+
Re-raises anything else unchanged, so a genuine schema or syntax error at ``BEGIN`` time — not
|
|
144
|
+
that one is expected — is never miscast as contention.
|
|
145
|
+
"""
|
|
146
|
+
orig = exc.orig
|
|
147
|
+
code = getattr(orig, "sqlite_errorcode", None)
|
|
148
|
+
if code in (sqlite3.SQLITE_BUSY, getattr(sqlite3, "SQLITE_BUSY_SNAPSHOT", -1)):
|
|
149
|
+
raise StorageBusy(
|
|
150
|
+
f"SQLite database is locked (busy_timeout={busy_timeout_ms}ms exceeded).",
|
|
151
|
+
details={"busy_timeout_ms": busy_timeout_ms},
|
|
152
|
+
) from exc
|
|
153
|
+
raise exc
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _configure_sqlite(engine: Engine, *, busy_timeout_ms: int) -> None:
|
|
157
|
+
"""Wire pragmas and ``BEGIN IMMEDIATE`` onto every connection this engine opens."""
|
|
158
|
+
|
|
159
|
+
@event.listens_for(engine, "connect")
|
|
160
|
+
def _on_connect(dbapi_connection: Any, _connection_record: Any) -> None:
|
|
161
|
+
# Hand transaction control to SQLAlchemy entirely: pysqlite's own implicit-BEGIN behaviour
|
|
162
|
+
# and our explicit "begin" hook below would otherwise fight over who opens the
|
|
163
|
+
# transaction, producing "cannot start a transaction within a transaction".
|
|
164
|
+
dbapi_connection.isolation_level = None
|
|
165
|
+
cursor = dbapi_connection.cursor()
|
|
166
|
+
try:
|
|
167
|
+
cursor.execute("PRAGMA foreign_keys=ON")
|
|
168
|
+
cursor.execute("PRAGMA journal_mode=WAL")
|
|
169
|
+
cursor.execute(f"PRAGMA busy_timeout={int(busy_timeout_ms)}")
|
|
170
|
+
cursor.execute("PRAGMA synchronous=NORMAL")
|
|
171
|
+
finally:
|
|
172
|
+
cursor.close()
|
|
173
|
+
|
|
174
|
+
@event.listens_for(engine, "begin")
|
|
175
|
+
def _on_begin(connection: Any) -> None:
|
|
176
|
+
# A connection explicitly switched to AUTOCOMMIT (e.g. VACUUM, which cannot run inside any
|
|
177
|
+
# transaction on SQLite) still fires this event — SQLAlchemy's "begin" hook does not know
|
|
178
|
+
# what AUTOCOMMIT means, only this dialect-specific listener does. Forcing a literal
|
|
179
|
+
# BEGIN IMMEDIATE onto such a connection is exactly the failure that opts it out here.
|
|
180
|
+
options = connection.get_execution_options()
|
|
181
|
+
if options.get("isolation_level") == "AUTOCOMMIT":
|
|
182
|
+
connection.exec_driver_sql("PRAGMA query_only=OFF")
|
|
183
|
+
return
|
|
184
|
+
if options.get(READ_ONLY_EXECUTION_OPTION):
|
|
185
|
+
# Plain BEGIN — deferred. A transaction that only reads never has to upgrade to the
|
|
186
|
+
# write lock, so the snapshot hazard IMMEDIATE exists to avoid cannot arise, and
|
|
187
|
+
# declaring it a writer would throw away WAL's concurrent reads for nothing.
|
|
188
|
+
connection.exec_driver_sql("PRAGMA query_only=ON")
|
|
189
|
+
try:
|
|
190
|
+
connection.exec_driver_sql("BEGIN")
|
|
191
|
+
except SAOperationalError as exc:
|
|
192
|
+
_raise_if_busy(exc, busy_timeout_ms=busy_timeout_ms)
|
|
193
|
+
return
|
|
194
|
+
# `query_only` is set explicitly on *both* paths, every transaction, rather than being
|
|
195
|
+
# cleaned up after the read-only one. A reset that only runs on the way out is a reset
|
|
196
|
+
# that does not run when the way out is an exception, and the failure mode — a pooled
|
|
197
|
+
# connection stuck read-only, rejecting writes for the rest of its life — is both silent
|
|
198
|
+
# and very hard to trace back to here.
|
|
199
|
+
connection.exec_driver_sql("PRAGMA query_only=OFF")
|
|
200
|
+
try:
|
|
201
|
+
connection.exec_driver_sql("BEGIN IMMEDIATE")
|
|
202
|
+
except SAOperationalError as exc:
|
|
203
|
+
_raise_if_busy(exc, busy_timeout_ms=busy_timeout_ms)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _configure_postgresql(
|
|
207
|
+
engine: Engine, *, statement_timeout_ms: int | None, application_name: str | None
|
|
208
|
+
) -> None:
|
|
209
|
+
"""Apply ``statement_timeout``/``lock_timeout``/``application_name`` on every connection."""
|
|
210
|
+
|
|
211
|
+
@event.listens_for(engine, "connect")
|
|
212
|
+
def _on_connect(dbapi_connection: Any, _connection_record: Any) -> None:
|
|
213
|
+
cursor = dbapi_connection.cursor()
|
|
214
|
+
try:
|
|
215
|
+
# `set_config(...)`, not `SET ... = %s`. PostgreSQL's SET is a utility statement whose
|
|
216
|
+
# value is parsed as a literal, so a bind parameter in it is a syntax error at "$1" —
|
|
217
|
+
# every connection carrying an application_name failed outright. set_config is the
|
|
218
|
+
# function form and takes ordinary parameters, which also keeps the value off the
|
|
219
|
+
# statement string entirely.
|
|
220
|
+
if statement_timeout_ms is not None:
|
|
221
|
+
timeout = str(int(statement_timeout_ms))
|
|
222
|
+
cursor.execute("SELECT set_config('statement_timeout', %s, false)", (timeout,))
|
|
223
|
+
cursor.execute("SELECT set_config('lock_timeout', %s, false)", (timeout,))
|
|
224
|
+
if application_name is not None:
|
|
225
|
+
cursor.execute(
|
|
226
|
+
"SELECT set_config('application_name', %s, false)", (application_name,)
|
|
227
|
+
)
|
|
228
|
+
finally:
|
|
229
|
+
cursor.close()
|
weightsdb/errors.py
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""weightsdb.errors — the database error hierarchy.
|
|
2
|
+
|
|
3
|
+
The contract every application in the suite maps its own database-layer failures to
|
|
4
|
+
([spec §13](../../../docs/packages/weightsdb/spec.md)). Extracted from FreeWeight's
|
|
5
|
+
``infrastructure.db.errors``, which was written "as if it were WeightsDB's own error module" for
|
|
6
|
+
exactly this move; the codes below are unchanged by the extraction.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import ClassVar
|
|
12
|
+
|
|
13
|
+
from baseaicore import SuiteError
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"DatabaseError",
|
|
17
|
+
"DatabaseUnavailable",
|
|
18
|
+
"MigrationFailed",
|
|
19
|
+
"MigrationRequired",
|
|
20
|
+
"SchemaAhead",
|
|
21
|
+
"StorageBusy",
|
|
22
|
+
"StorageFull",
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class DatabaseError(SuiteError):
|
|
27
|
+
"""Base for every error raised by this package."""
|
|
28
|
+
|
|
29
|
+
code: ClassVar[str] = "DATABASE_ERROR"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class MigrationRequired(DatabaseError):
|
|
33
|
+
"""The schema is behind head and this dialect does not auto-migrate.
|
|
34
|
+
|
|
35
|
+
``details`` always carries ``current``, ``head`` and ``command`` so the caller can print the
|
|
36
|
+
exact upgrade invocation without composing it itself.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
code: ClassVar[str] = "MIGRATION_REQUIRED"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class MigrationFailed(DatabaseError):
|
|
43
|
+
"""A migration raised partway through ``upgrade`` or ``downgrade``.
|
|
44
|
+
|
|
45
|
+
On SQLite the pre-migration backup has already been restored by the time this is raised, and
|
|
46
|
+
``details["restored"]`` is ``True``. On PostgreSQL no automatic restore is attempted (spec
|
|
47
|
+
§11.4); ``details`` instead names the revision reached and the backup to restore from.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
code: ClassVar[str] = "MIGRATION_FAILED"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class DatabaseUnavailable(DatabaseError):
|
|
54
|
+
"""The configured database could not be reached or opened.
|
|
55
|
+
|
|
56
|
+
``details["database_url"]`` is always redacted (credentials stripped, see
|
|
57
|
+
:func:`weightsdb.redaction.redact_url`) before this error is constructed — never pass a raw URL
|
|
58
|
+
here.
|
|
59
|
+
"""
|
|
60
|
+
|
|
61
|
+
code: ClassVar[str] = "DATABASE_UNAVAILABLE"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class SchemaAhead(DatabaseError):
|
|
65
|
+
"""The database's revision is newer than any this build knows about.
|
|
66
|
+
|
|
67
|
+
Means the database was written by a later application version; running against it would risk
|
|
68
|
+
misreading rows this build's models do not describe.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
code: ClassVar[str] = "SCHEMA_AHEAD"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class StorageBusy(DatabaseError):
|
|
75
|
+
"""SQLite reported ``SQLITE_BUSY`` beyond the configured ``busy_timeout``."""
|
|
76
|
+
|
|
77
|
+
code: ClassVar[str] = "STORAGE_BUSY"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class StorageFull(DatabaseError):
|
|
81
|
+
"""The device backing the database or its backup directory is out of space."""
|
|
82
|
+
|
|
83
|
+
code: ClassVar[str] = "STORAGE_FULL"
|
weightsdb/health.py
ADDED
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
"""weightsdb.health — the ``database`` component every application's health endpoint reports from.
|
|
2
|
+
|
|
3
|
+
Never raises: a health check that itself crashes takes the whole health endpoint down with it,
|
|
4
|
+
which is exactly the outcome graceful degradation exists to prevent (the same convention FreeWeight
|
|
5
|
+
and LoadCoach already apply at their own health-report layer, moved one level down so both build
|
|
6
|
+
their ``database`` component from a single, shared implementation — spec §17).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import shutil
|
|
12
|
+
import time
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import TYPE_CHECKING, Literal
|
|
16
|
+
|
|
17
|
+
from sqlalchemy import text
|
|
18
|
+
from sqlalchemy.exc import SQLAlchemyError
|
|
19
|
+
|
|
20
|
+
from weightsdb.backup import database_size_bytes, integrity_check
|
|
21
|
+
|
|
22
|
+
if TYPE_CHECKING:
|
|
23
|
+
from sqlalchemy import Engine
|
|
24
|
+
|
|
25
|
+
from weightsdb.migrations import MigrationRunner
|
|
26
|
+
|
|
27
|
+
__all__ = ["DatabaseHealth", "database_health", "is_network_filesystem"]
|
|
28
|
+
|
|
29
|
+
type ComponentStatus = Literal["ok", "degraded", "unavailable"]
|
|
30
|
+
|
|
31
|
+
# No configuration reaches this module (spec §12 — WeightsDB reads no config file or environment
|
|
32
|
+
# variable), so the two judgement calls below are fixed, documented constants rather than tunable
|
|
33
|
+
# parameters. An application that wants a different bar reports raw values itself instead of
|
|
34
|
+
# calling this function — see docs/adoption-checklist.md.
|
|
35
|
+
_LOW_DISK_FLOOR_BYTES = 100 * 1024 * 1024 # 100 MiB
|
|
36
|
+
_STALE_BACKUP_AFTER_SECONDS = 7 * 24 * 3600 # 7 days
|
|
37
|
+
|
|
38
|
+
_NETWORK_FILESYSTEM_TYPES = frozenset(
|
|
39
|
+
{
|
|
40
|
+
"nfs",
|
|
41
|
+
"nfs4",
|
|
42
|
+
"cifs",
|
|
43
|
+
"smb",
|
|
44
|
+
"smbfs",
|
|
45
|
+
"smb2",
|
|
46
|
+
"9p",
|
|
47
|
+
"afs",
|
|
48
|
+
"afpfs",
|
|
49
|
+
"fuse.sshfs",
|
|
50
|
+
"glusterfs",
|
|
51
|
+
"ceph",
|
|
52
|
+
"davfs",
|
|
53
|
+
}
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass(frozen=True, slots=True)
|
|
58
|
+
class DatabaseHealth:
|
|
59
|
+
"""The full ``database`` health snapshot, dialect-portable.
|
|
60
|
+
|
|
61
|
+
Attributes:
|
|
62
|
+
dialect: ``"sqlite"`` or ``"postgresql"``.
|
|
63
|
+
version: The backend's own version string, or ``None`` if it could not be read.
|
|
64
|
+
current_revision: The database's current Alembic revision, or ``None`` — either unmigrated,
|
|
65
|
+
or ``runner`` was not given.
|
|
66
|
+
head_revision: The revision this build's migrations produce, or ``None`` when ``runner``
|
|
67
|
+
was not given.
|
|
68
|
+
is_at_head: Whether ``current_revision == head_revision``, or ``None`` when ``runner`` was
|
|
69
|
+
not given.
|
|
70
|
+
journal_mode: SQLite's journal mode (normally ``"wal"``), or ``None`` on PostgreSQL.
|
|
71
|
+
size_bytes: How much disk the database occupies.
|
|
72
|
+
free_space_bytes: Free space on the containing device, or ``None`` if it could not be read.
|
|
73
|
+
last_backup_age_seconds: Age of the newest file under the conventional
|
|
74
|
+
``<sqlite file>/../backups`` directory, or ``None`` on PostgreSQL (which has no such
|
|
75
|
+
convention) or when no backup has ever been taken.
|
|
76
|
+
integrity_ok: Whether the integrity check passed.
|
|
77
|
+
integrity_detail: The backend's own integrity report.
|
|
78
|
+
network_filesystem: Whether the SQLite file appears to live on a network filesystem
|
|
79
|
+
(spec §16); always ``None`` on PostgreSQL and whenever this cannot be determined
|
|
80
|
+
(non-Linux, ``/proc/mounts`` unreadable) — ``None`` means "undetermined", not "no".
|
|
81
|
+
status: The overall verdict.
|
|
82
|
+
degraded_reasons: Every condition that contributed to a non-``ok`` status, empty when
|
|
83
|
+
``status == "ok"``.
|
|
84
|
+
"""
|
|
85
|
+
|
|
86
|
+
dialect: str
|
|
87
|
+
version: str | None
|
|
88
|
+
current_revision: str | None
|
|
89
|
+
head_revision: str | None
|
|
90
|
+
is_at_head: bool | None
|
|
91
|
+
journal_mode: str | None
|
|
92
|
+
size_bytes: int
|
|
93
|
+
free_space_bytes: int | None
|
|
94
|
+
last_backup_age_seconds: float | None
|
|
95
|
+
integrity_ok: bool
|
|
96
|
+
integrity_detail: str
|
|
97
|
+
network_filesystem: bool | None
|
|
98
|
+
status: ComponentStatus
|
|
99
|
+
degraded_reasons: tuple[str, ...]
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _read_proc_mounts() -> list[tuple[str, str]] | None:
|
|
103
|
+
"""Return ``(mount_point, fs_type)`` pairs from ``/proc/mounts``, or ``None`` if unreadable."""
|
|
104
|
+
try:
|
|
105
|
+
with Path("/proc/mounts").open(encoding="utf-8") as handle:
|
|
106
|
+
lines = handle.readlines()
|
|
107
|
+
except OSError:
|
|
108
|
+
return None
|
|
109
|
+
entries: list[tuple[str, str]] = []
|
|
110
|
+
for line in lines:
|
|
111
|
+
parts = line.split()
|
|
112
|
+
if len(parts) >= 3:
|
|
113
|
+
entries.append((parts[1], parts[2]))
|
|
114
|
+
return entries
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def is_network_filesystem(
|
|
118
|
+
path: Path, *, mounts: list[tuple[str, str]] | None = None
|
|
119
|
+
) -> bool | None:
|
|
120
|
+
"""Return whether ``path`` appears to live on a network filesystem (spec §16).
|
|
121
|
+
|
|
122
|
+
Best-effort and Linux-only: reads ``/proc/mounts`` (or the ``mounts`` table given, for tests)
|
|
123
|
+
and finds the longest-prefix mount point containing ``path``, then checks its filesystem type
|
|
124
|
+
against a list of known network filesystem types (NFS, CIFS/SMB, 9p, and similar).
|
|
125
|
+
|
|
126
|
+
Args:
|
|
127
|
+
path: The file to check. Need not exist yet — only its ancestry is examined.
|
|
128
|
+
mounts: ``(mount_point, fs_type)`` pairs, as read from ``/proc/mounts``. ``None`` reads the
|
|
129
|
+
real table; a test supplies one directly rather than mocking the filesystem.
|
|
130
|
+
|
|
131
|
+
Returns:
|
|
132
|
+
``True``/``False`` when determinable; ``None`` when it cannot be (no ``mounts`` table
|
|
133
|
+
available, or no entry contains ``path``) — undetermined, not "no".
|
|
134
|
+
"""
|
|
135
|
+
entries = mounts if mounts is not None else _read_proc_mounts()
|
|
136
|
+
if not entries:
|
|
137
|
+
return None
|
|
138
|
+
resolved = path.absolute()
|
|
139
|
+
best_match: str | None = None
|
|
140
|
+
best_depth = -1
|
|
141
|
+
for mount_point, fs_type in entries:
|
|
142
|
+
mount_path = Path(mount_point)
|
|
143
|
+
if mount_path != resolved and mount_path not in resolved.parents:
|
|
144
|
+
continue
|
|
145
|
+
depth = len(mount_path.parts)
|
|
146
|
+
if depth > best_depth:
|
|
147
|
+
best_depth = depth
|
|
148
|
+
best_match = fs_type
|
|
149
|
+
if best_match is None:
|
|
150
|
+
return None
|
|
151
|
+
return best_match.lower() in _NETWORK_FILESYSTEM_TYPES
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _backend_version(engine: Engine) -> str | None:
|
|
155
|
+
try:
|
|
156
|
+
with engine.connect() as connection:
|
|
157
|
+
if engine.dialect.name == "sqlite":
|
|
158
|
+
return str(connection.execute(text("SELECT sqlite_version()")).scalar_one())
|
|
159
|
+
return str(connection.execute(text("SHOW server_version")).scalar_one())
|
|
160
|
+
except SQLAlchemyError:
|
|
161
|
+
return None
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _journal_mode(engine: Engine) -> str | None:
|
|
165
|
+
if engine.dialect.name != "sqlite":
|
|
166
|
+
return None
|
|
167
|
+
try:
|
|
168
|
+
with engine.connect() as connection:
|
|
169
|
+
return str(connection.execute(text("PRAGMA journal_mode")).scalar_one())
|
|
170
|
+
except SQLAlchemyError:
|
|
171
|
+
return None
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _free_space_bytes(engine: Engine) -> int | None:
|
|
175
|
+
try:
|
|
176
|
+
if engine.dialect.name == "sqlite":
|
|
177
|
+
database = engine.url.database
|
|
178
|
+
if not database or database == ":memory:":
|
|
179
|
+
return None
|
|
180
|
+
target = Path(database).parent
|
|
181
|
+
else:
|
|
182
|
+
# No local path for a remote PostgreSQL server; report on the current process's own
|
|
183
|
+
# filesystem as the best available proxy — an application that needs the *server's*
|
|
184
|
+
# free space monitors that server directly.
|
|
185
|
+
target = Path.cwd()
|
|
186
|
+
while not target.exists():
|
|
187
|
+
parent = target.parent
|
|
188
|
+
if parent == target:
|
|
189
|
+
return None
|
|
190
|
+
target = parent
|
|
191
|
+
return shutil.disk_usage(target).free
|
|
192
|
+
except OSError:
|
|
193
|
+
return None
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _last_backup_age_seconds(engine: Engine, *, now: float) -> float | None:
|
|
197
|
+
if engine.dialect.name != "sqlite":
|
|
198
|
+
return None
|
|
199
|
+
database = engine.url.database
|
|
200
|
+
if not database or database == ":memory:":
|
|
201
|
+
return None
|
|
202
|
+
backups_directory = Path(database).parent / "backups"
|
|
203
|
+
if not backups_directory.is_dir():
|
|
204
|
+
return None
|
|
205
|
+
newest: float | None = None
|
|
206
|
+
for candidate in backups_directory.iterdir():
|
|
207
|
+
if not candidate.is_file():
|
|
208
|
+
continue
|
|
209
|
+
mtime = candidate.stat().st_mtime
|
|
210
|
+
if newest is None or mtime > newest:
|
|
211
|
+
newest = mtime
|
|
212
|
+
if newest is None:
|
|
213
|
+
return None
|
|
214
|
+
return max(0.0, now - newest)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _network_filesystem_for(engine: Engine) -> bool | None:
|
|
218
|
+
if engine.dialect.name != "sqlite":
|
|
219
|
+
return None
|
|
220
|
+
database = engine.url.database
|
|
221
|
+
if not database or database == ":memory:":
|
|
222
|
+
return None
|
|
223
|
+
return is_network_filesystem(Path(database))
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def database_health(engine: Engine, runner: MigrationRunner | None = None) -> DatabaseHealth:
|
|
227
|
+
"""Build the full ``database`` health snapshot for ``engine``.
|
|
228
|
+
|
|
229
|
+
Args:
|
|
230
|
+
engine: The engine to report on — normally the same one the application is serving from
|
|
231
|
+
(spec §17: this function supplies the ``database`` component of every application's
|
|
232
|
+
health endpoint).
|
|
233
|
+
runner: The application's own :class:`~weightsdb.migrations.MigrationRunner`, for the
|
|
234
|
+
revision-vs-head comparison. ``None`` skips it — ``current_revision``, ``head_revision``
|
|
235
|
+
and ``is_at_head`` are reported as ``None`` rather than raising, since a package that
|
|
236
|
+
owns no schema has no revision history of its own to fall back on.
|
|
237
|
+
|
|
238
|
+
Returns:
|
|
239
|
+
The :class:`DatabaseHealth`. Never raises: a database that cannot be reached at all is
|
|
240
|
+
reported as ``status="unavailable"``, not an exception.
|
|
241
|
+
"""
|
|
242
|
+
dialect = engine.dialect.name
|
|
243
|
+
try:
|
|
244
|
+
integrity = integrity_check(engine)
|
|
245
|
+
except Exception as exc: # noqa: BLE001 — a health check must not itself raise
|
|
246
|
+
return DatabaseHealth(
|
|
247
|
+
dialect=dialect,
|
|
248
|
+
version=None,
|
|
249
|
+
current_revision=None,
|
|
250
|
+
head_revision=None,
|
|
251
|
+
is_at_head=None,
|
|
252
|
+
journal_mode=None,
|
|
253
|
+
size_bytes=0,
|
|
254
|
+
free_space_bytes=None,
|
|
255
|
+
last_backup_age_seconds=None,
|
|
256
|
+
integrity_ok=False,
|
|
257
|
+
integrity_detail="unreachable",
|
|
258
|
+
network_filesystem=None,
|
|
259
|
+
status="unavailable",
|
|
260
|
+
degraded_reasons=(f"database unreachable: {exc}",),
|
|
261
|
+
)
|
|
262
|
+
|
|
263
|
+
version = _backend_version(engine)
|
|
264
|
+
journal_mode = _journal_mode(engine)
|
|
265
|
+
size_bytes = database_size_bytes(engine)
|
|
266
|
+
free_space = _free_space_bytes(engine)
|
|
267
|
+
backup_age = _last_backup_age_seconds(engine, now=time.time())
|
|
268
|
+
network_fs = _network_filesystem_for(engine)
|
|
269
|
+
|
|
270
|
+
current_revision: str | None = None
|
|
271
|
+
head_revision: str | None = None
|
|
272
|
+
is_at_head: bool | None = None
|
|
273
|
+
is_ahead_of_head = False
|
|
274
|
+
if runner is not None:
|
|
275
|
+
current_revision = runner.current()
|
|
276
|
+
heads = runner.heads()
|
|
277
|
+
head_revision = heads[0] if heads else None
|
|
278
|
+
is_at_head = current_revision == head_revision if head_revision is not None else None
|
|
279
|
+
is_ahead_of_head = (
|
|
280
|
+
current_revision is not None
|
|
281
|
+
and not is_at_head
|
|
282
|
+
and current_revision not in runner.known_revisions()
|
|
283
|
+
)
|
|
284
|
+
|
|
285
|
+
reasons: list[str] = []
|
|
286
|
+
if is_ahead_of_head:
|
|
287
|
+
reasons.append(
|
|
288
|
+
f"database is ahead of this build: at {current_revision!r}, a revision this build's "
|
|
289
|
+
"migrations do not produce — it was likely written by a newer application version"
|
|
290
|
+
)
|
|
291
|
+
elif is_at_head is False:
|
|
292
|
+
reasons.append(f"pending migration: at {current_revision!r}, head is {head_revision!r}")
|
|
293
|
+
if not integrity.ok:
|
|
294
|
+
reasons.append(f"integrity check failed: {integrity.detail}")
|
|
295
|
+
if free_space is not None and free_space < _LOW_DISK_FLOOR_BYTES:
|
|
296
|
+
reasons.append(f"low disk space: {free_space} bytes free")
|
|
297
|
+
if backup_age is not None and backup_age > _STALE_BACKUP_AFTER_SECONDS:
|
|
298
|
+
reasons.append(f"stale backup: last one is {backup_age / 86400:.1f} days old")
|
|
299
|
+
if network_fs:
|
|
300
|
+
reasons.append("database file appears to be on a network filesystem")
|
|
301
|
+
|
|
302
|
+
status: ComponentStatus = "degraded" if reasons else "ok"
|
|
303
|
+
return DatabaseHealth(
|
|
304
|
+
dialect=dialect,
|
|
305
|
+
version=version,
|
|
306
|
+
current_revision=current_revision,
|
|
307
|
+
head_revision=head_revision,
|
|
308
|
+
is_at_head=is_at_head,
|
|
309
|
+
journal_mode=journal_mode,
|
|
310
|
+
size_bytes=size_bytes,
|
|
311
|
+
free_space_bytes=free_space,
|
|
312
|
+
last_backup_age_seconds=backup_age,
|
|
313
|
+
integrity_ok=integrity.ok,
|
|
314
|
+
integrity_detail=integrity.detail,
|
|
315
|
+
network_filesystem=network_fs,
|
|
316
|
+
status=status,
|
|
317
|
+
degraded_reasons=tuple(reasons),
|
|
318
|
+
)
|