django-aiogram 4.0.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- django_aiogram/__init__.py +64 -0
- django_aiogram/_singleton.py +22 -0
- django_aiogram/admin.py +380 -0
- django_aiogram/api.py +38 -0
- django_aiogram/apps.py +56 -0
- django_aiogram/config/__init__.py +15 -0
- django_aiogram/config/checks.py +759 -0
- django_aiogram/config/defaults.py +102 -0
- django_aiogram/config/enums.py +108 -0
- django_aiogram/config/settings.py +266 -0
- django_aiogram/consumer/__init__.py +13 -0
- django_aiogram/consumer/delivery.py +625 -0
- django_aiogram/consumer/routers.py +19 -0
- django_aiogram/consumer/webhook.py +154 -0
- django_aiogram/context.py +34 -0
- django_aiogram/eventlog/__init__.py +16 -0
- django_aiogram/eventlog/dbrouter.py +55 -0
- django_aiogram/eventlog/events.py +120 -0
- django_aiogram/eventlog/instrumentation.py +231 -0
- django_aiogram/eventlog/recorder.py +922 -0
- django_aiogram/eventlog/signals.py +84 -0
- django_aiogram/eventlog/writer.py +231 -0
- django_aiogram/exceptions.py +60 -0
- django_aiogram/healthcheck.py +412 -0
- django_aiogram/management/__init__.py +1 -0
- django_aiogram/management/commands/__init__.py +1 -0
- django_aiogram/management/commands/start_tgbot.py +308 -0
- django_aiogram/management/commands/tgbot_healthcheck.py +57 -0
- django_aiogram/management/commands/tgbot_prune_events.py +144 -0
- django_aiogram/management/commands/tgbot_reclaim.py +135 -0
- django_aiogram/management/commands/tgbot_webhook.py +87 -0
- django_aiogram/migrations/0001_initial.py +50 -0
- django_aiogram/migrations/0002_kind_id_index.py +32 -0
- django_aiogram/migrations/__init__.py +1 -0
- django_aiogram/models.py +79 -0
- django_aiogram/producer/__init__.py +13 -0
- django_aiogram/producer/client.py +1540 -0
- django_aiogram/producer/throttling.py +336 -0
- django_aiogram/py.typed +0 -0
- django_aiogram/redis.py +394 -0
- django_aiogram/wire/__init__.py +14 -0
- django_aiogram/wire/envelope.py +146 -0
- django_aiogram/wire/payloads.py +195 -0
- django_aiogram/wire/serializers.py +533 -0
- django_aiogram-4.0.0.dev0.dist-info/METADATA +145 -0
- django_aiogram-4.0.0.dev0.dist-info/RECORD +48 -0
- django_aiogram-4.0.0.dev0.dist-info/WHEEL +4 -0
- django_aiogram-4.0.0.dev0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,922 @@
|
|
|
1
|
+
"""Record what happened, without making anyone wait for the database.
|
|
2
|
+
|
|
3
|
+
Every seam that records runs somewhere the ORM cannot be used directly: the
|
|
4
|
+
send coroutine runs on the bot's event loop, the delivery consumer runs on a
|
|
5
|
+
thread nothing manages a connection for, and a Django view must not pay a
|
|
6
|
+
second round trip to log the one it just made. All of them hand an :class:`Event`
|
|
7
|
+
to a bounded queue that one writer thread drains in batches.
|
|
8
|
+
|
|
9
|
+
``record()`` reaches only ``Queue.put_nowait`` — a lock, a deque append and a
|
|
10
|
+
notify. Nothing in that chain is decorated ``@async_unsafe``, which is what
|
|
11
|
+
makes it legal from a coroutine with no ``sync_to_async`` and no
|
|
12
|
+
``SynchronousOnlyOperation``. One setting suspends that, and only one:
|
|
13
|
+
``EVENT_LOG_SYNC`` inserts on the calling thread on purpose, which is why it is
|
|
14
|
+
documented as a testing setting and why it declines to act inside a running loop.
|
|
15
|
+
|
|
16
|
+
Going through the queue also avoids what a synchronous insert would do
|
|
17
|
+
inside a caller's ``atomic()`` block: on PostgreSQL a failed statement aborts
|
|
18
|
+
the whole transaction, so logging would corrupt the caller's data.
|
|
19
|
+
|
|
20
|
+
This module must not import ``django.db``. :mod:`django_aiogram.eventlog.writer`
|
|
21
|
+
does, and the writer thread imports it on its first flush — on its *first write*,
|
|
22
|
+
more precisely, because since 3.1.0 the writer also runs with the log off, for
|
|
23
|
+
``events_recorded`` receivers alone, and such a process never reaches it at all.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
import asyncio
|
|
27
|
+
import atexit
|
|
28
|
+
import contextlib
|
|
29
|
+
import logging
|
|
30
|
+
import os
|
|
31
|
+
import queue
|
|
32
|
+
import threading
|
|
33
|
+
import time
|
|
34
|
+
import uuid
|
|
35
|
+
from collections.abc import Callable
|
|
36
|
+
from dataclasses import dataclass, field, replace
|
|
37
|
+
from typing import Any
|
|
38
|
+
|
|
39
|
+
from django.core.exceptions import ImproperlyConfigured
|
|
40
|
+
from django.core.signals import setting_changed
|
|
41
|
+
|
|
42
|
+
from django_aiogram.config.defaults import DEFAULTS
|
|
43
|
+
from django_aiogram.config.enums import EventKind
|
|
44
|
+
from django_aiogram.config.settings import SETTINGS_NAME, coerce_bool, conf
|
|
45
|
+
from django_aiogram.eventlog.events import known_kinds, new_correlation_id, worker_identity
|
|
46
|
+
from django_aiogram.eventlog.signals import events_recorded
|
|
47
|
+
|
|
48
|
+
logger = logging.getLogger('django_aiogram')
|
|
49
|
+
|
|
50
|
+
#: the writer's thread name, so a log line or a test can name it
|
|
51
|
+
WRITER_THREAD = 'tgbot-event-writer'
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
#: what a signed BIGINT holds, which is the width of every id column here
|
|
55
|
+
ID_RANGE = range(-(2**63), 2**63)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def as_identifier(value: object) -> int | None:
|
|
59
|
+
"""Keep what a BIGINT column can hold, and nothing else.
|
|
60
|
+
|
|
61
|
+
A Telegram chat_id may be a `@username`, which is a valid destination and
|
|
62
|
+
not a number; `True` is an int to Python and not an id to anyone; and a
|
|
63
|
+
Python integer has no width, so one off an untrusted queue can be wider
|
|
64
|
+
than the column and cost the row it was meant to describe.
|
|
65
|
+
"""
|
|
66
|
+
if isinstance(value, bool) or not isinstance(value, int):
|
|
67
|
+
return None
|
|
68
|
+
return value if value in ID_RANGE else None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
#: how long stop() waits for the writer before giving up on what it holds
|
|
72
|
+
STOP_TIMEOUT = 5.0
|
|
73
|
+
#: consecutive failed flushes after which the writer stops trying for a while
|
|
74
|
+
FAILURE_LIMIT = 5
|
|
75
|
+
#: how long it drains and discards before probing the database again
|
|
76
|
+
FAILURE_BACKOFF = 60.0
|
|
77
|
+
#: how often the drop counter is allowed to reach the log
|
|
78
|
+
DROP_REPORT_INTERVAL = 60.0
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass(frozen=True)
|
|
82
|
+
class Wake:
|
|
83
|
+
"""A marker that ends the writer's current wait.
|
|
84
|
+
|
|
85
|
+
``done`` is what makes :meth:`EventRecorder.flush` honest: the queue going
|
|
86
|
+
empty means the writer has *taken* the batch, not that it has written it.
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
done: threading.Event | None = None
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass(frozen=True)
|
|
93
|
+
class Event:
|
|
94
|
+
"""One thing that happened. Indexed columns first, the rest in ``detail``."""
|
|
95
|
+
|
|
96
|
+
kind: str
|
|
97
|
+
correlation_id: uuid.UUID = field(default_factory=new_correlation_id)
|
|
98
|
+
created_at: float = field(default_factory=time.time)
|
|
99
|
+
function: str = ''
|
|
100
|
+
chat_id: int | None = None
|
|
101
|
+
user_id: int | None = None
|
|
102
|
+
message_id: int | None = None
|
|
103
|
+
update_id: int | None = None
|
|
104
|
+
worker: str = ''
|
|
105
|
+
attempt: int = 0
|
|
106
|
+
duration_ms: int | None = None
|
|
107
|
+
error_code: str = ''
|
|
108
|
+
error: str = ''
|
|
109
|
+
#: already JSON-safe by the time it arrives: encoding aiogram objects is the
|
|
110
|
+
#: caller's job, because this module must stay free of aiogram
|
|
111
|
+
detail: dict[str, Any] | None = None
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _number(key: str, cast: Callable[[Any], float]) -> float:
|
|
115
|
+
"""Read one of the writer's own dials, falling back to its default.
|
|
116
|
+
|
|
117
|
+
Checks E036-E038 report a value that cannot be read, at boot and once. This
|
|
118
|
+
runs on the writer thread, in a loop, on the far side of `_flush`'s net — a
|
|
119
|
+
raise here ends the writer and takes the whole buffer with it, which is a
|
|
120
|
+
steep price for a typo in a batch size.
|
|
121
|
+
"""
|
|
122
|
+
try:
|
|
123
|
+
return cast(conf[key])
|
|
124
|
+
except (ImproperlyConfigured, KeyError, TypeError, OverflowError, ValueError):
|
|
125
|
+
# ImproperlyConfigured from resolving the settings, the rest from the cast.
|
|
126
|
+
# OverflowError is the one that is not a typo: `int(float('inf'))` raises it, and a
|
|
127
|
+
# settings dict can hold `inf` directly — the environment cannot, it is refused
|
|
128
|
+
# there — so without this the writer thread ends on a value E044 only reports
|
|
129
|
+
return cast(DEFAULTS[key])
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _receiver_name(receiver: object) -> str:
|
|
133
|
+
"""Name a signal receiver for a log line, without calling anything that can raise.
|
|
134
|
+
|
|
135
|
+
``repr()`` is deliberately not the fallback. Python evaluates every argument
|
|
136
|
+
before the call, so ``getattr(receiver, '__qualname__', repr(receiver))``
|
|
137
|
+
evaluates ``repr`` *even when the attribute is there* — and a receiver whose
|
|
138
|
+
``__repr__`` raises would then take this line, and with it the rest of the batch's
|
|
139
|
+
receivers, out through :meth:`EventRecorder._flush`'s ``except``, where it would
|
|
140
|
+
be counted as a failed write. That is the failure this whole method exists to
|
|
141
|
+
contain, arriving through the code that reports it.
|
|
142
|
+
|
|
143
|
+
``type(receiver).__name__`` is the last resort because reading it runs nothing.
|
|
144
|
+
"""
|
|
145
|
+
for attribute in ('__qualname__', '__name__'):
|
|
146
|
+
name = getattr(receiver, attribute, None)
|
|
147
|
+
if isinstance(name, str) and name:
|
|
148
|
+
return name
|
|
149
|
+
return type(receiver).__name__
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _acknowledge(wakes: list[Wake]) -> None:
|
|
153
|
+
"""Release everything waiting on this batch."""
|
|
154
|
+
for wake in wakes:
|
|
155
|
+
if wake.done is not None:
|
|
156
|
+
wake.done.set()
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class EventRecorder:
|
|
160
|
+
"""A bounded queue, and the one thread that drains it into the database."""
|
|
161
|
+
|
|
162
|
+
def __init__(self) -> None:
|
|
163
|
+
"""Hold nothing: no setting is read and no thread starts until the first event."""
|
|
164
|
+
self._queue: queue.Queue[Event | Wake] | None = None
|
|
165
|
+
self._thread: threading.Thread | None = None
|
|
166
|
+
self._guard = threading.Lock()
|
|
167
|
+
self._stopping = threading.Event()
|
|
168
|
+
self._enabled: bool | None = None
|
|
169
|
+
self._kinds: frozenset[str] | None = None
|
|
170
|
+
self._worker: str | None = None
|
|
171
|
+
self._owner_pid = os.getpid()
|
|
172
|
+
self._fork_hook = False
|
|
173
|
+
self._dropped = 0
|
|
174
|
+
# which threads have handed a batch to the ORM, and so have a connection to
|
|
175
|
+
# close on the way out. Per thread rather than one flag: this object is
|
|
176
|
+
# process-wide, `close_old_connections()` acts on the calling thread, and two
|
|
177
|
+
# writers can overlap — a `stop()` whose join times out leaves the old one
|
|
178
|
+
# running while a replacement starts
|
|
179
|
+
self._touched_database: set[int] = set()
|
|
180
|
+
# its own lock, not _guard: _guard is held across starting a thread, and the
|
|
181
|
+
# counter and the touch set are reached from paths that must not wait on that
|
|
182
|
+
self._counter = threading.Lock()
|
|
183
|
+
# far enough back that the first drop always reports: monotonic() is
|
|
184
|
+
# time since boot on Linux, so a fresh container starts it near zero
|
|
185
|
+
self._reported_at = -DROP_REPORT_INTERVAL
|
|
186
|
+
|
|
187
|
+
@property
|
|
188
|
+
def enabled(self) -> bool:
|
|
189
|
+
"""Whether this process writes the event feed at all."""
|
|
190
|
+
# one read, kept local: a reset() between two reads would return None
|
|
191
|
+
enabled = self._enabled
|
|
192
|
+
if enabled is None:
|
|
193
|
+
enabled = self._enabled = self._read_flag()
|
|
194
|
+
return enabled
|
|
195
|
+
|
|
196
|
+
@property
|
|
197
|
+
def active(self) -> bool:
|
|
198
|
+
"""Whether anything at all is reading events: the table, a receiver, or both.
|
|
199
|
+
|
|
200
|
+
This is the gate every seam that *produces* an event belongs behind, and
|
|
201
|
+
:attr:`enabled` is not — a project that connects a receiver and leaves the
|
|
202
|
+
table off must still get its events, and gating on the table alone is how
|
|
203
|
+
an advertised metric comes out silently empty.
|
|
204
|
+
|
|
205
|
+
``bool(receivers)`` rather than ``has_listeners()``: 7ns against 172ns
|
|
206
|
+
measured, and this is read once per event. The difference is a receiver
|
|
207
|
+
whose weak reference has died but whose entry has not been cleaned up yet,
|
|
208
|
+
which makes this answer yes while nothing listens — so an event is recorded
|
|
209
|
+
that nobody reads, and ``send_robust`` short-circuits on it. Wasted work,
|
|
210
|
+
never a wrong row. ``tests/test_metrics_seam.py`` pins the attribute, since
|
|
211
|
+
it is Django's to rename.
|
|
212
|
+
"""
|
|
213
|
+
return self.enabled or bool(events_recorded.receivers)
|
|
214
|
+
|
|
215
|
+
@property
|
|
216
|
+
def wants_payload(self) -> bool:
|
|
217
|
+
"""Whether anything will read a payload summary, which is the costly part.
|
|
218
|
+
|
|
219
|
+
Only the table does. ``describe()`` redacts credentials, walks the
|
|
220
|
+
structure and bounds the result — measured in tens of microseconds, against
|
|
221
|
+
nothing for a counter keyed on ``kind`` and ``function``. So unless the log
|
|
222
|
+
is on too, a receiver gets ``Event`` objects whose ``detail`` carries what
|
|
223
|
+
the seam measured itself and not the summarized arguments. Rows are what the
|
|
224
|
+
table gets; with the log off there are none.
|
|
225
|
+
"""
|
|
226
|
+
return self.enabled
|
|
227
|
+
|
|
228
|
+
@property
|
|
229
|
+
def worker(self) -> str:
|
|
230
|
+
"""Name the process recording, cached: gethostname() is a system call."""
|
|
231
|
+
# one read, kept local, for the same reason `enabled` does it
|
|
232
|
+
worker = self._worker
|
|
233
|
+
if worker is None:
|
|
234
|
+
worker = self._worker = worker_identity()
|
|
235
|
+
return worker
|
|
236
|
+
|
|
237
|
+
def _read_flag(self) -> bool:
|
|
238
|
+
"""Read the flag once, treating an unreadable one as off."""
|
|
239
|
+
try:
|
|
240
|
+
return coerce_bool(conf['EVENT_LOG'], f"{SETTINGS_NAME}['EVENT_LOG']")
|
|
241
|
+
except Exception:
|
|
242
|
+
# a misconfigured flag is E031's finding at boot; at runtime it must
|
|
243
|
+
# not become the reason a message was not sent
|
|
244
|
+
logger.exception('could not read the event log flag; recording is off')
|
|
245
|
+
return False
|
|
246
|
+
|
|
247
|
+
def wants(self, kind: str) -> bool:
|
|
248
|
+
"""Whether this kind is one the project asked to keep."""
|
|
249
|
+
kinds = self._kinds
|
|
250
|
+
if kinds is None:
|
|
251
|
+
configured = conf['EVENT_LOG_KINDS'] or ()
|
|
252
|
+
kinds = self._kinds = frozenset(str(name) for name in configured) or known_kinds()
|
|
253
|
+
return kind in kinds
|
|
254
|
+
|
|
255
|
+
def record(self, event: Event) -> None:
|
|
256
|
+
"""Hand one event over. Never blocks, never raises, never touches the ORM."""
|
|
257
|
+
if not self.active:
|
|
258
|
+
return
|
|
259
|
+
try:
|
|
260
|
+
if not self.wants(event.kind):
|
|
261
|
+
return
|
|
262
|
+
if not event.worker:
|
|
263
|
+
# every row says which process recorded it, and only the
|
|
264
|
+
# consumer knew its own name before
|
|
265
|
+
event = replace(event, worker=self.worker)
|
|
266
|
+
if self._write_here():
|
|
267
|
+
# counted, which it was not: under EVENT_LOG_SYNC the row is written on
|
|
268
|
+
# this thread, and a database that refuses it raises `EventLogRefusedError`
|
|
269
|
+
# — which the broad `except` below logged and did not count, so the row
|
|
270
|
+
# vanished with no gap marker. A one-row batch either lands or raises, so
|
|
271
|
+
# there is no partial case to weigh here, only this one
|
|
272
|
+
try:
|
|
273
|
+
self._deliver([event])
|
|
274
|
+
except Exception:
|
|
275
|
+
self._drop(1)
|
|
276
|
+
logger.exception('could not record an event on the calling thread', extra={'tg_kind': event.kind})
|
|
277
|
+
finally:
|
|
278
|
+
# this thread wrote, and this thread is not the one that closes: the
|
|
279
|
+
# mark exists for the writer's exit, and a caller's mark left behind
|
|
280
|
+
# outlives the caller. Thread idents are reused, so a receiver-only
|
|
281
|
+
# writer could inherit one and close a connection it never opened —
|
|
282
|
+
# importing `eventlog`, and `django.db` with it
|
|
283
|
+
self._forget_touch()
|
|
284
|
+
return
|
|
285
|
+
buffer = self._buffer()
|
|
286
|
+
buffer.put_nowait(event)
|
|
287
|
+
if self._queue is not buffer:
|
|
288
|
+
# stop() detached this queue between the lookup and the put, and
|
|
289
|
+
# nothing will ever drain a detached one again
|
|
290
|
+
self._rehome(buffer)
|
|
291
|
+
except queue.Full:
|
|
292
|
+
self._drop(1)
|
|
293
|
+
except Exception:
|
|
294
|
+
# the recorder failing is not the caller's problem to handle
|
|
295
|
+
logger.exception('could not record an event', extra={'tg_kind': event.kind})
|
|
296
|
+
|
|
297
|
+
def _rehome(self, orphan: 'queue.Queue[Event | Wake]') -> None:
|
|
298
|
+
"""Move what is in a detached queue onto the live one.
|
|
299
|
+
|
|
300
|
+
Not "move my own event": ``get_nowait`` may hand back somebody else's, and
|
|
301
|
+
it does not matter which — what matters is that the detached queue ends up
|
|
302
|
+
empty and everything in it lands somewhere that will be drained. Two
|
|
303
|
+
producers doing this at once take disjoint items, and stop()'s own drain
|
|
304
|
+
competing with them is equally harmless.
|
|
305
|
+
|
|
306
|
+
Both halves are non-blocking, so :meth:`record`'s promise holds: an event
|
|
307
|
+
that cannot be rehomed because the live queue is full is counted, which is
|
|
308
|
+
what would have happened to it there anyway.
|
|
309
|
+
"""
|
|
310
|
+
while True:
|
|
311
|
+
try:
|
|
312
|
+
item = orphan.get_nowait()
|
|
313
|
+
except queue.Empty:
|
|
314
|
+
return
|
|
315
|
+
try:
|
|
316
|
+
self._buffer().put_nowait(item)
|
|
317
|
+
except queue.Full:
|
|
318
|
+
if isinstance(item, Wake):
|
|
319
|
+
_acknowledge([item])
|
|
320
|
+
continue
|
|
321
|
+
self._drop(1)
|
|
322
|
+
|
|
323
|
+
def _write_here(self) -> bool:
|
|
324
|
+
"""Whether to write on this thread instead of handing it to the writer.
|
|
325
|
+
|
|
326
|
+
Only when asked to, and never from inside a running loop: the ORM is
|
|
327
|
+
@async_unsafe there, so the seam that records an update would raise
|
|
328
|
+
SynchronousOnlyOperation instead of recording anything.
|
|
329
|
+
|
|
330
|
+
Requires the log as well as the flag. ``EVENT_LOG_SYNC`` is about *where the
|
|
331
|
+
insert happens*, so with nothing being inserted it has nothing to say — and
|
|
332
|
+
answering yes would have a process that only has receivers run them on the
|
|
333
|
+
thread that recorded the event, which is the one thing this design exists
|
|
334
|
+
to avoid.
|
|
335
|
+
"""
|
|
336
|
+
if not self.enabled:
|
|
337
|
+
return False
|
|
338
|
+
if not coerce_bool(conf['EVENT_LOG_SYNC'], f"{SETTINGS_NAME}['EVENT_LOG_SYNC']"):
|
|
339
|
+
return False
|
|
340
|
+
try:
|
|
341
|
+
asyncio.get_running_loop()
|
|
342
|
+
except RuntimeError:
|
|
343
|
+
return True
|
|
344
|
+
return False
|
|
345
|
+
|
|
346
|
+
def _drop(self, count: int) -> None:
|
|
347
|
+
"""Count lost events, and say so at most once a minute.
|
|
348
|
+
|
|
349
|
+
The counter is guarded because more than one thread reaches it: producers
|
|
350
|
+
on a full queue, the writer on a failed flush, and whichever thread called
|
|
351
|
+
stop() draining what the writer left. `+=` is a read and a write, so
|
|
352
|
+
without this a drop is silently swallowed by a concurrent one.
|
|
353
|
+
"""
|
|
354
|
+
if not count:
|
|
355
|
+
# callers pass the refused count straight through, and that is zero on every
|
|
356
|
+
# successful write — reporting "the event log is falling behind" for a batch
|
|
357
|
+
# that landed in full is a false alarm on the line people watch for real ones
|
|
358
|
+
return
|
|
359
|
+
with self._counter:
|
|
360
|
+
self._dropped += count
|
|
361
|
+
dropped = self._dropped
|
|
362
|
+
now = time.monotonic()
|
|
363
|
+
if now - self._reported_at < DROP_REPORT_INTERVAL:
|
|
364
|
+
return
|
|
365
|
+
self._reported_at = now
|
|
366
|
+
logger.error(
|
|
367
|
+
'the event log is falling behind; events are being dropped',
|
|
368
|
+
extra={'tg_dropped': dropped},
|
|
369
|
+
)
|
|
370
|
+
|
|
371
|
+
def _buffer(self) -> queue.Queue[Event | Wake]:
|
|
372
|
+
"""Return the queue, starting the writer the first time anything is recorded."""
|
|
373
|
+
if self._owner_pid != os.getpid():
|
|
374
|
+
# a thread does not survive fork(), but the queue object does, so a
|
|
375
|
+
# child would fill one nobody drains
|
|
376
|
+
self._forget()
|
|
377
|
+
buffer = self._queue
|
|
378
|
+
if buffer is not None:
|
|
379
|
+
return buffer
|
|
380
|
+
with self._guard:
|
|
381
|
+
if self._queue is None:
|
|
382
|
+
self._install_fork_hook()
|
|
383
|
+
self._stopping.clear()
|
|
384
|
+
self._owner_pid = os.getpid()
|
|
385
|
+
buffer = queue.Queue(maxsize=max(1, int(_number('EVENT_LOG_BUFFER_SIZE', int))))
|
|
386
|
+
thread = threading.Thread(target=self._run, args=(buffer,), name=WRITER_THREAD, daemon=True)
|
|
387
|
+
self._queue, self._thread = buffer, thread
|
|
388
|
+
try:
|
|
389
|
+
thread.start()
|
|
390
|
+
except RuntimeError:
|
|
391
|
+
# out of threads: leave nothing half-built for the next call,
|
|
392
|
+
# and count the event this loses so the gap row still says so
|
|
393
|
+
self._queue = self._thread = None
|
|
394
|
+
self._drop(1)
|
|
395
|
+
raise
|
|
396
|
+
# CPython runs atexit callbacks while daemon threads are still
|
|
397
|
+
# alive, so the writer is still joinable from one
|
|
398
|
+
atexit.register(self.stop)
|
|
399
|
+
return self._queue
|
|
400
|
+
|
|
401
|
+
def _install_fork_hook(self) -> None:
|
|
402
|
+
"""Reset in the child as well as on the pid check, where the platform allows."""
|
|
403
|
+
if self._fork_hook or not hasattr(os, 'register_at_fork'):
|
|
404
|
+
return
|
|
405
|
+
self._fork_hook = True
|
|
406
|
+
os.register_at_fork(after_in_child=self._forget)
|
|
407
|
+
|
|
408
|
+
def _forget(self) -> None:
|
|
409
|
+
"""Drop everything a fork invalidated, so the next event starts fresh."""
|
|
410
|
+
# a new lock: the parent may have held this one at the moment of the fork
|
|
411
|
+
self._guard = threading.Lock()
|
|
412
|
+
self._counter = threading.Lock()
|
|
413
|
+
self._queue = None
|
|
414
|
+
self._thread = None
|
|
415
|
+
self._owner_pid = os.getpid()
|
|
416
|
+
self._touched_database = set()
|
|
417
|
+
self._dropped = 0
|
|
418
|
+
self._reported_at = -DROP_REPORT_INTERVAL
|
|
419
|
+
|
|
420
|
+
def _run(self, buffer: queue.Queue[Event | Wake]) -> None:
|
|
421
|
+
"""Drain the queue into the database until stopped.
|
|
422
|
+
|
|
423
|
+
A thread target: anything escaping it would end recording for the life
|
|
424
|
+
of the process, so the slot is cleared on the way out and the next
|
|
425
|
+
record() starts a replacement.
|
|
426
|
+
"""
|
|
427
|
+
failures = 0
|
|
428
|
+
blocked_until = 0.0
|
|
429
|
+
try:
|
|
430
|
+
while True:
|
|
431
|
+
batch, wakes = self._collect(buffer)
|
|
432
|
+
try:
|
|
433
|
+
if batch:
|
|
434
|
+
if time.monotonic() < blocked_until:
|
|
435
|
+
# the database has been refusing us; keep draining so
|
|
436
|
+
# producers never fill up, but do not hammer it
|
|
437
|
+
self._drop(len(batch))
|
|
438
|
+
else:
|
|
439
|
+
failures, blocked_until = self._flush(batch, failures=failures)
|
|
440
|
+
finally:
|
|
441
|
+
# after the write, never before: a waiter released early was
|
|
442
|
+
# told the batch was durable while it was still in flight
|
|
443
|
+
_acknowledge(wakes)
|
|
444
|
+
# not `wakes and ...`: the wake `stop()` queues is the usual way this
|
|
445
|
+
# thread learns, but it is not the only one — `stop()` called from a
|
|
446
|
+
# receiver runs on *this* thread and drains this very buffer through
|
|
447
|
+
# `_abandon`, taking that wake with it. The loop then never saw one and
|
|
448
|
+
# spun for the life of the process, holding a connection.
|
|
449
|
+
#
|
|
450
|
+
# `_queue is not buffer` is the other half, and it is per writer where the
|
|
451
|
+
# flag is not: `stop()` detaches this queue and sets `_stopping`, and a
|
|
452
|
+
# `record()` that lands next calls `_buffer()`, which *clears* the flag and
|
|
453
|
+
# starts a replacement. This writer then saw an empty detached queue with
|
|
454
|
+
# the flag down and waited on it for the life of the process. Its own
|
|
455
|
+
# buffer no longer being the recorder's queue says the same thing and
|
|
456
|
+
# cannot be undone by anybody else
|
|
457
|
+
if buffer.empty() and (self._stopping.is_set() or self._queue is not buffer):
|
|
458
|
+
return
|
|
459
|
+
except Exception:
|
|
460
|
+
logger.exception('the event writer stopped; it restarts on the next event')
|
|
461
|
+
finally:
|
|
462
|
+
with self._guard:
|
|
463
|
+
if self._queue is buffer:
|
|
464
|
+
self._queue = self._thread = None
|
|
465
|
+
# the slot is cleared above, so nothing will ever drain this queue
|
|
466
|
+
# again: without this, everything still in it disappears with no row
|
|
467
|
+
# and no counter, and the gap reads as quiet traffic
|
|
468
|
+
self._abandon(buffer)
|
|
469
|
+
if self._took_the_touch():
|
|
470
|
+
# a process that only has receivers never opened one, and importing
|
|
471
|
+
# `eventlog` to close it would pull in `django.db` — the one import
|
|
472
|
+
# this module exists to keep out of a process that does not need it
|
|
473
|
+
self._close_connections()
|
|
474
|
+
|
|
475
|
+
def _forget_touch(self) -> None:
|
|
476
|
+
"""Drop this thread's mark without acting on it, for a thread that does not close.
|
|
477
|
+
|
|
478
|
+
Only the writer's exit closes connections. `record()` under ``EVENT_LOG_SYNC`` and
|
|
479
|
+
`drain_once()` write on their caller's thread, and Django owns that thread's
|
|
480
|
+
connection — so their marks are bookkeeping nobody reads, and idents get reused.
|
|
481
|
+
"""
|
|
482
|
+
with self._counter:
|
|
483
|
+
self._touched_database.discard(threading.get_ident())
|
|
484
|
+
|
|
485
|
+
def _took_the_touch(self) -> bool:
|
|
486
|
+
"""Whether *this* thread handed a batch to the ORM, clearing the mark as it answers.
|
|
487
|
+
|
|
488
|
+
Read and cleared together, because the mark describes this writer: left set it
|
|
489
|
+
outlives the thread that earned it, and a later writer with only receivers closes
|
|
490
|
+
a connection it never opened — importing `eventlog`, and with it `django.db`, into
|
|
491
|
+
the one process this module exists to keep it out of. Only a fork cleared it before,
|
|
492
|
+
so a process that wrote once and then had the log turned off carried it for good.
|
|
493
|
+
|
|
494
|
+
Per thread, because one flag was not enough either: `stop()` detaches the queue
|
|
495
|
+
before joining, so a join that times out leaves the old writer running while a
|
|
496
|
+
replacement starts, and the old one's exit cleared the new one's flag — the
|
|
497
|
+
replacement then skipped closing the connection it had opened. It is also what the
|
|
498
|
+
mark always meant, since `close_old_connections()` acts on the calling thread.
|
|
499
|
+
"""
|
|
500
|
+
ident = threading.get_ident()
|
|
501
|
+
with self._counter:
|
|
502
|
+
touched = ident in self._touched_database
|
|
503
|
+
self._touched_database.discard(ident)
|
|
504
|
+
return touched
|
|
505
|
+
|
|
506
|
+
@staticmethod
|
|
507
|
+
def _empty(buffer: 'queue.Queue[Event | Wake]') -> tuple[list[Event], list[Wake]]:
|
|
508
|
+
"""Take everything left in a queue, without waiting for more."""
|
|
509
|
+
events: list[Event] = []
|
|
510
|
+
wakes: list[Wake] = []
|
|
511
|
+
while True:
|
|
512
|
+
try:
|
|
513
|
+
item = buffer.get_nowait()
|
|
514
|
+
except queue.Empty:
|
|
515
|
+
return events, wakes
|
|
516
|
+
if isinstance(item, Wake):
|
|
517
|
+
wakes.append(item)
|
|
518
|
+
else:
|
|
519
|
+
events.append(item)
|
|
520
|
+
|
|
521
|
+
def _abandon(self, buffer: 'queue.Queue[Event | Wake]') -> None:
|
|
522
|
+
"""Write what is left in a queue nobody will drain again, or count it lost.
|
|
523
|
+
|
|
524
|
+
**Receivers run on whatever thread calls this**, which is the writer's own
|
|
525
|
+
when it is exiting and the caller's when :meth:`stop` reached a queue the
|
|
526
|
+
writer had already left behind. That is not a lapse in the writer-thread
|
|
527
|
+
contract so much as the end of it: this queue exists precisely because no
|
|
528
|
+
writer will ever drain it, so there is no writer thread to route through.
|
|
529
|
+
|
|
530
|
+
Publishing anyway rather than dropping, because these are the last events
|
|
531
|
+
before the process goes — the same reasoning that makes this method write
|
|
532
|
+
them instead of discarding them. The contract says so on all three surfaces
|
|
533
|
+
that state it.
|
|
534
|
+
"""
|
|
535
|
+
leftover, wakes = self._empty(buffer)
|
|
536
|
+
_acknowledge(wakes)
|
|
537
|
+
if not leftover:
|
|
538
|
+
return
|
|
539
|
+
try:
|
|
540
|
+
# the refused count, not only the raise: a database that takes some of these
|
|
541
|
+
# rows and refuses others leaves a hole exactly as large as what it refused,
|
|
542
|
+
# and ignoring the return counted that hole as zero
|
|
543
|
+
refused = self._deliver(leftover)
|
|
544
|
+
except Exception:
|
|
545
|
+
logger.exception('could not write the events a stopping writer left behind')
|
|
546
|
+
else:
|
|
547
|
+
self._drop(refused)
|
|
548
|
+
return
|
|
549
|
+
# counted, not silent: the next flush that succeeds turns this into a
|
|
550
|
+
# log.dropped row, which is the only place the gap becomes visible
|
|
551
|
+
self._drop(len(leftover))
|
|
552
|
+
|
|
553
|
+
@staticmethod
|
|
554
|
+
def flush_interval() -> int:
|
|
555
|
+
"""Seconds before a partial batch is written anyway, as ``E038`` defines it.
|
|
556
|
+
|
|
557
|
+
An integer, matching the check and the settings page. Read as a float this
|
|
558
|
+
honoured a fractional interval the check refuses, so a value could pass
|
|
559
|
+
``manage.py check`` and then behave in a way the check called impossible — one
|
|
560
|
+
setting with two rules. Named rather than inline so the rule has one reader and a
|
|
561
|
+
test can ask it directly.
|
|
562
|
+
"""
|
|
563
|
+
return int(max(1, _number('EVENT_LOG_FLUSH_INTERVAL', int)))
|
|
564
|
+
|
|
565
|
+
def _collect(self, buffer: queue.Queue[Event | Wake]) -> tuple[list[Event], list[Wake]]:
|
|
566
|
+
"""Gather up to one batch, with any wake-ups that ended the wait."""
|
|
567
|
+
interval = self.flush_interval()
|
|
568
|
+
limit = max(1, int(_number('EVENT_LOG_BATCH_SIZE', int)))
|
|
569
|
+
deadline = time.monotonic() + interval
|
|
570
|
+
batch: list[Event] = []
|
|
571
|
+
wakes: list[Wake] = []
|
|
572
|
+
while len(batch) < limit:
|
|
573
|
+
remaining = deadline - time.monotonic()
|
|
574
|
+
if remaining <= 0:
|
|
575
|
+
break
|
|
576
|
+
try:
|
|
577
|
+
item = buffer.get(timeout=remaining)
|
|
578
|
+
except queue.Empty:
|
|
579
|
+
break
|
|
580
|
+
if isinstance(item, Wake):
|
|
581
|
+
wakes.append(item)
|
|
582
|
+
break
|
|
583
|
+
batch.append(item)
|
|
584
|
+
return batch, wakes
|
|
585
|
+
|
|
586
|
+
def _flush(self, batch: list[Event], *, failures: int) -> tuple[int, float]:
|
|
587
|
+
"""Write one batch and publish it, containing whatever the write raises.
|
|
588
|
+
|
|
589
|
+
Both happen inside :meth:`_deliver`, which writes first and publishes in a
|
|
590
|
+
``finally`` — so a failing database costs rows and not metrics, and nothing
|
|
591
|
+
reaching the ``except`` here came from a receiver: :meth:`_publish` cannot
|
|
592
|
+
raise.
|
|
593
|
+
|
|
594
|
+
Which means **receivers run on whatever thread calls this**, and that is not
|
|
595
|
+
only the writer's: :meth:`drain_once` calls it on the caller's, which is what
|
|
596
|
+
lets a test drive the real flush path. The signal's own documentation states
|
|
597
|
+
the rule that way round rather than listing the threads, so a fourth one does
|
|
598
|
+
not make it wrong.
|
|
599
|
+
"""
|
|
600
|
+
# under the counter's lock, both of them: `_drop`'s docstring already names
|
|
601
|
+
# "the writer on a failed flush" among the threads it protects against, and
|
|
602
|
+
# this was the one place that read and wrote the count without taking it —
|
|
603
|
+
# so a producer's drop landing between this `+=`'s read and its write was
|
|
604
|
+
# silently discarded, and the `log.dropped` row then under-reported the gap
|
|
605
|
+
with self._counter:
|
|
606
|
+
dropped_before = self._dropped
|
|
607
|
+
try:
|
|
608
|
+
refused = self._deliver(batch)
|
|
609
|
+
except Exception:
|
|
610
|
+
failures += 1
|
|
611
|
+
with self._counter:
|
|
612
|
+
self._dropped += len(batch)
|
|
613
|
+
# one line per failure, not two: the suspension is a different
|
|
614
|
+
# sentence about the same exception, not a second thing that broke
|
|
615
|
+
if failures >= FAILURE_LIMIT:
|
|
616
|
+
logger.exception(
|
|
617
|
+
'the event log is suspended after repeated failures; run migrate or check the database',
|
|
618
|
+
extra={'tg_count': len(batch), 'tg_failures': failures},
|
|
619
|
+
)
|
|
620
|
+
return 0, time.monotonic() + FAILURE_BACKOFF
|
|
621
|
+
logger.exception(
|
|
622
|
+
'could not write an event batch',
|
|
623
|
+
extra={'tg_count': len(batch), 'tg_failures': failures},
|
|
624
|
+
)
|
|
625
|
+
return failures, 0.0
|
|
626
|
+
if refused:
|
|
627
|
+
# rows this very batch lost, one at a time, on the ladder below `write_batch`.
|
|
628
|
+
# Counted rather than reported now: the gap row belongs to the *next*
|
|
629
|
+
# successful flush, the same way a producer's drop does
|
|
630
|
+
with self._counter:
|
|
631
|
+
self._dropped += refused
|
|
632
|
+
logger.warning(
|
|
633
|
+
'the database refused part of an event batch',
|
|
634
|
+
extra={'tg_count': refused, 'tg_batch': len(batch)},
|
|
635
|
+
)
|
|
636
|
+
if dropped_before:
|
|
637
|
+
self._record_gap(dropped_before)
|
|
638
|
+
return 0, 0.0
|
|
639
|
+
|
|
640
|
+
def _record_gap(self, dropped: int) -> None:
|
|
641
|
+
"""Put the gap in the feed, not only in the log: a silent hole reads as coverage.
|
|
642
|
+
|
|
643
|
+
**Claimed, then written, and given back if the write fails.** Two flushes can be
|
|
644
|
+
in progress at once — the writer thread's and a ``drain_once()`` on somebody
|
|
645
|
+
else's — and both snapshot the drop count before their batch. Subtracting after
|
|
646
|
+
the write let each of them report the same hole and take it off twice, which
|
|
647
|
+
drives the count negative; subtracting before the write, which is where this
|
|
648
|
+
started, lost the hole whenever the gap row itself was refused. Taking the count
|
|
649
|
+
out of the counter first makes the claim exclusive, and putting it back on failure
|
|
650
|
+
keeps it for the next flush. Both properties, one lock.
|
|
651
|
+
|
|
652
|
+
Claims no more than is there: a count that another flush has already taken leaves
|
|
653
|
+
nothing to report, and this returns rather than writing a row about zero events.
|
|
654
|
+
Anything a producer drops while the write is in flight stays for the next one.
|
|
655
|
+
|
|
656
|
+
A refusal counts as a failure here, not only an exception. ``_deliver`` returns how
|
|
657
|
+
many rows the database refused one at a time, and this batch is one row — so a
|
|
658
|
+
return of 1 means the gap row did *not* land, which is the same loss as a raise and
|
|
659
|
+
was the one path this method used to ignore. Both give the claim back.
|
|
660
|
+
|
|
661
|
+
The failure stays suppressed either way: the batch this follows did land, and a
|
|
662
|
+
gap row that cannot be written must not turn a successful flush into a failed one.
|
|
663
|
+
"""
|
|
664
|
+
with self._counter:
|
|
665
|
+
claimed = min(dropped, self._dropped)
|
|
666
|
+
self._dropped -= claimed
|
|
667
|
+
if not claimed:
|
|
668
|
+
return
|
|
669
|
+
try:
|
|
670
|
+
refused = self._deliver([Event(kind=EventKind.LOG_DROPPED.value, detail={'dropped': claimed})])
|
|
671
|
+
except Exception:
|
|
672
|
+
self._reclaim(claimed)
|
|
673
|
+
logger.exception('could not record the gap; keeping the count for the next flush')
|
|
674
|
+
return
|
|
675
|
+
if refused:
|
|
676
|
+
self._reclaim(claimed)
|
|
677
|
+
logger.error(
|
|
678
|
+
'the database refused the gap row; keeping the count for the next flush',
|
|
679
|
+
extra={'tg_dropped': claimed},
|
|
680
|
+
)
|
|
681
|
+
|
|
682
|
+
def _reclaim(self, claimed: int) -> None:
|
|
683
|
+
"""Put a claim back, so a gap nobody could record survives to be recorded."""
|
|
684
|
+
with self._counter:
|
|
685
|
+
self._dropped += claimed
|
|
686
|
+
|
|
687
|
+
@staticmethod
|
|
688
|
+
def _write(batch: list[Event]) -> int:
|
|
689
|
+
"""Hand a batch to the ORM, importing it here so a disabled process never does.
|
|
690
|
+
|
|
691
|
+
Returns how many rows did not land, which only a partial refusal produces.
|
|
692
|
+
"""
|
|
693
|
+
from django_aiogram.eventlog.writer import write_batch # noqa: PLC0415 - the point: no django.db above
|
|
694
|
+
|
|
695
|
+
return write_batch(batch)
|
|
696
|
+
|
|
697
|
+
def _publish(self, batch: list[Event]) -> None:
|
|
698
|
+
"""Hand a batch to whoever connected to :data:`events_recorded`.
|
|
699
|
+
|
|
700
|
+
``send_robust``, so one broken receiver neither loses the batch for the
|
|
701
|
+
others nor stops the writer, and it is logged here because a receiver that
|
|
702
|
+
fails silently is a metric that reads as zero traffic. Django logs it too, on
|
|
703
|
+
its own ``django.dispatch`` logger; the line here is on the logger a project
|
|
704
|
+
configures for this package, which is where it will actually be seen.
|
|
705
|
+
|
|
706
|
+
**Wrapped anyway, because ``send_robust`` does not contain everything.**
|
|
707
|
+
Django's own failure logging reads ``receiver.__qualname__`` unguarded, and a
|
|
708
|
+
callable *instance* — an ordinary shape for a metrics collector — has no such
|
|
709
|
+
attribute. So a receiver like that raising makes ``send_robust`` itself raise
|
|
710
|
+
``AttributeError``, measured on Django 6.1, and without this ``try`` it would
|
|
711
|
+
land in :meth:`_flush`'s ``except`` and be counted as a failed *write*: the
|
|
712
|
+
other receivers lose the batch, a ``log.dropped`` row appears, and the log
|
|
713
|
+
blames the database for something a receiver did. Containing it here makes
|
|
714
|
+
this method's promise true whatever Django does with it, on any supported
|
|
715
|
+
version.
|
|
716
|
+
|
|
717
|
+
The upshot is a method that **cannot raise**, which is the property the rest
|
|
718
|
+
of the writer needs from it rather than a defensive habit.
|
|
719
|
+
|
|
720
|
+
One limit worth stating, because it is Django's and not ours: when
|
|
721
|
+
``send_robust`` raises on that unnamed receiver it abandons **its own loop**, so
|
|
722
|
+
receivers connected after the offending one do not run for that batch at all.
|
|
723
|
+
Containing it here keeps the write and every earlier receiver whole; it cannot
|
|
724
|
+
reach past Django into a dispatch that already stopped. Calling receivers
|
|
725
|
+
ourselves would need ``Signal._live_receivers``, a private API, which is a worse
|
|
726
|
+
trade than one documented sentence. A collector written as a callable instance
|
|
727
|
+
can close the gap on its side by defining ``__qualname__``; one written as a
|
|
728
|
+
function or a bound method has it already, and is the shape every recipe uses.
|
|
729
|
+
|
|
730
|
+
A tuple rather than the list itself: receivers run one after another with
|
|
731
|
+
the same argument, so one of them sorting or clearing a list would decide
|
|
732
|
+
what the next one sees.
|
|
733
|
+
"""
|
|
734
|
+
if not events_recorded.receivers:
|
|
735
|
+
return
|
|
736
|
+
# the reporting loop is inside the guard as well as the dispatch, because
|
|
737
|
+
# `getattr(..., None)` absorbs only `AttributeError` — a receiver whose
|
|
738
|
+
# `__getattr__` raises anything else makes naming it raise, and the whole
|
|
739
|
+
# point is that nothing about a receiver reaches `_flush`'s failure counter
|
|
740
|
+
try:
|
|
741
|
+
for receiver, outcome in events_recorded.send_robust(sender=self, events=tuple(batch)):
|
|
742
|
+
if isinstance(outcome, BaseException):
|
|
743
|
+
logger.error(
|
|
744
|
+
'an events_recorded receiver raised',
|
|
745
|
+
exc_info=outcome,
|
|
746
|
+
extra={'tg_receiver': _receiver_name(receiver), 'tg_count': len(batch)},
|
|
747
|
+
)
|
|
748
|
+
except Exception:
|
|
749
|
+
# even this is suppressed: `logger.exception` is `logger.error` with
|
|
750
|
+
# `exc_info`, so a project whose handler or formatter raises would take
|
|
751
|
+
# the fallback out too — and the whole purpose here is that **nothing**
|
|
752
|
+
# about publishing reaches `_flush`'s failure counter, where it would be
|
|
753
|
+
# reported as a database refusing a batch it never saw
|
|
754
|
+
with contextlib.suppress(Exception):
|
|
755
|
+
logger.exception('publishing recorded events failed', extra={'tg_count': len(batch)})
|
|
756
|
+
|
|
757
|
+
def _deliver(self, batch: list[Event]) -> int:
|
|
758
|
+
"""Write a batch if this process keeps the table, then publish it either way.
|
|
759
|
+
|
|
760
|
+
Returns how many rows the database refused individually — zero unless a partial
|
|
761
|
+
refusal happened, and zero in a process that does not write at all.
|
|
762
|
+
|
|
763
|
+
|
|
764
|
+
The order is the contract, and it is two claims rather than one. The write
|
|
765
|
+
is **attempted first**, so nothing a receiver does can change a row that was
|
|
766
|
+
written — which is why receivers get the real ``Event`` objects and not
|
|
767
|
+
copies. And the publish is in a ``finally``, so a write that *failed* still
|
|
768
|
+
reaches them: a database that is down or unmigrated is exactly when someone
|
|
769
|
+
is watching a dashboard, and the metrics have no reason to go with it. A
|
|
770
|
+
receiver seeing a batch is therefore not evidence that a row exists for it.
|
|
771
|
+
|
|
772
|
+
Publishing first was the original order, and it handed receivers the same
|
|
773
|
+
list and the same ``Event`` objects the ORM was about to read. A frozen
|
|
774
|
+
dataclass does not freeze the ``detail`` dict inside it, so a receiver
|
|
775
|
+
clearing the list or editing a ``detail`` changed what got persisted. This
|
|
776
|
+
way round makes that impossible instead of asking receivers to be careful.
|
|
777
|
+
"""
|
|
778
|
+
refused = 0
|
|
779
|
+
try:
|
|
780
|
+
if self.enabled:
|
|
781
|
+
with self._counter:
|
|
782
|
+
self._touched_database.add(threading.get_ident())
|
|
783
|
+
refused = self._write(batch)
|
|
784
|
+
finally:
|
|
785
|
+
self._publish(batch)
|
|
786
|
+
return refused
|
|
787
|
+
|
|
788
|
+
@staticmethod
|
|
789
|
+
def _close_connections() -> None:
|
|
790
|
+
"""Release the writer thread's own connection on the way out."""
|
|
791
|
+
try:
|
|
792
|
+
from django_aiogram.eventlog.writer import close_connections # noqa: PLC0415 - as above
|
|
793
|
+
except Exception:
|
|
794
|
+
logger.exception('could not import the event log to close its connection')
|
|
795
|
+
return
|
|
796
|
+
close_connections()
|
|
797
|
+
|
|
798
|
+
def drain_once(self, timeout: float = 0.0) -> int:
|
|
799
|
+
"""Write whatever is buffered, on the calling thread. Returns events processed.
|
|
800
|
+
|
|
801
|
+
Events taken off the queue, not rows that landed: `_flush` swallows a failed write
|
|
802
|
+
and the refused count `_deliver` returns is dropped here, so a batch the database
|
|
803
|
+
rejected still counts. A caller reading this as a durability signal is reading the
|
|
804
|
+
wrong number — the gap rows and `log.dropped` are what say what was lost.
|
|
805
|
+
|
|
806
|
+
Goes through the same flush the writer uses, gap recording included, so
|
|
807
|
+
a test driving this exercises the path production takes.
|
|
808
|
+
"""
|
|
809
|
+
buffer = self._queue
|
|
810
|
+
if buffer is None:
|
|
811
|
+
return 0
|
|
812
|
+
batch: list[Event] = []
|
|
813
|
+
wakes: list[Wake] = []
|
|
814
|
+
deadline = time.monotonic() + timeout
|
|
815
|
+
while True:
|
|
816
|
+
try:
|
|
817
|
+
item = buffer.get(timeout=max(0.0, deadline - time.monotonic())) if timeout else buffer.get_nowait()
|
|
818
|
+
except queue.Empty:
|
|
819
|
+
break
|
|
820
|
+
if isinstance(item, Wake):
|
|
821
|
+
wakes.append(item)
|
|
822
|
+
continue
|
|
823
|
+
batch.append(item)
|
|
824
|
+
try:
|
|
825
|
+
if batch:
|
|
826
|
+
self._flush(batch, failures=0)
|
|
827
|
+
finally:
|
|
828
|
+
# same reason as the synchronous `record()` path: this thread wrote, and it is
|
|
829
|
+
# not the thread whose exit closes connections
|
|
830
|
+
self._forget_touch()
|
|
831
|
+
_acknowledge(wakes)
|
|
832
|
+
return len(batch)
|
|
833
|
+
|
|
834
|
+
def flush(self, timeout: float = STOP_TIMEOUT) -> None:
|
|
835
|
+
"""Wait until what has been recorded so far has reached the database.
|
|
836
|
+
|
|
837
|
+
Waits on an acknowledgement from the writer rather than on the queue
|
|
838
|
+
going empty: the queue empties when a batch is *taken*, which is before
|
|
839
|
+
it is written, so polling it would return mid-insert.
|
|
840
|
+
"""
|
|
841
|
+
buffer = self._queue
|
|
842
|
+
if buffer is None:
|
|
843
|
+
return
|
|
844
|
+
done = threading.Event()
|
|
845
|
+
try:
|
|
846
|
+
buffer.put_nowait(Wake(done))
|
|
847
|
+
except queue.Full:
|
|
848
|
+
# no room even for the marker, so there is nothing to wait behind
|
|
849
|
+
return
|
|
850
|
+
if not done.wait(timeout):
|
|
851
|
+
logger.warning('the event writer did not flush in time', extra={'tg_timeout': timeout})
|
|
852
|
+
|
|
853
|
+
def stop(self, timeout: float = STOP_TIMEOUT) -> None:
|
|
854
|
+
"""Flush and end the writer. Idempotent: atexit and start_tgbot both call it."""
|
|
855
|
+
with self._guard:
|
|
856
|
+
buffer, thread = self._queue, self._thread
|
|
857
|
+
self._queue = self._thread = None
|
|
858
|
+
if buffer is None:
|
|
859
|
+
return
|
|
860
|
+
with contextlib.suppress(Exception):
|
|
861
|
+
atexit.unregister(self.stop)
|
|
862
|
+
self._stopping.set()
|
|
863
|
+
with contextlib.suppress(queue.Full):
|
|
864
|
+
buffer.put_nowait(Wake())
|
|
865
|
+
if thread is not None and thread is not threading.current_thread():
|
|
866
|
+
# a thread that never started cannot be joined, and this runs from
|
|
867
|
+
# atexit where raising is noise nobody can act on
|
|
868
|
+
with contextlib.suppress(RuntimeError):
|
|
869
|
+
thread.join(timeout)
|
|
870
|
+
if thread.is_alive():
|
|
871
|
+
logger.warning('the event writer did not finish in time', extra={'tg_timeout': timeout})
|
|
872
|
+
elif thread is not None:
|
|
873
|
+
# `stop()` from a receiver, which runs on the writer's own thread: joining
|
|
874
|
+
# would be waiting for itself, and the old code reported that as a writer
|
|
875
|
+
# that missed its deadline. It has not missed anything — it unwinds through
|
|
876
|
+
# the loop below as soon as this returns
|
|
877
|
+
logger.debug('stop() was called on the writer thread; it will unwind on its own')
|
|
878
|
+
# a record() that read self._queue before the swap above puts into a queue
|
|
879
|
+
# this method has already detached, and nothing else will ever look at it.
|
|
880
|
+
# Draining after the join is what keeps those events; the few instructions
|
|
881
|
+
# between this drain and the producer's put stay a gap, because closing it
|
|
882
|
+
# would mean a lock on the one path that may never wait
|
|
883
|
+
try:
|
|
884
|
+
self._abandon(buffer)
|
|
885
|
+
finally:
|
|
886
|
+
# the same rule as the synchronous `record()` and `drain_once()`: this ran on
|
|
887
|
+
# whoever called `stop()`, and that thread is not the one whose exit closes
|
|
888
|
+
# connections. Left behind, its mark can be inherited by a later
|
|
889
|
+
# receiver-only writer through a reused ident.
|
|
890
|
+
#
|
|
891
|
+
# Unless `stop()` was called *from* the writer, where `_run` is still on the
|
|
892
|
+
# stack below and has that mark to consume on its way out. Reached by a
|
|
893
|
+
# receiver calling `stop()`, since receivers run on that thread —
|
|
894
|
+
# `test_a_receiver_that_stops_the_log_does_not_strand_the_writer` covers both
|
|
895
|
+
# halves and fails without either
|
|
896
|
+
if threading.current_thread() is not thread:
|
|
897
|
+
self._forget_touch()
|
|
898
|
+
|
|
899
|
+
def reset(self) -> None:
|
|
900
|
+
"""Re-read the settings next time; used by override_settings.
|
|
901
|
+
|
|
902
|
+
It does not flush. Every ``override_settings(TELEGRAM_BOT=...)`` in a
|
|
903
|
+
consumer's own test suite fires this twice, and waiting for the writer
|
|
904
|
+
there would put a second on each one. A test that needs its rows calls
|
|
905
|
+
:meth:`flush`; queued events survive the reset either way.
|
|
906
|
+
"""
|
|
907
|
+
self._enabled = None
|
|
908
|
+
self._kinds = None
|
|
909
|
+
self._worker = None
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
recorder = EventRecorder()
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
def _reset_on_setting_change(setting: str, **_kwargs: object) -> None:
|
|
916
|
+
"""Drop the cached flags after writing whatever was recorded under the old ones."""
|
|
917
|
+
if setting == SETTINGS_NAME:
|
|
918
|
+
recorder.reset()
|
|
919
|
+
|
|
920
|
+
|
|
921
|
+
# dispatch_uid keeps autoreload from stacking duplicate receivers
|
|
922
|
+
setting_changed.connect(_reset_on_setting_change, dispatch_uid='django_aiogram.eventlog.recorder')
|