django-aiogram 4.0.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. django_aiogram/__init__.py +64 -0
  2. django_aiogram/_singleton.py +22 -0
  3. django_aiogram/admin.py +380 -0
  4. django_aiogram/api.py +38 -0
  5. django_aiogram/apps.py +56 -0
  6. django_aiogram/config/__init__.py +15 -0
  7. django_aiogram/config/checks.py +759 -0
  8. django_aiogram/config/defaults.py +102 -0
  9. django_aiogram/config/enums.py +108 -0
  10. django_aiogram/config/settings.py +266 -0
  11. django_aiogram/consumer/__init__.py +13 -0
  12. django_aiogram/consumer/delivery.py +625 -0
  13. django_aiogram/consumer/routers.py +19 -0
  14. django_aiogram/consumer/webhook.py +154 -0
  15. django_aiogram/context.py +34 -0
  16. django_aiogram/eventlog/__init__.py +16 -0
  17. django_aiogram/eventlog/dbrouter.py +55 -0
  18. django_aiogram/eventlog/events.py +120 -0
  19. django_aiogram/eventlog/instrumentation.py +231 -0
  20. django_aiogram/eventlog/recorder.py +922 -0
  21. django_aiogram/eventlog/signals.py +84 -0
  22. django_aiogram/eventlog/writer.py +231 -0
  23. django_aiogram/exceptions.py +60 -0
  24. django_aiogram/healthcheck.py +412 -0
  25. django_aiogram/management/__init__.py +1 -0
  26. django_aiogram/management/commands/__init__.py +1 -0
  27. django_aiogram/management/commands/start_tgbot.py +308 -0
  28. django_aiogram/management/commands/tgbot_healthcheck.py +57 -0
  29. django_aiogram/management/commands/tgbot_prune_events.py +144 -0
  30. django_aiogram/management/commands/tgbot_reclaim.py +135 -0
  31. django_aiogram/management/commands/tgbot_webhook.py +87 -0
  32. django_aiogram/migrations/0001_initial.py +50 -0
  33. django_aiogram/migrations/0002_kind_id_index.py +32 -0
  34. django_aiogram/migrations/__init__.py +1 -0
  35. django_aiogram/models.py +79 -0
  36. django_aiogram/producer/__init__.py +13 -0
  37. django_aiogram/producer/client.py +1540 -0
  38. django_aiogram/producer/throttling.py +336 -0
  39. django_aiogram/py.typed +0 -0
  40. django_aiogram/redis.py +394 -0
  41. django_aiogram/wire/__init__.py +14 -0
  42. django_aiogram/wire/envelope.py +146 -0
  43. django_aiogram/wire/payloads.py +195 -0
  44. django_aiogram/wire/serializers.py +533 -0
  45. django_aiogram-4.0.0.dev0.dist-info/METADATA +145 -0
  46. django_aiogram-4.0.0.dev0.dist-info/RECORD +48 -0
  47. django_aiogram-4.0.0.dev0.dist-info/WHEEL +4 -0
  48. django_aiogram-4.0.0.dev0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,922 @@
1
+ """Record what happened, without making anyone wait for the database.
2
+
3
+ Every seam that records runs somewhere the ORM cannot be used directly: the
4
+ send coroutine runs on the bot's event loop, the delivery consumer runs on a
5
+ thread nothing manages a connection for, and a Django view must not pay a
6
+ second round trip to log the one it just made. All of them hand an :class:`Event`
7
+ to a bounded queue that one writer thread drains in batches.
8
+
9
+ ``record()`` reaches only ``Queue.put_nowait`` — a lock, a deque append and a
10
+ notify. Nothing in that chain is decorated ``@async_unsafe``, which is what
11
+ makes it legal from a coroutine with no ``sync_to_async`` and no
12
+ ``SynchronousOnlyOperation``. One setting suspends that, and only one:
13
+ ``EVENT_LOG_SYNC`` inserts on the calling thread on purpose, which is why it is
14
+ documented as a testing setting and why it declines to act inside a running loop.
15
+
16
+ Going through the queue also avoids what a synchronous insert would do
17
+ inside a caller's ``atomic()`` block: on PostgreSQL a failed statement aborts
18
+ the whole transaction, so logging would corrupt the caller's data.
19
+
20
+ This module must not import ``django.db``. :mod:`django_aiogram.eventlog.writer`
21
+ does, and the writer thread imports it on its first flush — on its *first write*,
22
+ more precisely, because since 3.1.0 the writer also runs with the log off, for
23
+ ``events_recorded`` receivers alone, and such a process never reaches it at all.
24
+ """
25
+
26
+ import asyncio
27
+ import atexit
28
+ import contextlib
29
+ import logging
30
+ import os
31
+ import queue
32
+ import threading
33
+ import time
34
+ import uuid
35
+ from collections.abc import Callable
36
+ from dataclasses import dataclass, field, replace
37
+ from typing import Any
38
+
39
+ from django.core.exceptions import ImproperlyConfigured
40
+ from django.core.signals import setting_changed
41
+
42
+ from django_aiogram.config.defaults import DEFAULTS
43
+ from django_aiogram.config.enums import EventKind
44
+ from django_aiogram.config.settings import SETTINGS_NAME, coerce_bool, conf
45
+ from django_aiogram.eventlog.events import known_kinds, new_correlation_id, worker_identity
46
+ from django_aiogram.eventlog.signals import events_recorded
47
+
48
+ logger = logging.getLogger('django_aiogram')
49
+
50
+ #: the writer's thread name, so a log line or a test can name it
51
+ WRITER_THREAD = 'tgbot-event-writer'
52
+
53
+
54
+ #: what a signed BIGINT holds, which is the width of every id column here
55
+ ID_RANGE = range(-(2**63), 2**63)
56
+
57
+
58
+ def as_identifier(value: object) -> int | None:
59
+ """Keep what a BIGINT column can hold, and nothing else.
60
+
61
+ A Telegram chat_id may be a `@username`, which is a valid destination and
62
+ not a number; `True` is an int to Python and not an id to anyone; and a
63
+ Python integer has no width, so one off an untrusted queue can be wider
64
+ than the column and cost the row it was meant to describe.
65
+ """
66
+ if isinstance(value, bool) or not isinstance(value, int):
67
+ return None
68
+ return value if value in ID_RANGE else None
69
+
70
+
71
+ #: how long stop() waits for the writer before giving up on what it holds
72
+ STOP_TIMEOUT = 5.0
73
+ #: consecutive failed flushes after which the writer stops trying for a while
74
+ FAILURE_LIMIT = 5
75
+ #: how long it drains and discards before probing the database again
76
+ FAILURE_BACKOFF = 60.0
77
+ #: how often the drop counter is allowed to reach the log
78
+ DROP_REPORT_INTERVAL = 60.0
79
+
80
+
81
+ @dataclass(frozen=True)
82
+ class Wake:
83
+ """A marker that ends the writer's current wait.
84
+
85
+ ``done`` is what makes :meth:`EventRecorder.flush` honest: the queue going
86
+ empty means the writer has *taken* the batch, not that it has written it.
87
+ """
88
+
89
+ done: threading.Event | None = None
90
+
91
+
92
+ @dataclass(frozen=True)
93
+ class Event:
94
+ """One thing that happened. Indexed columns first, the rest in ``detail``."""
95
+
96
+ kind: str
97
+ correlation_id: uuid.UUID = field(default_factory=new_correlation_id)
98
+ created_at: float = field(default_factory=time.time)
99
+ function: str = ''
100
+ chat_id: int | None = None
101
+ user_id: int | None = None
102
+ message_id: int | None = None
103
+ update_id: int | None = None
104
+ worker: str = ''
105
+ attempt: int = 0
106
+ duration_ms: int | None = None
107
+ error_code: str = ''
108
+ error: str = ''
109
+ #: already JSON-safe by the time it arrives: encoding aiogram objects is the
110
+ #: caller's job, because this module must stay free of aiogram
111
+ detail: dict[str, Any] | None = None
112
+
113
+
114
+ def _number(key: str, cast: Callable[[Any], float]) -> float:
115
+ """Read one of the writer's own dials, falling back to its default.
116
+
117
+ Checks E036-E038 report a value that cannot be read, at boot and once. This
118
+ runs on the writer thread, in a loop, on the far side of `_flush`'s net — a
119
+ raise here ends the writer and takes the whole buffer with it, which is a
120
+ steep price for a typo in a batch size.
121
+ """
122
+ try:
123
+ return cast(conf[key])
124
+ except (ImproperlyConfigured, KeyError, TypeError, OverflowError, ValueError):
125
+ # ImproperlyConfigured from resolving the settings, the rest from the cast.
126
+ # OverflowError is the one that is not a typo: `int(float('inf'))` raises it, and a
127
+ # settings dict can hold `inf` directly — the environment cannot, it is refused
128
+ # there — so without this the writer thread ends on a value E044 only reports
129
+ return cast(DEFAULTS[key])
130
+
131
+
132
+ def _receiver_name(receiver: object) -> str:
133
+ """Name a signal receiver for a log line, without calling anything that can raise.
134
+
135
+ ``repr()`` is deliberately not the fallback. Python evaluates every argument
136
+ before the call, so ``getattr(receiver, '__qualname__', repr(receiver))``
137
+ evaluates ``repr`` *even when the attribute is there* — and a receiver whose
138
+ ``__repr__`` raises would then take this line, and with it the rest of the batch's
139
+ receivers, out through :meth:`EventRecorder._flush`'s ``except``, where it would
140
+ be counted as a failed write. That is the failure this whole method exists to
141
+ contain, arriving through the code that reports it.
142
+
143
+ ``type(receiver).__name__`` is the last resort because reading it runs nothing.
144
+ """
145
+ for attribute in ('__qualname__', '__name__'):
146
+ name = getattr(receiver, attribute, None)
147
+ if isinstance(name, str) and name:
148
+ return name
149
+ return type(receiver).__name__
150
+
151
+
152
+ def _acknowledge(wakes: list[Wake]) -> None:
153
+ """Release everything waiting on this batch."""
154
+ for wake in wakes:
155
+ if wake.done is not None:
156
+ wake.done.set()
157
+
158
+
159
+ class EventRecorder:
160
+ """A bounded queue, and the one thread that drains it into the database."""
161
+
162
+ def __init__(self) -> None:
163
+ """Hold nothing: no setting is read and no thread starts until the first event."""
164
+ self._queue: queue.Queue[Event | Wake] | None = None
165
+ self._thread: threading.Thread | None = None
166
+ self._guard = threading.Lock()
167
+ self._stopping = threading.Event()
168
+ self._enabled: bool | None = None
169
+ self._kinds: frozenset[str] | None = None
170
+ self._worker: str | None = None
171
+ self._owner_pid = os.getpid()
172
+ self._fork_hook = False
173
+ self._dropped = 0
174
+ # which threads have handed a batch to the ORM, and so have a connection to
175
+ # close on the way out. Per thread rather than one flag: this object is
176
+ # process-wide, `close_old_connections()` acts on the calling thread, and two
177
+ # writers can overlap — a `stop()` whose join times out leaves the old one
178
+ # running while a replacement starts
179
+ self._touched_database: set[int] = set()
180
+ # its own lock, not _guard: _guard is held across starting a thread, and the
181
+ # counter and the touch set are reached from paths that must not wait on that
182
+ self._counter = threading.Lock()
183
+ # far enough back that the first drop always reports: monotonic() is
184
+ # time since boot on Linux, so a fresh container starts it near zero
185
+ self._reported_at = -DROP_REPORT_INTERVAL
186
+
187
+ @property
188
+ def enabled(self) -> bool:
189
+ """Whether this process writes the event feed at all."""
190
+ # one read, kept local: a reset() between two reads would return None
191
+ enabled = self._enabled
192
+ if enabled is None:
193
+ enabled = self._enabled = self._read_flag()
194
+ return enabled
195
+
196
+ @property
197
+ def active(self) -> bool:
198
+ """Whether anything at all is reading events: the table, a receiver, or both.
199
+
200
+ This is the gate every seam that *produces* an event belongs behind, and
201
+ :attr:`enabled` is not — a project that connects a receiver and leaves the
202
+ table off must still get its events, and gating on the table alone is how
203
+ an advertised metric comes out silently empty.
204
+
205
+ ``bool(receivers)`` rather than ``has_listeners()``: 7ns against 172ns
206
+ measured, and this is read once per event. The difference is a receiver
207
+ whose weak reference has died but whose entry has not been cleaned up yet,
208
+ which makes this answer yes while nothing listens — so an event is recorded
209
+ that nobody reads, and ``send_robust`` short-circuits on it. Wasted work,
210
+ never a wrong row. ``tests/test_metrics_seam.py`` pins the attribute, since
211
+ it is Django's to rename.
212
+ """
213
+ return self.enabled or bool(events_recorded.receivers)
214
+
215
+ @property
216
+ def wants_payload(self) -> bool:
217
+ """Whether anything will read a payload summary, which is the costly part.
218
+
219
+ Only the table does. ``describe()`` redacts credentials, walks the
220
+ structure and bounds the result — measured in tens of microseconds, against
221
+ nothing for a counter keyed on ``kind`` and ``function``. So unless the log
222
+ is on too, a receiver gets ``Event`` objects whose ``detail`` carries what
223
+ the seam measured itself and not the summarized arguments. Rows are what the
224
+ table gets; with the log off there are none.
225
+ """
226
+ return self.enabled
227
+
228
+ @property
229
+ def worker(self) -> str:
230
+ """Name the process recording, cached: gethostname() is a system call."""
231
+ # one read, kept local, for the same reason `enabled` does it
232
+ worker = self._worker
233
+ if worker is None:
234
+ worker = self._worker = worker_identity()
235
+ return worker
236
+
237
+ def _read_flag(self) -> bool:
238
+ """Read the flag once, treating an unreadable one as off."""
239
+ try:
240
+ return coerce_bool(conf['EVENT_LOG'], f"{SETTINGS_NAME}['EVENT_LOG']")
241
+ except Exception:
242
+ # a misconfigured flag is E031's finding at boot; at runtime it must
243
+ # not become the reason a message was not sent
244
+ logger.exception('could not read the event log flag; recording is off')
245
+ return False
246
+
247
+ def wants(self, kind: str) -> bool:
248
+ """Whether this kind is one the project asked to keep."""
249
+ kinds = self._kinds
250
+ if kinds is None:
251
+ configured = conf['EVENT_LOG_KINDS'] or ()
252
+ kinds = self._kinds = frozenset(str(name) for name in configured) or known_kinds()
253
+ return kind in kinds
254
+
255
+ def record(self, event: Event) -> None:
256
+ """Hand one event over. Never blocks, never raises, never touches the ORM."""
257
+ if not self.active:
258
+ return
259
+ try:
260
+ if not self.wants(event.kind):
261
+ return
262
+ if not event.worker:
263
+ # every row says which process recorded it, and only the
264
+ # consumer knew its own name before
265
+ event = replace(event, worker=self.worker)
266
+ if self._write_here():
267
+ # counted, which it was not: under EVENT_LOG_SYNC the row is written on
268
+ # this thread, and a database that refuses it raises `EventLogRefusedError`
269
+ # — which the broad `except` below logged and did not count, so the row
270
+ # vanished with no gap marker. A one-row batch either lands or raises, so
271
+ # there is no partial case to weigh here, only this one
272
+ try:
273
+ self._deliver([event])
274
+ except Exception:
275
+ self._drop(1)
276
+ logger.exception('could not record an event on the calling thread', extra={'tg_kind': event.kind})
277
+ finally:
278
+ # this thread wrote, and this thread is not the one that closes: the
279
+ # mark exists for the writer's exit, and a caller's mark left behind
280
+ # outlives the caller. Thread idents are reused, so a receiver-only
281
+ # writer could inherit one and close a connection it never opened —
282
+ # importing `eventlog`, and `django.db` with it
283
+ self._forget_touch()
284
+ return
285
+ buffer = self._buffer()
286
+ buffer.put_nowait(event)
287
+ if self._queue is not buffer:
288
+ # stop() detached this queue between the lookup and the put, and
289
+ # nothing will ever drain a detached one again
290
+ self._rehome(buffer)
291
+ except queue.Full:
292
+ self._drop(1)
293
+ except Exception:
294
+ # the recorder failing is not the caller's problem to handle
295
+ logger.exception('could not record an event', extra={'tg_kind': event.kind})
296
+
297
+ def _rehome(self, orphan: 'queue.Queue[Event | Wake]') -> None:
298
+ """Move what is in a detached queue onto the live one.
299
+
300
+ Not "move my own event": ``get_nowait`` may hand back somebody else's, and
301
+ it does not matter which — what matters is that the detached queue ends up
302
+ empty and everything in it lands somewhere that will be drained. Two
303
+ producers doing this at once take disjoint items, and stop()'s own drain
304
+ competing with them is equally harmless.
305
+
306
+ Both halves are non-blocking, so :meth:`record`'s promise holds: an event
307
+ that cannot be rehomed because the live queue is full is counted, which is
308
+ what would have happened to it there anyway.
309
+ """
310
+ while True:
311
+ try:
312
+ item = orphan.get_nowait()
313
+ except queue.Empty:
314
+ return
315
+ try:
316
+ self._buffer().put_nowait(item)
317
+ except queue.Full:
318
+ if isinstance(item, Wake):
319
+ _acknowledge([item])
320
+ continue
321
+ self._drop(1)
322
+
323
+ def _write_here(self) -> bool:
324
+ """Whether to write on this thread instead of handing it to the writer.
325
+
326
+ Only when asked to, and never from inside a running loop: the ORM is
327
+ @async_unsafe there, so the seam that records an update would raise
328
+ SynchronousOnlyOperation instead of recording anything.
329
+
330
+ Requires the log as well as the flag. ``EVENT_LOG_SYNC`` is about *where the
331
+ insert happens*, so with nothing being inserted it has nothing to say — and
332
+ answering yes would have a process that only has receivers run them on the
333
+ thread that recorded the event, which is the one thing this design exists
334
+ to avoid.
335
+ """
336
+ if not self.enabled:
337
+ return False
338
+ if not coerce_bool(conf['EVENT_LOG_SYNC'], f"{SETTINGS_NAME}['EVENT_LOG_SYNC']"):
339
+ return False
340
+ try:
341
+ asyncio.get_running_loop()
342
+ except RuntimeError:
343
+ return True
344
+ return False
345
+
346
+ def _drop(self, count: int) -> None:
347
+ """Count lost events, and say so at most once a minute.
348
+
349
+ The counter is guarded because more than one thread reaches it: producers
350
+ on a full queue, the writer on a failed flush, and whichever thread called
351
+ stop() draining what the writer left. `+=` is a read and a write, so
352
+ without this a drop is silently swallowed by a concurrent one.
353
+ """
354
+ if not count:
355
+ # callers pass the refused count straight through, and that is zero on every
356
+ # successful write — reporting "the event log is falling behind" for a batch
357
+ # that landed in full is a false alarm on the line people watch for real ones
358
+ return
359
+ with self._counter:
360
+ self._dropped += count
361
+ dropped = self._dropped
362
+ now = time.monotonic()
363
+ if now - self._reported_at < DROP_REPORT_INTERVAL:
364
+ return
365
+ self._reported_at = now
366
+ logger.error(
367
+ 'the event log is falling behind; events are being dropped',
368
+ extra={'tg_dropped': dropped},
369
+ )
370
+
371
+ def _buffer(self) -> queue.Queue[Event | Wake]:
372
+ """Return the queue, starting the writer the first time anything is recorded."""
373
+ if self._owner_pid != os.getpid():
374
+ # a thread does not survive fork(), but the queue object does, so a
375
+ # child would fill one nobody drains
376
+ self._forget()
377
+ buffer = self._queue
378
+ if buffer is not None:
379
+ return buffer
380
+ with self._guard:
381
+ if self._queue is None:
382
+ self._install_fork_hook()
383
+ self._stopping.clear()
384
+ self._owner_pid = os.getpid()
385
+ buffer = queue.Queue(maxsize=max(1, int(_number('EVENT_LOG_BUFFER_SIZE', int))))
386
+ thread = threading.Thread(target=self._run, args=(buffer,), name=WRITER_THREAD, daemon=True)
387
+ self._queue, self._thread = buffer, thread
388
+ try:
389
+ thread.start()
390
+ except RuntimeError:
391
+ # out of threads: leave nothing half-built for the next call,
392
+ # and count the event this loses so the gap row still says so
393
+ self._queue = self._thread = None
394
+ self._drop(1)
395
+ raise
396
+ # CPython runs atexit callbacks while daemon threads are still
397
+ # alive, so the writer is still joinable from one
398
+ atexit.register(self.stop)
399
+ return self._queue
400
+
401
+ def _install_fork_hook(self) -> None:
402
+ """Reset in the child as well as on the pid check, where the platform allows."""
403
+ if self._fork_hook or not hasattr(os, 'register_at_fork'):
404
+ return
405
+ self._fork_hook = True
406
+ os.register_at_fork(after_in_child=self._forget)
407
+
408
+ def _forget(self) -> None:
409
+ """Drop everything a fork invalidated, so the next event starts fresh."""
410
+ # a new lock: the parent may have held this one at the moment of the fork
411
+ self._guard = threading.Lock()
412
+ self._counter = threading.Lock()
413
+ self._queue = None
414
+ self._thread = None
415
+ self._owner_pid = os.getpid()
416
+ self._touched_database = set()
417
+ self._dropped = 0
418
+ self._reported_at = -DROP_REPORT_INTERVAL
419
+
420
+ def _run(self, buffer: queue.Queue[Event | Wake]) -> None:
421
+ """Drain the queue into the database until stopped.
422
+
423
+ A thread target: anything escaping it would end recording for the life
424
+ of the process, so the slot is cleared on the way out and the next
425
+ record() starts a replacement.
426
+ """
427
+ failures = 0
428
+ blocked_until = 0.0
429
+ try:
430
+ while True:
431
+ batch, wakes = self._collect(buffer)
432
+ try:
433
+ if batch:
434
+ if time.monotonic() < blocked_until:
435
+ # the database has been refusing us; keep draining so
436
+ # producers never fill up, but do not hammer it
437
+ self._drop(len(batch))
438
+ else:
439
+ failures, blocked_until = self._flush(batch, failures=failures)
440
+ finally:
441
+ # after the write, never before: a waiter released early was
442
+ # told the batch was durable while it was still in flight
443
+ _acknowledge(wakes)
444
+ # not `wakes and ...`: the wake `stop()` queues is the usual way this
445
+ # thread learns, but it is not the only one — `stop()` called from a
446
+ # receiver runs on *this* thread and drains this very buffer through
447
+ # `_abandon`, taking that wake with it. The loop then never saw one and
448
+ # spun for the life of the process, holding a connection.
449
+ #
450
+ # `_queue is not buffer` is the other half, and it is per writer where the
451
+ # flag is not: `stop()` detaches this queue and sets `_stopping`, and a
452
+ # `record()` that lands next calls `_buffer()`, which *clears* the flag and
453
+ # starts a replacement. This writer then saw an empty detached queue with
454
+ # the flag down and waited on it for the life of the process. Its own
455
+ # buffer no longer being the recorder's queue says the same thing and
456
+ # cannot be undone by anybody else
457
+ if buffer.empty() and (self._stopping.is_set() or self._queue is not buffer):
458
+ return
459
+ except Exception:
460
+ logger.exception('the event writer stopped; it restarts on the next event')
461
+ finally:
462
+ with self._guard:
463
+ if self._queue is buffer:
464
+ self._queue = self._thread = None
465
+ # the slot is cleared above, so nothing will ever drain this queue
466
+ # again: without this, everything still in it disappears with no row
467
+ # and no counter, and the gap reads as quiet traffic
468
+ self._abandon(buffer)
469
+ if self._took_the_touch():
470
+ # a process that only has receivers never opened one, and importing
471
+ # `eventlog` to close it would pull in `django.db` — the one import
472
+ # this module exists to keep out of a process that does not need it
473
+ self._close_connections()
474
+
475
+ def _forget_touch(self) -> None:
476
+ """Drop this thread's mark without acting on it, for a thread that does not close.
477
+
478
+ Only the writer's exit closes connections. `record()` under ``EVENT_LOG_SYNC`` and
479
+ `drain_once()` write on their caller's thread, and Django owns that thread's
480
+ connection — so their marks are bookkeeping nobody reads, and idents get reused.
481
+ """
482
+ with self._counter:
483
+ self._touched_database.discard(threading.get_ident())
484
+
485
+ def _took_the_touch(self) -> bool:
486
+ """Whether *this* thread handed a batch to the ORM, clearing the mark as it answers.
487
+
488
+ Read and cleared together, because the mark describes this writer: left set it
489
+ outlives the thread that earned it, and a later writer with only receivers closes
490
+ a connection it never opened — importing `eventlog`, and with it `django.db`, into
491
+ the one process this module exists to keep it out of. Only a fork cleared it before,
492
+ so a process that wrote once and then had the log turned off carried it for good.
493
+
494
+ Per thread, because one flag was not enough either: `stop()` detaches the queue
495
+ before joining, so a join that times out leaves the old writer running while a
496
+ replacement starts, and the old one's exit cleared the new one's flag — the
497
+ replacement then skipped closing the connection it had opened. It is also what the
498
+ mark always meant, since `close_old_connections()` acts on the calling thread.
499
+ """
500
+ ident = threading.get_ident()
501
+ with self._counter:
502
+ touched = ident in self._touched_database
503
+ self._touched_database.discard(ident)
504
+ return touched
505
+
506
+ @staticmethod
507
+ def _empty(buffer: 'queue.Queue[Event | Wake]') -> tuple[list[Event], list[Wake]]:
508
+ """Take everything left in a queue, without waiting for more."""
509
+ events: list[Event] = []
510
+ wakes: list[Wake] = []
511
+ while True:
512
+ try:
513
+ item = buffer.get_nowait()
514
+ except queue.Empty:
515
+ return events, wakes
516
+ if isinstance(item, Wake):
517
+ wakes.append(item)
518
+ else:
519
+ events.append(item)
520
+
521
+ def _abandon(self, buffer: 'queue.Queue[Event | Wake]') -> None:
522
+ """Write what is left in a queue nobody will drain again, or count it lost.
523
+
524
+ **Receivers run on whatever thread calls this**, which is the writer's own
525
+ when it is exiting and the caller's when :meth:`stop` reached a queue the
526
+ writer had already left behind. That is not a lapse in the writer-thread
527
+ contract so much as the end of it: this queue exists precisely because no
528
+ writer will ever drain it, so there is no writer thread to route through.
529
+
530
+ Publishing anyway rather than dropping, because these are the last events
531
+ before the process goes — the same reasoning that makes this method write
532
+ them instead of discarding them. The contract says so on all three surfaces
533
+ that state it.
534
+ """
535
+ leftover, wakes = self._empty(buffer)
536
+ _acknowledge(wakes)
537
+ if not leftover:
538
+ return
539
+ try:
540
+ # the refused count, not only the raise: a database that takes some of these
541
+ # rows and refuses others leaves a hole exactly as large as what it refused,
542
+ # and ignoring the return counted that hole as zero
543
+ refused = self._deliver(leftover)
544
+ except Exception:
545
+ logger.exception('could not write the events a stopping writer left behind')
546
+ else:
547
+ self._drop(refused)
548
+ return
549
+ # counted, not silent: the next flush that succeeds turns this into a
550
+ # log.dropped row, which is the only place the gap becomes visible
551
+ self._drop(len(leftover))
552
+
553
+ @staticmethod
554
+ def flush_interval() -> int:
555
+ """Seconds before a partial batch is written anyway, as ``E038`` defines it.
556
+
557
+ An integer, matching the check and the settings page. Read as a float this
558
+ honoured a fractional interval the check refuses, so a value could pass
559
+ ``manage.py check`` and then behave in a way the check called impossible — one
560
+ setting with two rules. Named rather than inline so the rule has one reader and a
561
+ test can ask it directly.
562
+ """
563
+ return int(max(1, _number('EVENT_LOG_FLUSH_INTERVAL', int)))
564
+
565
+ def _collect(self, buffer: queue.Queue[Event | Wake]) -> tuple[list[Event], list[Wake]]:
566
+ """Gather up to one batch, with any wake-ups that ended the wait."""
567
+ interval = self.flush_interval()
568
+ limit = max(1, int(_number('EVENT_LOG_BATCH_SIZE', int)))
569
+ deadline = time.monotonic() + interval
570
+ batch: list[Event] = []
571
+ wakes: list[Wake] = []
572
+ while len(batch) < limit:
573
+ remaining = deadline - time.monotonic()
574
+ if remaining <= 0:
575
+ break
576
+ try:
577
+ item = buffer.get(timeout=remaining)
578
+ except queue.Empty:
579
+ break
580
+ if isinstance(item, Wake):
581
+ wakes.append(item)
582
+ break
583
+ batch.append(item)
584
+ return batch, wakes
585
+
586
+ def _flush(self, batch: list[Event], *, failures: int) -> tuple[int, float]:
587
+ """Write one batch and publish it, containing whatever the write raises.
588
+
589
+ Both happen inside :meth:`_deliver`, which writes first and publishes in a
590
+ ``finally`` — so a failing database costs rows and not metrics, and nothing
591
+ reaching the ``except`` here came from a receiver: :meth:`_publish` cannot
592
+ raise.
593
+
594
+ Which means **receivers run on whatever thread calls this**, and that is not
595
+ only the writer's: :meth:`drain_once` calls it on the caller's, which is what
596
+ lets a test drive the real flush path. The signal's own documentation states
597
+ the rule that way round rather than listing the threads, so a fourth one does
598
+ not make it wrong.
599
+ """
600
+ # under the counter's lock, both of them: `_drop`'s docstring already names
601
+ # "the writer on a failed flush" among the threads it protects against, and
602
+ # this was the one place that read and wrote the count without taking it —
603
+ # so a producer's drop landing between this `+=`'s read and its write was
604
+ # silently discarded, and the `log.dropped` row then under-reported the gap
605
+ with self._counter:
606
+ dropped_before = self._dropped
607
+ try:
608
+ refused = self._deliver(batch)
609
+ except Exception:
610
+ failures += 1
611
+ with self._counter:
612
+ self._dropped += len(batch)
613
+ # one line per failure, not two: the suspension is a different
614
+ # sentence about the same exception, not a second thing that broke
615
+ if failures >= FAILURE_LIMIT:
616
+ logger.exception(
617
+ 'the event log is suspended after repeated failures; run migrate or check the database',
618
+ extra={'tg_count': len(batch), 'tg_failures': failures},
619
+ )
620
+ return 0, time.monotonic() + FAILURE_BACKOFF
621
+ logger.exception(
622
+ 'could not write an event batch',
623
+ extra={'tg_count': len(batch), 'tg_failures': failures},
624
+ )
625
+ return failures, 0.0
626
+ if refused:
627
+ # rows this very batch lost, one at a time, on the ladder below `write_batch`.
628
+ # Counted rather than reported now: the gap row belongs to the *next*
629
+ # successful flush, the same way a producer's drop does
630
+ with self._counter:
631
+ self._dropped += refused
632
+ logger.warning(
633
+ 'the database refused part of an event batch',
634
+ extra={'tg_count': refused, 'tg_batch': len(batch)},
635
+ )
636
+ if dropped_before:
637
+ self._record_gap(dropped_before)
638
+ return 0, 0.0
639
+
640
+ def _record_gap(self, dropped: int) -> None:
641
+ """Put the gap in the feed, not only in the log: a silent hole reads as coverage.
642
+
643
+ **Claimed, then written, and given back if the write fails.** Two flushes can be
644
+ in progress at once — the writer thread's and a ``drain_once()`` on somebody
645
+ else's — and both snapshot the drop count before their batch. Subtracting after
646
+ the write let each of them report the same hole and take it off twice, which
647
+ drives the count negative; subtracting before the write, which is where this
648
+ started, lost the hole whenever the gap row itself was refused. Taking the count
649
+ out of the counter first makes the claim exclusive, and putting it back on failure
650
+ keeps it for the next flush. Both properties, one lock.
651
+
652
+ Claims no more than is there: a count that another flush has already taken leaves
653
+ nothing to report, and this returns rather than writing a row about zero events.
654
+ Anything a producer drops while the write is in flight stays for the next one.
655
+
656
+ A refusal counts as a failure here, not only an exception. ``_deliver`` returns how
657
+ many rows the database refused one at a time, and this batch is one row — so a
658
+ return of 1 means the gap row did *not* land, which is the same loss as a raise and
659
+ was the one path this method used to ignore. Both give the claim back.
660
+
661
+ The failure stays suppressed either way: the batch this follows did land, and a
662
+ gap row that cannot be written must not turn a successful flush into a failed one.
663
+ """
664
+ with self._counter:
665
+ claimed = min(dropped, self._dropped)
666
+ self._dropped -= claimed
667
+ if not claimed:
668
+ return
669
+ try:
670
+ refused = self._deliver([Event(kind=EventKind.LOG_DROPPED.value, detail={'dropped': claimed})])
671
+ except Exception:
672
+ self._reclaim(claimed)
673
+ logger.exception('could not record the gap; keeping the count for the next flush')
674
+ return
675
+ if refused:
676
+ self._reclaim(claimed)
677
+ logger.error(
678
+ 'the database refused the gap row; keeping the count for the next flush',
679
+ extra={'tg_dropped': claimed},
680
+ )
681
+
682
+ def _reclaim(self, claimed: int) -> None:
683
+ """Put a claim back, so a gap nobody could record survives to be recorded."""
684
+ with self._counter:
685
+ self._dropped += claimed
686
+
687
+ @staticmethod
688
+ def _write(batch: list[Event]) -> int:
689
+ """Hand a batch to the ORM, importing it here so a disabled process never does.
690
+
691
+ Returns how many rows did not land, which only a partial refusal produces.
692
+ """
693
+ from django_aiogram.eventlog.writer import write_batch # noqa: PLC0415 - the point: no django.db above
694
+
695
+ return write_batch(batch)
696
+
697
+ def _publish(self, batch: list[Event]) -> None:
698
+ """Hand a batch to whoever connected to :data:`events_recorded`.
699
+
700
+ ``send_robust``, so one broken receiver neither loses the batch for the
701
+ others nor stops the writer, and it is logged here because a receiver that
702
+ fails silently is a metric that reads as zero traffic. Django logs it too, on
703
+ its own ``django.dispatch`` logger; the line here is on the logger a project
704
+ configures for this package, which is where it will actually be seen.
705
+
706
+ **Wrapped anyway, because ``send_robust`` does not contain everything.**
707
+ Django's own failure logging reads ``receiver.__qualname__`` unguarded, and a
708
+ callable *instance* — an ordinary shape for a metrics collector — has no such
709
+ attribute. So a receiver like that raising makes ``send_robust`` itself raise
710
+ ``AttributeError``, measured on Django 6.1, and without this ``try`` it would
711
+ land in :meth:`_flush`'s ``except`` and be counted as a failed *write*: the
712
+ other receivers lose the batch, a ``log.dropped`` row appears, and the log
713
+ blames the database for something a receiver did. Containing it here makes
714
+ this method's promise true whatever Django does with it, on any supported
715
+ version.
716
+
717
+ The upshot is a method that **cannot raise**, which is the property the rest
718
+ of the writer needs from it rather than a defensive habit.
719
+
720
+ One limit worth stating, because it is Django's and not ours: when
721
+ ``send_robust`` raises on that unnamed receiver it abandons **its own loop**, so
722
+ receivers connected after the offending one do not run for that batch at all.
723
+ Containing it here keeps the write and every earlier receiver whole; it cannot
724
+ reach past Django into a dispatch that already stopped. Calling receivers
725
+ ourselves would need ``Signal._live_receivers``, a private API, which is a worse
726
+ trade than one documented sentence. A collector written as a callable instance
727
+ can close the gap on its side by defining ``__qualname__``; one written as a
728
+ function or a bound method has it already, and is the shape every recipe uses.
729
+
730
+ A tuple rather than the list itself: receivers run one after another with
731
+ the same argument, so one of them sorting or clearing a list would decide
732
+ what the next one sees.
733
+ """
734
+ if not events_recorded.receivers:
735
+ return
736
+ # the reporting loop is inside the guard as well as the dispatch, because
737
+ # `getattr(..., None)` absorbs only `AttributeError` — a receiver whose
738
+ # `__getattr__` raises anything else makes naming it raise, and the whole
739
+ # point is that nothing about a receiver reaches `_flush`'s failure counter
740
+ try:
741
+ for receiver, outcome in events_recorded.send_robust(sender=self, events=tuple(batch)):
742
+ if isinstance(outcome, BaseException):
743
+ logger.error(
744
+ 'an events_recorded receiver raised',
745
+ exc_info=outcome,
746
+ extra={'tg_receiver': _receiver_name(receiver), 'tg_count': len(batch)},
747
+ )
748
+ except Exception:
749
+ # even this is suppressed: `logger.exception` is `logger.error` with
750
+ # `exc_info`, so a project whose handler or formatter raises would take
751
+ # the fallback out too — and the whole purpose here is that **nothing**
752
+ # about publishing reaches `_flush`'s failure counter, where it would be
753
+ # reported as a database refusing a batch it never saw
754
+ with contextlib.suppress(Exception):
755
+ logger.exception('publishing recorded events failed', extra={'tg_count': len(batch)})
756
+
757
+ def _deliver(self, batch: list[Event]) -> int:
758
+ """Write a batch if this process keeps the table, then publish it either way.
759
+
760
+ Returns how many rows the database refused individually — zero unless a partial
761
+ refusal happened, and zero in a process that does not write at all.
762
+
763
+
764
+ The order is the contract, and it is two claims rather than one. The write
765
+ is **attempted first**, so nothing a receiver does can change a row that was
766
+ written — which is why receivers get the real ``Event`` objects and not
767
+ copies. And the publish is in a ``finally``, so a write that *failed* still
768
+ reaches them: a database that is down or unmigrated is exactly when someone
769
+ is watching a dashboard, and the metrics have no reason to go with it. A
770
+ receiver seeing a batch is therefore not evidence that a row exists for it.
771
+
772
+ Publishing first was the original order, and it handed receivers the same
773
+ list and the same ``Event`` objects the ORM was about to read. A frozen
774
+ dataclass does not freeze the ``detail`` dict inside it, so a receiver
775
+ clearing the list or editing a ``detail`` changed what got persisted. This
776
+ way round makes that impossible instead of asking receivers to be careful.
777
+ """
778
+ refused = 0
779
+ try:
780
+ if self.enabled:
781
+ with self._counter:
782
+ self._touched_database.add(threading.get_ident())
783
+ refused = self._write(batch)
784
+ finally:
785
+ self._publish(batch)
786
+ return refused
787
+
788
+ @staticmethod
789
+ def _close_connections() -> None:
790
+ """Release the writer thread's own connection on the way out."""
791
+ try:
792
+ from django_aiogram.eventlog.writer import close_connections # noqa: PLC0415 - as above
793
+ except Exception:
794
+ logger.exception('could not import the event log to close its connection')
795
+ return
796
+ close_connections()
797
+
798
+ def drain_once(self, timeout: float = 0.0) -> int:
799
+ """Write whatever is buffered, on the calling thread. Returns events processed.
800
+
801
+ Events taken off the queue, not rows that landed: `_flush` swallows a failed write
802
+ and the refused count `_deliver` returns is dropped here, so a batch the database
803
+ rejected still counts. A caller reading this as a durability signal is reading the
804
+ wrong number — the gap rows and `log.dropped` are what say what was lost.
805
+
806
+ Goes through the same flush the writer uses, gap recording included, so
807
+ a test driving this exercises the path production takes.
808
+ """
809
+ buffer = self._queue
810
+ if buffer is None:
811
+ return 0
812
+ batch: list[Event] = []
813
+ wakes: list[Wake] = []
814
+ deadline = time.monotonic() + timeout
815
+ while True:
816
+ try:
817
+ item = buffer.get(timeout=max(0.0, deadline - time.monotonic())) if timeout else buffer.get_nowait()
818
+ except queue.Empty:
819
+ break
820
+ if isinstance(item, Wake):
821
+ wakes.append(item)
822
+ continue
823
+ batch.append(item)
824
+ try:
825
+ if batch:
826
+ self._flush(batch, failures=0)
827
+ finally:
828
+ # same reason as the synchronous `record()` path: this thread wrote, and it is
829
+ # not the thread whose exit closes connections
830
+ self._forget_touch()
831
+ _acknowledge(wakes)
832
+ return len(batch)
833
+
834
+ def flush(self, timeout: float = STOP_TIMEOUT) -> None:
835
+ """Wait until what has been recorded so far has reached the database.
836
+
837
+ Waits on an acknowledgement from the writer rather than on the queue
838
+ going empty: the queue empties when a batch is *taken*, which is before
839
+ it is written, so polling it would return mid-insert.
840
+ """
841
+ buffer = self._queue
842
+ if buffer is None:
843
+ return
844
+ done = threading.Event()
845
+ try:
846
+ buffer.put_nowait(Wake(done))
847
+ except queue.Full:
848
+ # no room even for the marker, so there is nothing to wait behind
849
+ return
850
+ if not done.wait(timeout):
851
+ logger.warning('the event writer did not flush in time', extra={'tg_timeout': timeout})
852
+
853
+ def stop(self, timeout: float = STOP_TIMEOUT) -> None:
854
+ """Flush and end the writer. Idempotent: atexit and start_tgbot both call it."""
855
+ with self._guard:
856
+ buffer, thread = self._queue, self._thread
857
+ self._queue = self._thread = None
858
+ if buffer is None:
859
+ return
860
+ with contextlib.suppress(Exception):
861
+ atexit.unregister(self.stop)
862
+ self._stopping.set()
863
+ with contextlib.suppress(queue.Full):
864
+ buffer.put_nowait(Wake())
865
+ if thread is not None and thread is not threading.current_thread():
866
+ # a thread that never started cannot be joined, and this runs from
867
+ # atexit where raising is noise nobody can act on
868
+ with contextlib.suppress(RuntimeError):
869
+ thread.join(timeout)
870
+ if thread.is_alive():
871
+ logger.warning('the event writer did not finish in time', extra={'tg_timeout': timeout})
872
+ elif thread is not None:
873
+ # `stop()` from a receiver, which runs on the writer's own thread: joining
874
+ # would be waiting for itself, and the old code reported that as a writer
875
+ # that missed its deadline. It has not missed anything — it unwinds through
876
+ # the loop below as soon as this returns
877
+ logger.debug('stop() was called on the writer thread; it will unwind on its own')
878
+ # a record() that read self._queue before the swap above puts into a queue
879
+ # this method has already detached, and nothing else will ever look at it.
880
+ # Draining after the join is what keeps those events; the few instructions
881
+ # between this drain and the producer's put stay a gap, because closing it
882
+ # would mean a lock on the one path that may never wait
883
+ try:
884
+ self._abandon(buffer)
885
+ finally:
886
+ # the same rule as the synchronous `record()` and `drain_once()`: this ran on
887
+ # whoever called `stop()`, and that thread is not the one whose exit closes
888
+ # connections. Left behind, its mark can be inherited by a later
889
+ # receiver-only writer through a reused ident.
890
+ #
891
+ # Unless `stop()` was called *from* the writer, where `_run` is still on the
892
+ # stack below and has that mark to consume on its way out. Reached by a
893
+ # receiver calling `stop()`, since receivers run on that thread —
894
+ # `test_a_receiver_that_stops_the_log_does_not_strand_the_writer` covers both
895
+ # halves and fails without either
896
+ if threading.current_thread() is not thread:
897
+ self._forget_touch()
898
+
899
+ def reset(self) -> None:
900
+ """Re-read the settings next time; used by override_settings.
901
+
902
+ It does not flush. Every ``override_settings(TELEGRAM_BOT=...)`` in a
903
+ consumer's own test suite fires this twice, and waiting for the writer
904
+ there would put a second on each one. A test that needs its rows calls
905
+ :meth:`flush`; queued events survive the reset either way.
906
+ """
907
+ self._enabled = None
908
+ self._kinds = None
909
+ self._worker = None
910
+
911
+
912
+ recorder = EventRecorder()
913
+
914
+
915
+ def _reset_on_setting_change(setting: str, **_kwargs: object) -> None:
916
+ """Drop the cached flags after writing whatever was recorded under the old ones."""
917
+ if setting == SETTINGS_NAME:
918
+ recorder.reset()
919
+
920
+
921
+ # dispatch_uid keeps autoreload from stacking duplicate receivers
922
+ setting_changed.connect(_reset_on_setting_change, dispatch_uid='django_aiogram.eventlog.recorder')