django-aiogram 4.0.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. django_aiogram/__init__.py +64 -0
  2. django_aiogram/_singleton.py +22 -0
  3. django_aiogram/admin.py +380 -0
  4. django_aiogram/api.py +38 -0
  5. django_aiogram/apps.py +56 -0
  6. django_aiogram/config/__init__.py +15 -0
  7. django_aiogram/config/checks.py +759 -0
  8. django_aiogram/config/defaults.py +102 -0
  9. django_aiogram/config/enums.py +108 -0
  10. django_aiogram/config/settings.py +266 -0
  11. django_aiogram/consumer/__init__.py +13 -0
  12. django_aiogram/consumer/delivery.py +625 -0
  13. django_aiogram/consumer/routers.py +19 -0
  14. django_aiogram/consumer/webhook.py +154 -0
  15. django_aiogram/context.py +34 -0
  16. django_aiogram/eventlog/__init__.py +16 -0
  17. django_aiogram/eventlog/dbrouter.py +55 -0
  18. django_aiogram/eventlog/events.py +120 -0
  19. django_aiogram/eventlog/instrumentation.py +231 -0
  20. django_aiogram/eventlog/recorder.py +922 -0
  21. django_aiogram/eventlog/signals.py +84 -0
  22. django_aiogram/eventlog/writer.py +231 -0
  23. django_aiogram/exceptions.py +60 -0
  24. django_aiogram/healthcheck.py +412 -0
  25. django_aiogram/management/__init__.py +1 -0
  26. django_aiogram/management/commands/__init__.py +1 -0
  27. django_aiogram/management/commands/start_tgbot.py +308 -0
  28. django_aiogram/management/commands/tgbot_healthcheck.py +57 -0
  29. django_aiogram/management/commands/tgbot_prune_events.py +144 -0
  30. django_aiogram/management/commands/tgbot_reclaim.py +135 -0
  31. django_aiogram/management/commands/tgbot_webhook.py +87 -0
  32. django_aiogram/migrations/0001_initial.py +50 -0
  33. django_aiogram/migrations/0002_kind_id_index.py +32 -0
  34. django_aiogram/migrations/__init__.py +1 -0
  35. django_aiogram/models.py +79 -0
  36. django_aiogram/producer/__init__.py +13 -0
  37. django_aiogram/producer/client.py +1540 -0
  38. django_aiogram/producer/throttling.py +336 -0
  39. django_aiogram/py.typed +0 -0
  40. django_aiogram/redis.py +394 -0
  41. django_aiogram/wire/__init__.py +14 -0
  42. django_aiogram/wire/envelope.py +146 -0
  43. django_aiogram/wire/payloads.py +195 -0
  44. django_aiogram/wire/serializers.py +533 -0
  45. django_aiogram-4.0.0.dev0.dist-info/METADATA +145 -0
  46. django_aiogram-4.0.0.dev0.dist-info/RECORD +48 -0
  47. django_aiogram-4.0.0.dev0.dist-info/WHEEL +4 -0
  48. django_aiogram-4.0.0.dev0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,412 @@
1
+ """Answer whether the bot container is doing its job, without booting Django.
2
+
3
+ ``manage.py tgbot_healthcheck`` has always been correct and could not be used: a
4
+ management command runs ``django.setup()`` first, which populates the app registry
5
+ and executes every ``AppConfig.ready()`` in the *host* project. Measured in one
6
+ consumer — Django 5.2, twenty apps, one registering adapters in ``ready()`` — that
7
+ was 17.89s on top of 2.45s for the settings module, against ~0.01s for the probe's
8
+ own three Redis calls. Docker killed it at every timeout, and the container read
9
+ ``unhealthy`` for the best part of an hour while the bot was fine and its heartbeat
10
+ six seconds old.
11
+
12
+ So this module is the check, and both entry points are thin:
13
+
14
+ * ``python -m django_aiogram.healthcheck`` — for a container healthcheck. It
15
+ reads ``DJANGO_SETTINGS_MODULE`` the way any Django code does and never calls
16
+ ``django.setup()``.
17
+ * ``manage.py tgbot_healthcheck`` — unchanged, because consumers have it in compose
18
+ files today. It pays for ``django.setup()`` like every other command, and the
19
+ Deployment page says why you would not put it in a healthcheck.
20
+
21
+ **This module must not import anything that needs the app registry.** No models, no
22
+ aiogram, no :mod:`django_aiogram.producer.client`. Reading
23
+ ``django.conf.settings.TELEGRAM_BOT`` imports the settings module and nothing more,
24
+ which is the whole saving. ``tests/test_lazy_init.py`` asserts the registry is still
25
+ unpopulated after ``main()`` returns.
26
+ """
27
+
28
+ import argparse
29
+ import logging
30
+ import os
31
+ import sys
32
+ import time
33
+ from dataclasses import dataclass
34
+
35
+ from django.core.exceptions import ImproperlyConfigured
36
+ from redis import Redis
37
+ from redis.exceptions import RedisError, ResponseError
38
+
39
+ from django_aiogram.config.settings import SETTINGS_NAME, coerce_bool, conf
40
+ from django_aiogram.redis import (
41
+ get_redis,
42
+ heartbeat_key,
43
+ heartbeat_ttl,
44
+ processing_key,
45
+ processing_pattern,
46
+ queue_key,
47
+ )
48
+
49
+ logger = logging.getLogger('django_aiogram')
50
+
51
+ # round trips, not keys: MATCH filters on the server but SCAN walks the whole
52
+ # keyspace either way, and this probe runs on a timer
53
+ STRANDED_SCAN_ROUNDS = 20
54
+
55
+
56
+ @dataclass(frozen=True)
57
+ class Report:
58
+ """What the probe concluded: the verdict, the line to print, anything else to say.
59
+
60
+ ``warnings`` is separate from ``message`` because a warning must never change the
61
+ verdict — a stranded in-flight list may be one another worker is sending this
62
+ second, and an exit code that said otherwise would restart a healthy container.
63
+ """
64
+
65
+ ok: bool
66
+ message: str
67
+ warnings: tuple[str, ...] = ()
68
+ #: whether anything was actually examined. False only when this process is
69
+ #: disabled, which is not a verdict about the bot — and is why the management
70
+ #: command reports that one plainly rather than in success green, as it always has
71
+ checked: bool = True
72
+
73
+
74
+ class _UnhealthyError(Exception):
75
+ """One reason the probe is about to answer no.
76
+
77
+ Private, and raised only between the helpers below and :func:`check`, which turns
78
+ it back into a :class:`Report`. It exists because the check is a sequence of
79
+ reads where any one of them ends the answer — written as early returns, that was
80
+ twelve branches in one function and the reason for each was harder to see than the
81
+ control flow around it.
82
+ """
83
+
84
+
85
+ def _setting_int(key: str) -> int:
86
+ """Read one integer setting, refusing in a line rather than a traceback.
87
+
88
+ `E023` and `E024` say this in ``manage.py check`` — which runs under
89
+ ``django.setup()``, the thing this entry point exists to skip. So this is the one path
90
+ where a value like ``os.environ.get('HB', '')`` written straight into the settings
91
+ dict reaches ``int()`` with nothing between, and the probe is what has to say so.
92
+ """
93
+ raw = conf[key]
94
+ try:
95
+ return int(raw)
96
+ except (TypeError, ValueError) as error:
97
+ msg = f"{SETTINGS_NAME}['{key}'] is not a number: {raw!r}"
98
+ raise _UnhealthyError(msg) from error
99
+
100
+
101
+ def _connected() -> Redis:
102
+ """Return the shared connection, having proved it answers.
103
+
104
+ Two non-Redis failures are caught beside ``RedisError``, because from a probe's
105
+ point of view a connection it cannot build is a Redis it cannot reach — which is
106
+ what this command has always said about it, back when it caught ``Exception``:
107
+
108
+ * ``ImproperlyConfigured`` — an empty ``REDIS_URL``, which
109
+ :func:`~django_aiogram.redis.build_client` raises on.
110
+ * ``ValueError`` — a ``REDIS_URL`` with no scheme, which ``Redis.from_url`` rejects,
111
+ and a non-numeric ``REDIS_TIMEOUT``, which ``read_timeout()`` does. Both are
112
+ misconfiguration reaching us as a builtin, and both used to read as one line.
113
+
114
+ Narrowing to ``RedisError`` alone turned those into tracebacks and, in the management
115
+ command, into a bare exception where a ``CommandError`` belongs.
116
+ """
117
+ try:
118
+ connection = get_redis()
119
+ connection.ping()
120
+ except (RedisError, ImproperlyConfigured, ValueError) as error:
121
+ msg = f'redis is unreachable: {error}'
122
+ raise _UnhealthyError(msg) from error
123
+ return connection
124
+
125
+
126
+ def _heartbeat_age(connection: Redis, *, limit: int, ttl: int) -> int:
127
+ """How long ago the consumer last said it was turning.
128
+
129
+ Both numbers are needed: ``limit`` is what the verdict is judged by, and ``ttl`` is
130
+ what the key can actually show. A limit above the TTL is unreachable — the key is
131
+ gone by then — so the refusal names that rather than leaving an operator to wonder
132
+ why ``--max-age 600`` still refused at a minute.
133
+ """
134
+ try:
135
+ raw = connection.get(heartbeat_key())
136
+ except (RedisError, UnicodeDecodeError) as error:
137
+ # ping answering says nothing about the next command: a failover in between, or a
138
+ # key this replica cannot serve. The decode error is not hypothetical either —
139
+ # `decode_responses` in a URL shared with a cache backend makes redis-py decode
140
+ # this key, and the bytes there are not ours to promise anything about
141
+ msg = f'could not read the heartbeat: {error}'
142
+ raise _UnhealthyError(msg) from error
143
+ if raw is None:
144
+ # the applied limit, not the TTL: with `--max-age 600` the old wording said
145
+ # "within 30s", which contradicts the number the probe is judging by
146
+ msg = (
147
+ f'no heartbeat at {heartbeat_key()}: the consumer has not written one within {limit}s, or it never started'
148
+ )
149
+ if limit > ttl:
150
+ msg = (
151
+ f'{msg}. A limit over {ttl}s cannot be observed anyway, because the key expires then: '
152
+ f"raise {SETTINGS_NAME}['HEARTBEAT_INTERVAL'] instead"
153
+ )
154
+ raise _UnhealthyError(msg)
155
+ try:
156
+ age = int(time.time()) - int(raw)
157
+ except (TypeError, ValueError) as error:
158
+ msg = f'the heartbeat at {heartbeat_key()} is not a timestamp'
159
+ raise _UnhealthyError(msg) from error
160
+ if age > limit:
161
+ msg = f'the consumer last reported {age}s ago, over the {limit}s limit'
162
+ raise _UnhealthyError(msg)
163
+ return age
164
+
165
+
166
+ def _queue_depth(connection: Redis, *, limit: int) -> int:
167
+ """How many messages are waiting, refusing when that is over the limit."""
168
+ try:
169
+ queued = int(connection.llen(queue_key()) or 0)
170
+ except RedisError as error:
171
+ msg = f'could not read the queue length: {error}'
172
+ raise _UnhealthyError(msg) from error
173
+ if limit and queued > limit:
174
+ msg = f'{queued} messages are queued, over the limit of {limit}'
175
+ raise _UnhealthyError(msg)
176
+ return queued
177
+
178
+
179
+ def check(
180
+ *,
181
+ max_queue: int | None = None,
182
+ max_age: int | None = None,
183
+ stranded: bool = False,
184
+ guarantee: bool = False,
185
+ ) -> Report:
186
+ """Read Redis, the consumer's heartbeat and the queue length, in that order.
187
+
188
+ ``max_age`` defaults to the heartbeat key's own TTL — three ``HEARTBEAT_INTERVAL``s,
189
+ so one missed refresh is not a failure — which is also the most any reader can
190
+ observe; ``max_queue`` to ``HEALTHCHECK_MAX_QUEUE``, where 0 disables the limit.
191
+
192
+ ``stranded`` and ``guarantee`` are off by default and the management command turns
193
+ them on, which is the one place the two entry points differ. Both cost more than
194
+ everything else here and neither can change the verdict: the scan is up to twenty
195
+ ``SCAN`` rounds over a keyspace often shared with a cache backend, and the
196
+ guarantee probe is a write — a no-op ``LMOVE`` on a missing key, but still a write,
197
+ and on a read-only replica it answers ``unknown`` rather than the truth. Twice a
198
+ minute for a line nobody acts on is the wrong trade for a container healthcheck and
199
+ the right one for a command a person ran deliberately.
200
+ """
201
+ if not coerce_bool(conf['ENABLED'], f"{SETTINGS_NAME}['ENABLED']"):
202
+ # nothing is meant to be running here, so nothing is wrong
203
+ return Report(ok=True, message='disabled in this process; nothing to check', checked=False)
204
+
205
+ try:
206
+ # inside the guard: these two are read before Redis is touched, so an
207
+ # unreadable one is the first thing the probe meets rather than the last
208
+ ttl = heartbeat_ttl(max(1, _setting_int('HEARTBEAT_INTERVAL')))
209
+ age_limit = ttl if max_age is None else max_age
210
+ queue_limit = _setting_int('HEALTHCHECK_MAX_QUEUE') if max_queue is None else max_queue
211
+ connection = _connected()
212
+ age = _heartbeat_age(connection, limit=age_limit, ttl=ttl)
213
+ queued = _queue_depth(connection, limit=queue_limit)
214
+ except _UnhealthyError as refusal:
215
+ return Report(ok=False, message=str(refusal))
216
+
217
+ healthy = f'healthy: heartbeat {age}s old, {queued} queued'
218
+ if guarantee:
219
+ healthy = f'{healthy}, {_guarantee(connection)}'
220
+ warnings: list[str] = []
221
+ if stranded:
222
+ found, swept = _stranded(connection)
223
+ if found:
224
+ # not a failure: another worker may be sending them right now. But an
225
+ # invisible pile is how a stranded list stays stranded
226
+ warnings.append(
227
+ f'{found if swept else f"at least {found}"} message(s) are in flight under '
228
+ 'other worker names. If one of those workers is gone, '
229
+ '`manage.py tgbot_reclaim --worker <name>` requeues them.'
230
+ )
231
+ return Report(ok=True, message=healthy, warnings=tuple(warnings))
232
+
233
+
234
+ def _guarantee(connection: Redis) -> str:
235
+ """Which delivery guarantee this Redis can actually give.
236
+
237
+ Asked of the server, not of a consumer: ``Delivery.crash_safe`` starts true and is
238
+ only lowered by ``reclaim()``, which this probe must never call — requeueing a
239
+ running worker's in-flight list would send those messages twice.
240
+
241
+ It asks the same question ``reclaim()`` does, on a key that does not exist:
242
+ rotating an empty list is a no-op on a server that has ``LMOVE``, and
243
+ ``unknown command`` on one that does not.
244
+
245
+ ``COMMAND INFO lmove`` would be a read rather than a write, and does not work:
246
+ against Redis 6.0 — the servers where the answer actually differs — redis-py's own
247
+ response parser raises ``TypeError`` on the nil entry the server returns for an
248
+ unknown command. Catching a library's parser failing in a particular way is a
249
+ worse dependency than a no-op write, so the write stays and the *caller* decides
250
+ whether to pay for it.
251
+ """
252
+ probe = f'{queue_key()}:lmove-probe'
253
+ try:
254
+ connection.lmove(probe, probe, 'LEFT', 'RIGHT')
255
+ except ResponseError as error:
256
+ if 'unknown command' in str(error).lower():
257
+ return 'at-most-once'
258
+ logger.warning('could not establish which delivery guarantee is in force')
259
+ return 'unknown'
260
+ except RedisError:
261
+ logger.warning('could not establish which delivery guarantee is in force')
262
+ return 'unknown'
263
+ return 'at-least-once'
264
+
265
+
266
+ def _stranded(connection: Redis) -> tuple[int, bool]:
267
+ """Count what is in flight under a worker name that is not this one.
268
+
269
+ Read rather than acted on: a message under another name may be one another worker
270
+ is sending this second, and taking it back would send it twice.
271
+
272
+ Bounded, and returns whether it finished. ``MATCH`` filters on the server but
273
+ ``SCAN`` still walks the whole keyspace, and on a Redis shared with a cache
274
+ backend — which the settings page suggests is common — an unbounded sweep is a full
275
+ pass over someone else's keys. A partial answer is worth having; one that pretends
276
+ to be complete is not.
277
+ """
278
+ pattern = processing_pattern()
279
+ mine = processing_key()
280
+ # SCAN may return the same key more than once when the keyspace changes
281
+ # size mid-iteration, and counting one twice would invent a backlog
282
+ seen: set[str] = set()
283
+ total = 0
284
+ cursor = 0
285
+ try:
286
+ for _ in range(STRANDED_SCAN_ROUNDS):
287
+ cursor, keys = connection.scan(cursor=cursor, match=pattern, count=100)
288
+ for key in keys:
289
+ name = key.decode('utf-8') if isinstance(key, bytes) else str(key)
290
+ if name == mine or name in seen:
291
+ continue
292
+ seen.add(name)
293
+ total += int(connection.llen(name) or 0)
294
+ if cursor == 0:
295
+ return total, True
296
+ except (RedisError, UnicodeDecodeError):
297
+ # the decode too: a foreign key on a shared Redis can match this pattern and hold
298
+ # bytes that are not UTF-8, and aborting the whole probe over a warning nobody
299
+ # acts on is the opposite of what this sweep is for
300
+ logger.warning('could not scan for stranded in-flight lists', extra={'tg_key': pattern})
301
+ return total, False
302
+ return total, False
303
+
304
+
305
+ def add_limit_flags(parser: argparse.ArgumentParser) -> None:
306
+ """Declare the two limits on any parser, so the entry points cannot drift.
307
+
308
+ The management command adds them to the parser Django hands it, which is why this
309
+ takes one rather than making its own: the flags, their types and their defaults are
310
+ a property of :func:`check`, and a reader comparing ``--help`` of the two forms is
311
+ entitled to the same answer from both.
312
+ """
313
+ parser.add_argument(
314
+ '--max-queue',
315
+ type=int,
316
+ default=None,
317
+ help=f"messages allowed to be waiting; defaults to {SETTINGS_NAME}['HEALTHCHECK_MAX_QUEUE'], 0 disables",
318
+ )
319
+ parser.add_argument(
320
+ '--max-age',
321
+ type=int,
322
+ default=None,
323
+ help=(
324
+ f"seconds the heartbeat may be stale; defaults to three {SETTINGS_NAME}['HEARTBEAT_INTERVAL']s, "
325
+ "which is also the key's TTL and so the most that can be observed"
326
+ ),
327
+ )
328
+
329
+
330
+ def build_parser() -> argparse.ArgumentParser:
331
+ """Declare the container form's flags: the shared limits, plus what it leaves off."""
332
+ parser = argparse.ArgumentParser(
333
+ prog='python -m django_aiogram.healthcheck',
334
+ description='Exit 0 when the bot container is healthy, non-zero with a reason otherwise',
335
+ )
336
+ add_limit_flags(parser)
337
+ parser.add_argument(
338
+ '--stranded',
339
+ action='store_true',
340
+ help=(f'also scan for in-flight lists left by other worker names (up to {STRANDED_SCAN_ROUNDS} SCAN rounds)'),
341
+ )
342
+ parser.add_argument(
343
+ '--guarantee',
344
+ action='store_true',
345
+ help='also report which delivery guarantee this Redis gives (issues a no-op write)',
346
+ )
347
+ return parser
348
+
349
+
350
+ def _names_the_settings_module(error: ImportError) -> bool:
351
+ """Whether what failed to import is the configured settings module itself.
352
+
353
+ ``ModuleNotFoundError`` carries the dotted name it could not find, and a missing
354
+ parent package reports the parent — so ``core.settings`` with no ``core`` on the path
355
+ is recognized too.
356
+ """
357
+ missing = getattr(error, 'name', None)
358
+ configured = os.environ.get('DJANGO_SETTINGS_MODULE')
359
+ if not missing or not configured:
360
+ return False
361
+ return configured == missing or configured.startswith(f'{missing}.')
362
+
363
+
364
+ def _cannot_read(error: Exception) -> int:
365
+ """Write the one-line refusal and give the exit code that goes with it."""
366
+ sys.stderr.write(f'cannot read the settings: {error}\n')
367
+ return 1
368
+
369
+
370
+ def main(argv: list[str] | None = None) -> int:
371
+ """Run the check and print its report. Returns the exit code.
372
+
373
+ Nothing here calls ``django.setup()``, which is the point of the module. Reading a
374
+ setting still needs ``DJANGO_SETTINGS_MODULE`` in the environment, which a container
375
+ running ``manage.py`` does not necessarily have — see the refusal below.
376
+ """
377
+ options = build_parser().parse_args(argv)
378
+ try:
379
+ report = check(
380
+ max_queue=options.max_queue,
381
+ max_age=options.max_age,
382
+ stranded=options.stranded,
383
+ guarantee=options.guarantee,
384
+ )
385
+ except ImproperlyConfigured as error:
386
+ # the failure this form meets that the management command cannot: `manage.py` sets
387
+ # DJANGO_SETTINGS_MODULE inside its own process, so a container running it does not
388
+ # necessarily export the variable — and a healthcheck is a separate process. A
389
+ # traceback here would say "unhealthy" without saying why, from a probe whose whole
390
+ # job is to say why
391
+ return _cannot_read(error)
392
+ except ImportError as error:
393
+ # the recipe asks an operator to write that module name by hand, so a mistyped one
394
+ # is at least as likely as a missing variable, and it deserves the same line. Only
395
+ # that one: an import the settings module itself fails at is not ours to flatten,
396
+ # and its traceback is where the answer is
397
+ if not _names_the_settings_module(error):
398
+ raise
399
+ return _cannot_read(error)
400
+ stream = sys.stdout if report.ok else sys.stderr
401
+ stream.write(f'{report.message}\n')
402
+ # stdout is block-buffered when it is not a tty and stderr is not buffered at all, so
403
+ # without this the warnings reach a merged `docker inspect` log before the verdict
404
+ # they qualify
405
+ stream.flush()
406
+ for warning in report.warnings:
407
+ sys.stderr.write(f'{warning}\n')
408
+ return 0 if report.ok else 1
409
+
410
+
411
+ if __name__ == '__main__':
412
+ raise SystemExit(main())
@@ -0,0 +1 @@
1
+ """Where Django looks for the ``manage.py`` commands this package adds."""
@@ -0,0 +1 @@
1
+ """One module per command: ``start_tgbot``, ``tgbot_webhook``, ``tgbot_healthcheck``."""