django-aiogram 4.0.0.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. django_aiogram/__init__.py +64 -0
  2. django_aiogram/_singleton.py +22 -0
  3. django_aiogram/admin.py +380 -0
  4. django_aiogram/api.py +38 -0
  5. django_aiogram/apps.py +56 -0
  6. django_aiogram/config/__init__.py +15 -0
  7. django_aiogram/config/checks.py +759 -0
  8. django_aiogram/config/defaults.py +102 -0
  9. django_aiogram/config/enums.py +108 -0
  10. django_aiogram/config/settings.py +266 -0
  11. django_aiogram/consumer/__init__.py +13 -0
  12. django_aiogram/consumer/delivery.py +625 -0
  13. django_aiogram/consumer/routers.py +19 -0
  14. django_aiogram/consumer/webhook.py +154 -0
  15. django_aiogram/context.py +34 -0
  16. django_aiogram/eventlog/__init__.py +16 -0
  17. django_aiogram/eventlog/dbrouter.py +55 -0
  18. django_aiogram/eventlog/events.py +120 -0
  19. django_aiogram/eventlog/instrumentation.py +231 -0
  20. django_aiogram/eventlog/recorder.py +922 -0
  21. django_aiogram/eventlog/signals.py +84 -0
  22. django_aiogram/eventlog/writer.py +231 -0
  23. django_aiogram/exceptions.py +60 -0
  24. django_aiogram/healthcheck.py +412 -0
  25. django_aiogram/management/__init__.py +1 -0
  26. django_aiogram/management/commands/__init__.py +1 -0
  27. django_aiogram/management/commands/start_tgbot.py +308 -0
  28. django_aiogram/management/commands/tgbot_healthcheck.py +57 -0
  29. django_aiogram/management/commands/tgbot_prune_events.py +144 -0
  30. django_aiogram/management/commands/tgbot_reclaim.py +135 -0
  31. django_aiogram/management/commands/tgbot_webhook.py +87 -0
  32. django_aiogram/migrations/0001_initial.py +50 -0
  33. django_aiogram/migrations/0002_kind_id_index.py +32 -0
  34. django_aiogram/migrations/__init__.py +1 -0
  35. django_aiogram/models.py +79 -0
  36. django_aiogram/producer/__init__.py +13 -0
  37. django_aiogram/producer/client.py +1540 -0
  38. django_aiogram/producer/throttling.py +336 -0
  39. django_aiogram/py.typed +0 -0
  40. django_aiogram/redis.py +394 -0
  41. django_aiogram/wire/__init__.py +14 -0
  42. django_aiogram/wire/envelope.py +146 -0
  43. django_aiogram/wire/payloads.py +195 -0
  44. django_aiogram/wire/serializers.py +533 -0
  45. django_aiogram-4.0.0.dev0.dist-info/METADATA +145 -0
  46. django_aiogram-4.0.0.dev0.dist-info/RECORD +48 -0
  47. django_aiogram-4.0.0.dev0.dist-info/WHEEL +4 -0
  48. django_aiogram-4.0.0.dev0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,308 @@
1
+ """Run the bot: receive updates, and consume the queue Django writes to.
2
+
3
+ This is the long-running process a bot container is built around. It owns two
4
+ things at once — whatever brings updates in, and the consumer that drains the
5
+ Redis queue — and has to shut both down cleanly when the container stops.
6
+ """
7
+
8
+ import contextlib
9
+ import logging
10
+ import signal
11
+ import threading
12
+ from argparse import ArgumentParser
13
+ from collections.abc import Callable
14
+ from types import FrameType
15
+ from typing import Any
16
+
17
+ from django.core.management import BaseCommand, CommandError
18
+
19
+ from django_aiogram import bot
20
+ from django_aiogram.config.enums import UpdateMode
21
+ from django_aiogram.config.settings import SETTINGS_NAME, coerce_bool, conf
22
+ from django_aiogram.consumer.delivery import Delivery, get_delivery
23
+ from django_aiogram.consumer.webhook import MODES, current_mode
24
+ from django_aiogram.eventlog.events import worker_identity
25
+ from django_aiogram.eventlog.recorder import recorder
26
+ from django_aiogram.redis import read_timeout
27
+
28
+ logger = logging.getLogger('django_aiogram')
29
+
30
+ #: what signal.signal returns: a handler, one of the SIG_* constants, or None
31
+ Handler = Callable[[int, FrameType | None], Any] | int | None
32
+
33
+
34
+ class Command(BaseCommand):
35
+ """Start the bot and the queue consumer, and stop them together."""
36
+
37
+ help = 'Start telegram bot'
38
+
39
+ #: what --idle waits on; tests replace it so they can end the wait
40
+ idle_event: threading.Event | None = None
41
+
42
+ def add_arguments(self, parser: ArgumentParser) -> None:
43
+ """Declare --mode and --idle."""
44
+ parser.add_argument(
45
+ '--mode',
46
+ choices=sorted(MODES),
47
+ default=None,
48
+ help=(
49
+ "how updates reach the bot for this run. Defaults to TELEGRAM_BOT['MODE'] "
50
+ "(env: DJANGO_AIOGRAM_MODE), itself 'polling'. In webhook mode this "
51
+ 'process consumes the queue and never calls getUpdates, because the updates '
52
+ 'arrive over HTTP instead.'
53
+ ),
54
+ )
55
+ parser.add_argument(
56
+ '--idle',
57
+ action='store_true',
58
+ help=(
59
+ 'When the bot is disabled, block instead of exiting. Useful under '
60
+ 'restart policies that treat a clean exit as a crash loop.'
61
+ ),
62
+ )
63
+
64
+ def handle(self, *args: Any, **options: Any) -> None:
65
+ """Receive updates, drain the queue, and unwind both on a signal."""
66
+ if not bot.enabled:
67
+ self.stdout.write(
68
+ self.style.WARNING(
69
+ 'django-aiogram is disabled '
70
+ "(TELEGRAM_BOT['ENABLED'] or DJANGO_AIOGRAM_ENABLED); "
71
+ 'not starting the bot.'
72
+ )
73
+ )
74
+ if options['idle']:
75
+ self._idle_until_signalled()
76
+ return
77
+
78
+ configured = current_mode()
79
+ mode = options['mode'] or configured
80
+ self.stdout.write(f'Updates arrive by {mode}.')
81
+ if mode != configured:
82
+ # the webhook view reads the setting, not this flag, so it would
83
+ # refuse the updates this process is no longer polling for
84
+ self.stdout.write(
85
+ self.style.WARNING(
86
+ f"--mode {mode} disagrees with TELEGRAM_BOT['MODE'] ({configured}), and it "
87
+ 'changes this process only: '
88
+ + (
89
+ 'the webhook view still refuses updates while the setting says polling'
90
+ if mode == UpdateMode.WEBHOOK
91
+ else 'getUpdates fails while a webhook is registered'
92
+ )
93
+ )
94
+ )
95
+
96
+ delivery = get_delivery(handler=bot.send_raw)
97
+ self._preflight(delivery)
98
+ # before a thread exists, because `read_timeout()` refuses a non-numeric
99
+ # `REDIS_TIMEOUT` — and raised from the `finally` below it would skip `close()`,
100
+ # `collect()` and `recorder.stop()`, stranding the drain's own messages
101
+ join_timeout = read_timeout() + 1
102
+ threads: list[threading.Thread] = []
103
+
104
+ # Both modes: starting the consumer before the loop runs would let a
105
+ # backlog reach send_raw while loop.is_running() is still False, so the
106
+ # coroutine would be driven from the consumer thread. Deferring the start
107
+ # until the loop picks up this callback keeps the loop single-threaded.
108
+ # Webhook mode used to start it directly because nothing ran the loop
109
+ # there — something does now, which is what this change is about.
110
+ #
111
+ # Started through a callback on the loop, so it cannot begin before the loop is
112
+ # turning, and refused once the shutdown starts. close() runs one turn of the loop
113
+ # on purpose, so a callback still queued when we reach the finally would
114
+ # start the consumer *after* stop() and after the joins — a thread nobody
115
+ # waits for, doing Redis work, whose first act is reclaim()
116
+ shutting_down = threading.Event()
117
+
118
+ def start_consuming() -> None:
119
+ """Start the consumer thread on the loop, unless the shutdown got there first.
120
+
121
+ Queued with ``call_soon`` so the thread begins once the loop is turning, and
122
+ gated because the callback can still be pending when the teardown runs — see
123
+ the comment above for what an ungated one starts, and how late.
124
+ """
125
+ if shutting_down.is_set():
126
+ logger.info('not starting the consumer: the shutdown had already begun')
127
+ return
128
+ threads.append(delivery.start_thread())
129
+
130
+ bot.loop.call_soon(start_consuming)
131
+ previous = self._install_sigterm_handler()
132
+
133
+ try:
134
+ with contextlib.suppress(KeyboardInterrupt, SystemExit):
135
+ if mode == UpdateMode.WEBHOOK:
136
+ self.stdout.write('Consuming the queue; updates are expected over HTTP.')
137
+ self._idle_on_the_loop()
138
+ else:
139
+ bot.start_polling()
140
+ finally:
141
+ logger.info('shutting down')
142
+ # before stop(), so the callback above cannot slip a consumer in
143
+ # behind the joins below
144
+ shutting_down.set()
145
+ delivery.stop()
146
+ for thread in threads:
147
+ # derived from the bound that actually governs the thread: every
148
+ # call it makes is capped by REDIS_TIMEOUT, and its blocking pop
149
+ # by one less than that. BLPOP_TIMEOUT + 1 was six seconds against
150
+ # a worst case of ten, so a consumer that outlived the join went on
151
+ # to acknowledge a message close() had already refused
152
+ thread.join(timeout=join_timeout)
153
+ if thread.is_alive():
154
+ logger.warning(
155
+ 'the delivery consumer did not stop in time',
156
+ extra={'tg_timeout': join_timeout},
157
+ )
158
+ try:
159
+ bot.close()
160
+ finally:
161
+ # the sends close() just drained reported themselves finished into a
162
+ # queue whose only reader is the consumer loop, and that returned before
163
+ # the join above — so without this every message the drain delivered
164
+ # stays in the in-flight list and the next start sends it again. A
165
+ # graceful stop duplicated whatever the drain had time to finish, which
166
+ # is the one thing `Delivery.md` says a *kill* is needed for
167
+ try:
168
+ delivery.collect()
169
+ finally:
170
+ # after close(), never before: closing drains in-flight sends,
171
+ # and those are what produce the final rows. In its own finally
172
+ # because a close() that raises must not also lose the rows
173
+ recorder.stop()
174
+ if previous is not None:
175
+ # the command may be called in-process; leaving our handler
176
+ # installed would turn a later SIGTERM into a stray interrupt
177
+ with contextlib.suppress(ValueError):
178
+ signal.signal(signal.SIGTERM, previous)
179
+
180
+ def _idle_on_the_loop(self) -> None:
181
+ """Wait on the bot's loop rather than on an Event.
182
+
183
+ In webhook mode this process consumes the queue and nothing drove the
184
+ loop, so every send the consumer scheduled sat there until something else
185
+ happened to run it — the next update, or `close()`. `run_forever` is what
186
+ makes a scheduled send run when it is scheduled, and it unwinds on
187
+ SIGTERM exactly as `start_polling` does, so the teardown below is
188
+ unchanged.
189
+ """
190
+ stop = self.idle_event or threading.Event()
191
+ loop = bot.loop
192
+
193
+ def wait_then_stop() -> None:
194
+ """Wait for the idle event on this thread, then stop the loop from it.
195
+
196
+ The main thread is inside ``run_forever`` and cannot wait for anything, so
197
+ the wait lives here and reaches the loop through ``call_soon_threadsafe`` —
198
+ the only safe way in from another thread. A loop already closed raises
199
+ ``RuntimeError``, which is a race with the teardown and not a fault.
200
+ """
201
+ stop.wait()
202
+ with contextlib.suppress(RuntimeError):
203
+ loop.call_soon_threadsafe(loop.stop)
204
+
205
+ threading.Thread(target=wait_then_stop, name='tgbot-idle', daemon=True).start()
206
+ loop.run_forever()
207
+
208
+ def _idle_until_signalled(self) -> None:
209
+ """Hold a disabled container open, and unwind it the way the enabled path does.
210
+
211
+ The same SIGTERM handler, so `docker stop` exits 0 rather than 143: without it the
212
+ signal kills the process outright and a container idling on purpose looked like one
213
+ that crashed. And `recorder.stop()`, because a disabled process with the log on
214
+ still has a writer thread holding a database connection.
215
+ """
216
+ self.stdout.write('Idling. Send SIGINT or SIGTERM to stop.')
217
+ previous = self._install_sigterm_handler()
218
+ try:
219
+ with contextlib.suppress(KeyboardInterrupt):
220
+ (self.idle_event or threading.Event()).wait()
221
+ finally:
222
+ recorder.stop()
223
+ if previous is not None:
224
+ with contextlib.suppress(ValueError):
225
+ signal.signal(signal.SIGTERM, previous)
226
+
227
+ def _preflight(self, delivery: Delivery) -> None:
228
+ """Everything worth saying or refusing before a thread exists."""
229
+ self._warn_about_an_unstable_worker_name()
230
+ self._require_crash_safety(delivery)
231
+
232
+ def _warn_about_an_unstable_worker_name(self) -> None:
233
+ """Say it here, where being the consumer is known.
234
+
235
+ The in-flight list is keyed on the worker's name, so a name that changes when the
236
+ container is replaced strands whatever the old one was sending. As a system check
237
+ this can only be information: `manage.py check` runs in every process, and a check
238
+ cannot tell a consumer from a web tier — as a warning it failed
239
+ `check --fail-level WARNING` in containers that own no in-flight list at all.
240
+
241
+ This process is the consumer. The same rule, reused rather than restated, so the
242
+ two cannot drift.
243
+ """
244
+ from django_aiogram.config.checks import worker_name_problems # noqa: PLC0415 - no aiogram at import
245
+
246
+ for problem in worker_name_problems():
247
+ logger.warning(
248
+ 'the worker name will not survive a replacement container',
249
+ extra={'tg_worker': worker_identity()},
250
+ )
251
+ self.stdout.write(self.style.WARNING(f'WORKER_NAME {problem.message}'))
252
+
253
+ @staticmethod
254
+ def _require_crash_safety(delivery: Delivery) -> None:
255
+ """Refuse to start where a killed worker loses the message it was sending.
256
+
257
+ Probed here rather than from inside ``run()``: that is a daemon thread, so
258
+ a ``SystemExit`` raised there kills only the thread and leaves a process
259
+ polling updates with a dead consumer. A ``CommandError`` gives a non-zero
260
+ exit and a restart loop somebody can see.
261
+
262
+ ``reclaim()`` is the probe, and it is the same call ``run()`` opens with.
263
+ An unreachable Redis returns False with crash safety still intact, which
264
+ must not be read as an old server — a blip is not a reason to refuse to
265
+ start.
266
+ """
267
+ if not coerce_bool(conf['REQUIRE_CRASH_SAFE'], f"{SETTINGS_NAME}['REQUIRE_CRASH_SAFE']"):
268
+ return
269
+ settled = delivery.reclaim()
270
+ if delivery.crash_safe:
271
+ if not settled:
272
+ # NOPERM and WRONGTYPE come back this way too, and unlike a blip
273
+ # they do not clear. Refusing here would turn every restart into
274
+ # a crash loop, so say plainly that the guarantee is unproven
275
+ # rather than let silence read as a passed check
276
+ logger.warning(
277
+ 'could not verify crash-safe delivery: the probe did not settle',
278
+ extra={'tg_key': delivery.queue_key},
279
+ )
280
+ return
281
+ msg = (
282
+ 'This Redis predates LMOVE (6.2), so a worker killed mid-send loses that message, '
283
+ f"and {SETTINGS_NAME}['REQUIRE_CRASH_SAFE'] refuses to run that way. Upgrade the "
284
+ 'server, or set it to False to accept at-most-once delivery.'
285
+ )
286
+ raise CommandError(msg)
287
+
288
+ @staticmethod
289
+ def _install_sigterm_handler() -> Handler:
290
+ """Turn SIGTERM into KeyboardInterrupt so `docker stop` unwinds cleanly.
291
+
292
+ Returns the handler it replaced, or None when it could not install one —
293
+ signal.signal only works on the main thread.
294
+ """
295
+
296
+ def raise_interrupt(_signum: int, _frame: FrameType | None) -> None:
297
+ """Raise where the signal arrived, which is inside whatever was blocking.
298
+
299
+ That is the whole trick: ``KeyboardInterrupt`` unwinds ``start_polling`` and
300
+ ``run_forever`` through the same path a Ctrl-C takes, so one teardown covers
301
+ both an operator and ``docker stop``.
302
+ """
303
+ raise KeyboardInterrupt
304
+
305
+ try:
306
+ return signal.signal(signal.SIGTERM, raise_interrupt)
307
+ except ValueError:
308
+ return None
@@ -0,0 +1,57 @@
1
+ """Answer whether the bot container is doing its job.
2
+
3
+ `docker ps` says the process is up, which is not the same thing: the consumer
4
+ thread can be dead while polling continues, or Redis can be unreachable, and the
5
+ container stays "healthy" either way.
6
+
7
+ A wrapper, since 3.1.0, over :mod:`django_aiogram.healthcheck`. The decision
8
+ lives there so that ``python -m django_aiogram.healthcheck`` can make it without
9
+ ``django.setup()`` — a management command populates the app registry and runs every
10
+ ``AppConfig.ready()`` in the host project first, which is 17.89s in one measured
11
+ consumer against 0.01s of actual probing, and more than any Docker ``timeout`` can
12
+ honestly allow. This command is unchanged for anyone who has it in a compose file
13
+ today, and **Deployment** says why the other form belongs in a healthcheck.
14
+ """
15
+
16
+ from argparse import ArgumentParser
17
+ from typing import Any
18
+
19
+ from django.core.management import BaseCommand, CommandError
20
+
21
+ from django_aiogram.healthcheck import add_limit_flags, check
22
+
23
+
24
+ class Command(BaseCommand):
25
+ """Check Redis, the consumer's heartbeat and the queue length, in that order."""
26
+
27
+ help = 'Exit 0 when the bot container is healthy, non-zero with a reason otherwise'
28
+
29
+ def add_arguments(self, parser: ArgumentParser) -> None:
30
+ """Declare the two limits, both of which default to a setting.
31
+
32
+ Taken from the module that acts on them rather than restated here: a second copy
33
+ of a flag is how one form ends up with a default the other does not have.
34
+ """
35
+ add_limit_flags(parser)
36
+
37
+ def handle(self, *args: Any, **options: Any) -> None:
38
+ """Report the first thing that is wrong, or that everything is fine.
39
+
40
+ ``stranded`` and ``guarantee`` are asked for explicitly, because this command's
41
+ output must not change for anyone who has it in a compose file — while the
42
+ container-facing entry point leaves both off, since neither can alter the
43
+ verdict and both are the expensive part of the probe.
44
+ """
45
+ report = check(
46
+ max_queue=options['max_queue'],
47
+ max_age=options['max_age'],
48
+ stranded=True,
49
+ guarantee=True,
50
+ )
51
+ if not report.ok:
52
+ raise CommandError(report.message)
53
+ # plain when nothing was examined: a disabled process is not a healthy bot, and
54
+ # this command has never colored that line green
55
+ self.stdout.write(self.style.SUCCESS(report.message) if report.checked else report.message)
56
+ for warning in report.warnings:
57
+ self.stdout.write(self.style.WARNING(warning))
@@ -0,0 +1,144 @@
1
+ """Delete event log rows older than the retention window.
2
+
3
+ Nothing on the write path deletes anything, so the table grows until this runs.
4
+ Schedule it; check ``W006`` says so while the retention is unset.
5
+
6
+ The deletion walks **primary key ranges**, not ``pk__in``. Two reasons, both
7
+ practical: MySQL rejects ``DELETE ... WHERE id IN (SELECT ... LIMIT n)`` against
8
+ the same table, which is exactly what the ORM generates for the obvious
9
+ formulation; and a bounded range at the cold end of the table cannot conflict
10
+ with the inserts still arriving at the hot end, which is what keeps InnoDB's
11
+ next-key locks out of the picture.
12
+ """
13
+
14
+ import datetime
15
+ import logging
16
+ import time
17
+ from argparse import ArgumentParser
18
+ from typing import Any, NamedTuple
19
+
20
+ from django.core.management import BaseCommand, CommandError
21
+ from django.db import connections, models, transaction
22
+ from django.utils import timezone
23
+
24
+ from django_aiogram.config.settings import conf
25
+ from django_aiogram.eventlog.writer import log_alias
26
+ from django_aiogram.models import TelegramEvent
27
+
28
+ logger = logging.getLogger('django_aiogram')
29
+
30
+
31
+ class Window(NamedTuple):
32
+ """The range one run walks: the rows, its ends, the cutoff and the alias."""
33
+
34
+ rows: models.QuerySet[TelegramEvent]
35
+ low: int
36
+ watermark: int
37
+ cutoff: datetime.datetime
38
+ alias: str
39
+
40
+
41
+ class Command(BaseCommand):
42
+ """Prune the event log in bounded chunks, one transaction each."""
43
+
44
+ help = 'Delete bot event rows older than the retention window'
45
+
46
+ def add_arguments(self, parser: ArgumentParser) -> None:
47
+ """Declare the window, the chunking and the two safety valves."""
48
+ parser.add_argument(
49
+ '--days',
50
+ type=int,
51
+ default=None,
52
+ help="delete rows older than this. Defaults to TELEGRAM_BOT['EVENT_LOG_RETENTION_DAYS'].",
53
+ )
54
+ parser.add_argument(
55
+ '--chunk',
56
+ type=int,
57
+ default=1000,
58
+ help='width of the id range each transaction covers. Rows inside it that are still '
59
+ 'within the window are left alone, so a chunk deletes at most this many (default 1000)',
60
+ )
61
+ parser.add_argument(
62
+ '--sleep',
63
+ type=float,
64
+ default=0.1,
65
+ help='seconds between chunks. The valve for replica lag: with row-based binlogs '
66
+ 'every deleted row is an event (default 0.1)',
67
+ )
68
+ parser.add_argument(
69
+ '--max-chunks',
70
+ type=int,
71
+ default=0,
72
+ help='stop after this many chunks, so a nightly run has a bounded blast radius. 0 means no limit',
73
+ )
74
+ parser.add_argument('--database', default=None, help='the alias to prune; defaults to the configured one')
75
+ parser.add_argument('--dry-run', action='store_true', help='report what would be deleted, and delete nothing')
76
+
77
+ def handle(self, *args: Any, **options: Any) -> None:
78
+ """Walk the table by primary key, deleting one bounded range per commit."""
79
+ days = options['days'] if options['days'] is not None else int(conf['EVENT_LOG_RETENTION_DAYS'])
80
+ if days <= 0:
81
+ self.stdout.write('Retention is not set, so nothing is pruned. See EVENT_LOG_RETENTION_DAYS.')
82
+ return
83
+
84
+ alias = options['database'] or log_alias()
85
+ if alias not in connections:
86
+ # E041 guards the setting; the flag bypasses it, in the one command that runs
87
+ # from cron — where a Django traceback is the least useful thing to wake up to
88
+ msg = f'no database is configured under the alias {alias!r}; DATABASES has {sorted(connections)}.'
89
+ raise CommandError(msg)
90
+ cutoff = timezone.now() - datetime.timedelta(days=days)
91
+ rows = TelegramEvent.objects.using(alias)
92
+
93
+ # where the walk stops: nothing older than the cutoff lives above this id.
94
+ # `drai_event_recent` covers the cutoff range, so neither form touches the
95
+ # table — but ordering by id still sorts that range, and `EXPLAIN QUERY PLAN`
96
+ # gives this and `Min(id)` the same two steps: the covering search and one
97
+ # `USE TEMP B-TREE FOR ORDER BY`. Written as a limit rather than an aggregate
98
+ # to read like the `low` below it, not because it measures faster
99
+ expired = rows.filter(created_at__lt=cutoff)
100
+ watermark = expired.order_by('-id').values_list('id', flat=True).first()
101
+ if watermark is None:
102
+ self.stdout.write(f'Nothing older than {cutoff.isoformat()}.')
103
+ return
104
+ # filtered by the cutoff like the watermark is. Taking the table's lowest
105
+ # id instead meant one surviving row down there pinned the walk to
106
+ # restart from it every night, and a --max-chunks run never got past it
107
+ low = expired.order_by('id').values_list('id', flat=True).first() or 1
108
+
109
+ deleted = self._walk(Window(rows, low, watermark, cutoff, alias), options)
110
+ verb = 'would delete' if options['dry_run'] else 'deleted'
111
+ self.stdout.write(f'{verb} {deleted} events older than {cutoff.isoformat()} from {alias!r}.')
112
+
113
+ def _walk(self, window: 'Window', options: dict[str, Any]) -> int:
114
+ """Delete each id range in turn, pausing between them."""
115
+ rows, low, watermark, cutoff, alias = window
116
+ chunk = max(1, int(options['chunk']))
117
+ limit = max(0, int(options['max_chunks']))
118
+ pause = max(0.0, float(options['sleep']))
119
+ deleted = 0
120
+ rounds = 0
121
+
122
+ while low <= watermark:
123
+ high = min(low + chunk - 1, watermark)
124
+ # the id range is the access path; created_at is the correctness
125
+ # condition, because id order only approximates time order once
126
+ # several processes and a buffered writer are involved
127
+ batch = rows.filter(id__gte=low, id__lte=high, created_at__lt=cutoff)
128
+ if options['dry_run']:
129
+ removed_here = batch.count()
130
+ else:
131
+ with transaction.atomic(using=alias):
132
+ removed_here, _ = batch.delete()
133
+ deleted += removed_here
134
+ low = high + 1
135
+ rounds += 1
136
+ if limit and rounds >= limit:
137
+ self.stdout.write(f'Stopped after {rounds} chunks; rerun to continue.')
138
+ break
139
+ # nothing was deleted, so there is nothing for a replica to catch up
140
+ # on and nothing to vacuum; a dry run deletes nothing at all
141
+ if pause and removed_here and low <= watermark and not options['dry_run']:
142
+ # replicas and autovacuum both need the gaps
143
+ time.sleep(pause)
144
+ return deleted
@@ -0,0 +1,135 @@
1
+ """Put a dead worker's in-flight messages back on the queue.
2
+
3
+ A message being sent lives in ``<queue>:processing:<worker>`` until the send
4
+ finishes, and the worker that put it there reclaims it on its next start. That
5
+ only works while the name is stable — a container with no ``hostname:`` and no
6
+ ``WORKER_NAME`` gets a fresh one every time, and its messages are stranded where
7
+ nothing will ever look for them again.
8
+
9
+ This is the way out, and it is deliberately manual: naming the dead worker is a
10
+ human saying it is dead. Nothing here probes for liveness, because a worker that
11
+ is merely slow looks exactly like one that is gone, and taking a message back
12
+ from a live sender is how you send it twice.
13
+ """
14
+
15
+ import logging
16
+ from argparse import ArgumentParser
17
+ from typing import Any
18
+
19
+ from django.core.management import BaseCommand, CommandError
20
+ from redis import Redis
21
+ from redis.exceptions import ResponseError
22
+
23
+ from django_aiogram.eventlog.events import worker_identity
24
+ from django_aiogram.redis import get_redis, processing_key, queue_key
25
+
26
+ logger = logging.getLogger('django_aiogram')
27
+
28
+
29
+ class Command(BaseCommand):
30
+ """Move one worker's in-flight messages back to the queue."""
31
+
32
+ help = 'Requeue the messages a dead worker left in flight'
33
+
34
+ def add_arguments(self, parser: ArgumentParser) -> None:
35
+ """Declare the worker to reclaim from, and the safety valves."""
36
+ parser.add_argument(
37
+ '--worker',
38
+ required=True,
39
+ help='the WORKER_NAME (or hostname) whose in-flight list to drain. Naming it is you '
40
+ 'saying that worker is gone: reclaiming from a live one sends its message twice.',
41
+ )
42
+ parser.add_argument(
43
+ '--limit',
44
+ type=int,
45
+ default=0,
46
+ help='stop after this many messages, so one run has a bounded blast radius. 0 means no '
47
+ 'limit. A bounded run takes the newest in flight first, because that is the end of the '
48
+ 'list a reclaim pops from',
49
+ )
50
+ parser.add_argument('--dry-run', action='store_true', help='report what is there, and move nothing')
51
+
52
+ @staticmethod
53
+ def _move_one(connection: Redis, source: str, destination: str) -> object:
54
+ """Move the newest in-flight message back to the front of the queue.
55
+
56
+ Newest, because a message is taken with ``LEFT`` → ``RIGHT`` and so the tail of the
57
+ in-flight list is the most recent one — and this pops that tail. Draining the whole
58
+ list therefore restores the original order, the oldest ending up at the front; a run
59
+ stopped by ``--limit`` has reclaimed the newest and left the older ones in place.
60
+
61
+ ``LMOVE ... RIGHT LEFT`` is what ``reclaim()`` uses, so the order a real run
62
+ produces is the order a dry run promises. On a Redis older than 6.2 that command
63
+ does not exist, and this is the one path that can still recover those messages:
64
+ the consumer's own ``reclaim()`` gives up there and runs at-most-once, so a list
65
+ stranded before the downgrade would have nothing else to come back through.
66
+ ``RPOPLPUSH`` is the same move, and has been there since 1.2.
67
+ """
68
+ try:
69
+ return connection.lmove(source, destination, 'RIGHT', 'LEFT')
70
+ except ResponseError as error:
71
+ if 'unknown command' not in str(error).lower():
72
+ raise
73
+ return connection.rpoplpush(source, destination)
74
+
75
+ def handle(self, *args: Any, **options: Any) -> None:
76
+ """Walk the named worker's in-flight list back onto the queue."""
77
+ worker = str(options['worker']).strip()
78
+ if not worker:
79
+ msg = '--worker cannot be empty.'
80
+ raise CommandError(msg)
81
+ limit = int(options['limit'])
82
+ if limit < 0:
83
+ # max(0, ...) would have read this as "no limit", which is the
84
+ # opposite of what someone typing a limit is asking for. Judged with
85
+ # the other arguments, so --dry-run reports the mistake rather than
86
+ # returning happily on a run that would have been refused
87
+ msg = f'--limit cannot be negative, got {limit}. Use 0 for no limit.'
88
+ raise CommandError(msg)
89
+ if worker == worker_identity():
90
+ # this process would be reclaiming from whatever is running here now,
91
+ # which on a bot container is the consumer that is mid-send
92
+ msg = (
93
+ f"{worker!r} is this process's own worker name. A running consumer reclaims its own "
94
+ 'messages when it starts; taking them from underneath it sends them twice.'
95
+ )
96
+ raise CommandError(msg)
97
+
98
+ source, destination = processing_key(worker), queue_key()
99
+ connection = get_redis()
100
+ try:
101
+ waiting = int(connection.llen(source) or 0)
102
+ except Exception as error:
103
+ msg = f'could not read {source}: {error}'
104
+ raise CommandError(msg) from error
105
+
106
+ if not waiting:
107
+ self.stdout.write(f'Nothing in flight for {worker!r}.')
108
+ return
109
+ if options['dry_run']:
110
+ # through the same limit the real run applies. A rehearsal that
111
+ # promises to move two and then moves one is worse than none: it is
112
+ # read as the plan, and the difference shows up as messages left
113
+ # behind that nobody went looking for
114
+ would_move = min(waiting, limit) if limit else waiting
115
+ self.stdout.write(
116
+ f'{waiting} message(s) in flight for {worker!r}; would requeue {would_move} of them.'
117
+ if would_move != waiting
118
+ else f'{waiting} message(s) in flight for {worker!r}; would requeue them.'
119
+ )
120
+ return
121
+
122
+ moved = 0
123
+ while not limit or moved < limit:
124
+ try:
125
+ # `is None`, not falsy: nil is how both commands say the source list is
126
+ # empty, and a payload that happens to be empty is still a message
127
+ if self._move_one(connection, source, destination) is None:
128
+ break
129
+ except Exception as error:
130
+ msg = f'moved {moved} message(s), then failed: {error}'
131
+ raise CommandError(msg) from error
132
+ moved += 1
133
+
134
+ logger.info('reclaimed a dead worker', extra={'tg_key': source, 'tg_count': moved})
135
+ self.stdout.write(self.style.SUCCESS(f'Requeued {moved} message(s) from {worker!r}.'))