django-aiogram 4.0.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- django_aiogram/__init__.py +64 -0
- django_aiogram/_singleton.py +22 -0
- django_aiogram/admin.py +380 -0
- django_aiogram/api.py +38 -0
- django_aiogram/apps.py +56 -0
- django_aiogram/config/__init__.py +15 -0
- django_aiogram/config/checks.py +759 -0
- django_aiogram/config/defaults.py +102 -0
- django_aiogram/config/enums.py +108 -0
- django_aiogram/config/settings.py +266 -0
- django_aiogram/consumer/__init__.py +13 -0
- django_aiogram/consumer/delivery.py +625 -0
- django_aiogram/consumer/routers.py +19 -0
- django_aiogram/consumer/webhook.py +154 -0
- django_aiogram/context.py +34 -0
- django_aiogram/eventlog/__init__.py +16 -0
- django_aiogram/eventlog/dbrouter.py +55 -0
- django_aiogram/eventlog/events.py +120 -0
- django_aiogram/eventlog/instrumentation.py +231 -0
- django_aiogram/eventlog/recorder.py +922 -0
- django_aiogram/eventlog/signals.py +84 -0
- django_aiogram/eventlog/writer.py +231 -0
- django_aiogram/exceptions.py +60 -0
- django_aiogram/healthcheck.py +412 -0
- django_aiogram/management/__init__.py +1 -0
- django_aiogram/management/commands/__init__.py +1 -0
- django_aiogram/management/commands/start_tgbot.py +308 -0
- django_aiogram/management/commands/tgbot_healthcheck.py +57 -0
- django_aiogram/management/commands/tgbot_prune_events.py +144 -0
- django_aiogram/management/commands/tgbot_reclaim.py +135 -0
- django_aiogram/management/commands/tgbot_webhook.py +87 -0
- django_aiogram/migrations/0001_initial.py +50 -0
- django_aiogram/migrations/0002_kind_id_index.py +32 -0
- django_aiogram/migrations/__init__.py +1 -0
- django_aiogram/models.py +79 -0
- django_aiogram/producer/__init__.py +13 -0
- django_aiogram/producer/client.py +1540 -0
- django_aiogram/producer/throttling.py +336 -0
- django_aiogram/py.typed +0 -0
- django_aiogram/redis.py +394 -0
- django_aiogram/wire/__init__.py +14 -0
- django_aiogram/wire/envelope.py +146 -0
- django_aiogram/wire/payloads.py +195 -0
- django_aiogram/wire/serializers.py +533 -0
- django_aiogram-4.0.0.dev0.dist-info/METADATA +145 -0
- django_aiogram-4.0.0.dev0.dist-info/RECORD +48 -0
- django_aiogram-4.0.0.dev0.dist-info/WHEEL +4 -0
- django_aiogram-4.0.0.dev0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,308 @@
|
|
|
1
|
+
"""Run the bot: receive updates, and consume the queue Django writes to.
|
|
2
|
+
|
|
3
|
+
This is the long-running process a bot container is built around. It owns two
|
|
4
|
+
things at once — whatever brings updates in, and the consumer that drains the
|
|
5
|
+
Redis queue — and has to shut both down cleanly when the container stops.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import contextlib
|
|
9
|
+
import logging
|
|
10
|
+
import signal
|
|
11
|
+
import threading
|
|
12
|
+
from argparse import ArgumentParser
|
|
13
|
+
from collections.abc import Callable
|
|
14
|
+
from types import FrameType
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from django.core.management import BaseCommand, CommandError
|
|
18
|
+
|
|
19
|
+
from django_aiogram import bot
|
|
20
|
+
from django_aiogram.config.enums import UpdateMode
|
|
21
|
+
from django_aiogram.config.settings import SETTINGS_NAME, coerce_bool, conf
|
|
22
|
+
from django_aiogram.consumer.delivery import Delivery, get_delivery
|
|
23
|
+
from django_aiogram.consumer.webhook import MODES, current_mode
|
|
24
|
+
from django_aiogram.eventlog.events import worker_identity
|
|
25
|
+
from django_aiogram.eventlog.recorder import recorder
|
|
26
|
+
from django_aiogram.redis import read_timeout
|
|
27
|
+
|
|
28
|
+
logger = logging.getLogger('django_aiogram')
|
|
29
|
+
|
|
30
|
+
#: what signal.signal returns: a handler, one of the SIG_* constants, or None
|
|
31
|
+
Handler = Callable[[int, FrameType | None], Any] | int | None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class Command(BaseCommand):
|
|
35
|
+
"""Start the bot and the queue consumer, and stop them together."""
|
|
36
|
+
|
|
37
|
+
help = 'Start telegram bot'
|
|
38
|
+
|
|
39
|
+
#: what --idle waits on; tests replace it so they can end the wait
|
|
40
|
+
idle_event: threading.Event | None = None
|
|
41
|
+
|
|
42
|
+
def add_arguments(self, parser: ArgumentParser) -> None:
|
|
43
|
+
"""Declare --mode and --idle."""
|
|
44
|
+
parser.add_argument(
|
|
45
|
+
'--mode',
|
|
46
|
+
choices=sorted(MODES),
|
|
47
|
+
default=None,
|
|
48
|
+
help=(
|
|
49
|
+
"how updates reach the bot for this run. Defaults to TELEGRAM_BOT['MODE'] "
|
|
50
|
+
"(env: DJANGO_AIOGRAM_MODE), itself 'polling'. In webhook mode this "
|
|
51
|
+
'process consumes the queue and never calls getUpdates, because the updates '
|
|
52
|
+
'arrive over HTTP instead.'
|
|
53
|
+
),
|
|
54
|
+
)
|
|
55
|
+
parser.add_argument(
|
|
56
|
+
'--idle',
|
|
57
|
+
action='store_true',
|
|
58
|
+
help=(
|
|
59
|
+
'When the bot is disabled, block instead of exiting. Useful under '
|
|
60
|
+
'restart policies that treat a clean exit as a crash loop.'
|
|
61
|
+
),
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
def handle(self, *args: Any, **options: Any) -> None:
|
|
65
|
+
"""Receive updates, drain the queue, and unwind both on a signal."""
|
|
66
|
+
if not bot.enabled:
|
|
67
|
+
self.stdout.write(
|
|
68
|
+
self.style.WARNING(
|
|
69
|
+
'django-aiogram is disabled '
|
|
70
|
+
"(TELEGRAM_BOT['ENABLED'] or DJANGO_AIOGRAM_ENABLED); "
|
|
71
|
+
'not starting the bot.'
|
|
72
|
+
)
|
|
73
|
+
)
|
|
74
|
+
if options['idle']:
|
|
75
|
+
self._idle_until_signalled()
|
|
76
|
+
return
|
|
77
|
+
|
|
78
|
+
configured = current_mode()
|
|
79
|
+
mode = options['mode'] or configured
|
|
80
|
+
self.stdout.write(f'Updates arrive by {mode}.')
|
|
81
|
+
if mode != configured:
|
|
82
|
+
# the webhook view reads the setting, not this flag, so it would
|
|
83
|
+
# refuse the updates this process is no longer polling for
|
|
84
|
+
self.stdout.write(
|
|
85
|
+
self.style.WARNING(
|
|
86
|
+
f"--mode {mode} disagrees with TELEGRAM_BOT['MODE'] ({configured}), and it "
|
|
87
|
+
'changes this process only: '
|
|
88
|
+
+ (
|
|
89
|
+
'the webhook view still refuses updates while the setting says polling'
|
|
90
|
+
if mode == UpdateMode.WEBHOOK
|
|
91
|
+
else 'getUpdates fails while a webhook is registered'
|
|
92
|
+
)
|
|
93
|
+
)
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
delivery = get_delivery(handler=bot.send_raw)
|
|
97
|
+
self._preflight(delivery)
|
|
98
|
+
# before a thread exists, because `read_timeout()` refuses a non-numeric
|
|
99
|
+
# `REDIS_TIMEOUT` — and raised from the `finally` below it would skip `close()`,
|
|
100
|
+
# `collect()` and `recorder.stop()`, stranding the drain's own messages
|
|
101
|
+
join_timeout = read_timeout() + 1
|
|
102
|
+
threads: list[threading.Thread] = []
|
|
103
|
+
|
|
104
|
+
# Both modes: starting the consumer before the loop runs would let a
|
|
105
|
+
# backlog reach send_raw while loop.is_running() is still False, so the
|
|
106
|
+
# coroutine would be driven from the consumer thread. Deferring the start
|
|
107
|
+
# until the loop picks up this callback keeps the loop single-threaded.
|
|
108
|
+
# Webhook mode used to start it directly because nothing ran the loop
|
|
109
|
+
# there — something does now, which is what this change is about.
|
|
110
|
+
#
|
|
111
|
+
# Started through a callback on the loop, so it cannot begin before the loop is
|
|
112
|
+
# turning, and refused once the shutdown starts. close() runs one turn of the loop
|
|
113
|
+
# on purpose, so a callback still queued when we reach the finally would
|
|
114
|
+
# start the consumer *after* stop() and after the joins — a thread nobody
|
|
115
|
+
# waits for, doing Redis work, whose first act is reclaim()
|
|
116
|
+
shutting_down = threading.Event()
|
|
117
|
+
|
|
118
|
+
def start_consuming() -> None:
|
|
119
|
+
"""Start the consumer thread on the loop, unless the shutdown got there first.
|
|
120
|
+
|
|
121
|
+
Queued with ``call_soon`` so the thread begins once the loop is turning, and
|
|
122
|
+
gated because the callback can still be pending when the teardown runs — see
|
|
123
|
+
the comment above for what an ungated one starts, and how late.
|
|
124
|
+
"""
|
|
125
|
+
if shutting_down.is_set():
|
|
126
|
+
logger.info('not starting the consumer: the shutdown had already begun')
|
|
127
|
+
return
|
|
128
|
+
threads.append(delivery.start_thread())
|
|
129
|
+
|
|
130
|
+
bot.loop.call_soon(start_consuming)
|
|
131
|
+
previous = self._install_sigterm_handler()
|
|
132
|
+
|
|
133
|
+
try:
|
|
134
|
+
with contextlib.suppress(KeyboardInterrupt, SystemExit):
|
|
135
|
+
if mode == UpdateMode.WEBHOOK:
|
|
136
|
+
self.stdout.write('Consuming the queue; updates are expected over HTTP.')
|
|
137
|
+
self._idle_on_the_loop()
|
|
138
|
+
else:
|
|
139
|
+
bot.start_polling()
|
|
140
|
+
finally:
|
|
141
|
+
logger.info('shutting down')
|
|
142
|
+
# before stop(), so the callback above cannot slip a consumer in
|
|
143
|
+
# behind the joins below
|
|
144
|
+
shutting_down.set()
|
|
145
|
+
delivery.stop()
|
|
146
|
+
for thread in threads:
|
|
147
|
+
# derived from the bound that actually governs the thread: every
|
|
148
|
+
# call it makes is capped by REDIS_TIMEOUT, and its blocking pop
|
|
149
|
+
# by one less than that. BLPOP_TIMEOUT + 1 was six seconds against
|
|
150
|
+
# a worst case of ten, so a consumer that outlived the join went on
|
|
151
|
+
# to acknowledge a message close() had already refused
|
|
152
|
+
thread.join(timeout=join_timeout)
|
|
153
|
+
if thread.is_alive():
|
|
154
|
+
logger.warning(
|
|
155
|
+
'the delivery consumer did not stop in time',
|
|
156
|
+
extra={'tg_timeout': join_timeout},
|
|
157
|
+
)
|
|
158
|
+
try:
|
|
159
|
+
bot.close()
|
|
160
|
+
finally:
|
|
161
|
+
# the sends close() just drained reported themselves finished into a
|
|
162
|
+
# queue whose only reader is the consumer loop, and that returned before
|
|
163
|
+
# the join above — so without this every message the drain delivered
|
|
164
|
+
# stays in the in-flight list and the next start sends it again. A
|
|
165
|
+
# graceful stop duplicated whatever the drain had time to finish, which
|
|
166
|
+
# is the one thing `Delivery.md` says a *kill* is needed for
|
|
167
|
+
try:
|
|
168
|
+
delivery.collect()
|
|
169
|
+
finally:
|
|
170
|
+
# after close(), never before: closing drains in-flight sends,
|
|
171
|
+
# and those are what produce the final rows. In its own finally
|
|
172
|
+
# because a close() that raises must not also lose the rows
|
|
173
|
+
recorder.stop()
|
|
174
|
+
if previous is not None:
|
|
175
|
+
# the command may be called in-process; leaving our handler
|
|
176
|
+
# installed would turn a later SIGTERM into a stray interrupt
|
|
177
|
+
with contextlib.suppress(ValueError):
|
|
178
|
+
signal.signal(signal.SIGTERM, previous)
|
|
179
|
+
|
|
180
|
+
def _idle_on_the_loop(self) -> None:
|
|
181
|
+
"""Wait on the bot's loop rather than on an Event.
|
|
182
|
+
|
|
183
|
+
In webhook mode this process consumes the queue and nothing drove the
|
|
184
|
+
loop, so every send the consumer scheduled sat there until something else
|
|
185
|
+
happened to run it — the next update, or `close()`. `run_forever` is what
|
|
186
|
+
makes a scheduled send run when it is scheduled, and it unwinds on
|
|
187
|
+
SIGTERM exactly as `start_polling` does, so the teardown below is
|
|
188
|
+
unchanged.
|
|
189
|
+
"""
|
|
190
|
+
stop = self.idle_event or threading.Event()
|
|
191
|
+
loop = bot.loop
|
|
192
|
+
|
|
193
|
+
def wait_then_stop() -> None:
|
|
194
|
+
"""Wait for the idle event on this thread, then stop the loop from it.
|
|
195
|
+
|
|
196
|
+
The main thread is inside ``run_forever`` and cannot wait for anything, so
|
|
197
|
+
the wait lives here and reaches the loop through ``call_soon_threadsafe`` —
|
|
198
|
+
the only safe way in from another thread. A loop already closed raises
|
|
199
|
+
``RuntimeError``, which is a race with the teardown and not a fault.
|
|
200
|
+
"""
|
|
201
|
+
stop.wait()
|
|
202
|
+
with contextlib.suppress(RuntimeError):
|
|
203
|
+
loop.call_soon_threadsafe(loop.stop)
|
|
204
|
+
|
|
205
|
+
threading.Thread(target=wait_then_stop, name='tgbot-idle', daemon=True).start()
|
|
206
|
+
loop.run_forever()
|
|
207
|
+
|
|
208
|
+
def _idle_until_signalled(self) -> None:
|
|
209
|
+
"""Hold a disabled container open, and unwind it the way the enabled path does.
|
|
210
|
+
|
|
211
|
+
The same SIGTERM handler, so `docker stop` exits 0 rather than 143: without it the
|
|
212
|
+
signal kills the process outright and a container idling on purpose looked like one
|
|
213
|
+
that crashed. And `recorder.stop()`, because a disabled process with the log on
|
|
214
|
+
still has a writer thread holding a database connection.
|
|
215
|
+
"""
|
|
216
|
+
self.stdout.write('Idling. Send SIGINT or SIGTERM to stop.')
|
|
217
|
+
previous = self._install_sigterm_handler()
|
|
218
|
+
try:
|
|
219
|
+
with contextlib.suppress(KeyboardInterrupt):
|
|
220
|
+
(self.idle_event or threading.Event()).wait()
|
|
221
|
+
finally:
|
|
222
|
+
recorder.stop()
|
|
223
|
+
if previous is not None:
|
|
224
|
+
with contextlib.suppress(ValueError):
|
|
225
|
+
signal.signal(signal.SIGTERM, previous)
|
|
226
|
+
|
|
227
|
+
def _preflight(self, delivery: Delivery) -> None:
|
|
228
|
+
"""Everything worth saying or refusing before a thread exists."""
|
|
229
|
+
self._warn_about_an_unstable_worker_name()
|
|
230
|
+
self._require_crash_safety(delivery)
|
|
231
|
+
|
|
232
|
+
def _warn_about_an_unstable_worker_name(self) -> None:
|
|
233
|
+
"""Say it here, where being the consumer is known.
|
|
234
|
+
|
|
235
|
+
The in-flight list is keyed on the worker's name, so a name that changes when the
|
|
236
|
+
container is replaced strands whatever the old one was sending. As a system check
|
|
237
|
+
this can only be information: `manage.py check` runs in every process, and a check
|
|
238
|
+
cannot tell a consumer from a web tier — as a warning it failed
|
|
239
|
+
`check --fail-level WARNING` in containers that own no in-flight list at all.
|
|
240
|
+
|
|
241
|
+
This process is the consumer. The same rule, reused rather than restated, so the
|
|
242
|
+
two cannot drift.
|
|
243
|
+
"""
|
|
244
|
+
from django_aiogram.config.checks import worker_name_problems # noqa: PLC0415 - no aiogram at import
|
|
245
|
+
|
|
246
|
+
for problem in worker_name_problems():
|
|
247
|
+
logger.warning(
|
|
248
|
+
'the worker name will not survive a replacement container',
|
|
249
|
+
extra={'tg_worker': worker_identity()},
|
|
250
|
+
)
|
|
251
|
+
self.stdout.write(self.style.WARNING(f'WORKER_NAME {problem.message}'))
|
|
252
|
+
|
|
253
|
+
@staticmethod
|
|
254
|
+
def _require_crash_safety(delivery: Delivery) -> None:
|
|
255
|
+
"""Refuse to start where a killed worker loses the message it was sending.
|
|
256
|
+
|
|
257
|
+
Probed here rather than from inside ``run()``: that is a daemon thread, so
|
|
258
|
+
a ``SystemExit`` raised there kills only the thread and leaves a process
|
|
259
|
+
polling updates with a dead consumer. A ``CommandError`` gives a non-zero
|
|
260
|
+
exit and a restart loop somebody can see.
|
|
261
|
+
|
|
262
|
+
``reclaim()`` is the probe, and it is the same call ``run()`` opens with.
|
|
263
|
+
An unreachable Redis returns False with crash safety still intact, which
|
|
264
|
+
must not be read as an old server — a blip is not a reason to refuse to
|
|
265
|
+
start.
|
|
266
|
+
"""
|
|
267
|
+
if not coerce_bool(conf['REQUIRE_CRASH_SAFE'], f"{SETTINGS_NAME}['REQUIRE_CRASH_SAFE']"):
|
|
268
|
+
return
|
|
269
|
+
settled = delivery.reclaim()
|
|
270
|
+
if delivery.crash_safe:
|
|
271
|
+
if not settled:
|
|
272
|
+
# NOPERM and WRONGTYPE come back this way too, and unlike a blip
|
|
273
|
+
# they do not clear. Refusing here would turn every restart into
|
|
274
|
+
# a crash loop, so say plainly that the guarantee is unproven
|
|
275
|
+
# rather than let silence read as a passed check
|
|
276
|
+
logger.warning(
|
|
277
|
+
'could not verify crash-safe delivery: the probe did not settle',
|
|
278
|
+
extra={'tg_key': delivery.queue_key},
|
|
279
|
+
)
|
|
280
|
+
return
|
|
281
|
+
msg = (
|
|
282
|
+
'This Redis predates LMOVE (6.2), so a worker killed mid-send loses that message, '
|
|
283
|
+
f"and {SETTINGS_NAME}['REQUIRE_CRASH_SAFE'] refuses to run that way. Upgrade the "
|
|
284
|
+
'server, or set it to False to accept at-most-once delivery.'
|
|
285
|
+
)
|
|
286
|
+
raise CommandError(msg)
|
|
287
|
+
|
|
288
|
+
@staticmethod
|
|
289
|
+
def _install_sigterm_handler() -> Handler:
|
|
290
|
+
"""Turn SIGTERM into KeyboardInterrupt so `docker stop` unwinds cleanly.
|
|
291
|
+
|
|
292
|
+
Returns the handler it replaced, or None when it could not install one —
|
|
293
|
+
signal.signal only works on the main thread.
|
|
294
|
+
"""
|
|
295
|
+
|
|
296
|
+
def raise_interrupt(_signum: int, _frame: FrameType | None) -> None:
|
|
297
|
+
"""Raise where the signal arrived, which is inside whatever was blocking.
|
|
298
|
+
|
|
299
|
+
That is the whole trick: ``KeyboardInterrupt`` unwinds ``start_polling`` and
|
|
300
|
+
``run_forever`` through the same path a Ctrl-C takes, so one teardown covers
|
|
301
|
+
both an operator and ``docker stop``.
|
|
302
|
+
"""
|
|
303
|
+
raise KeyboardInterrupt
|
|
304
|
+
|
|
305
|
+
try:
|
|
306
|
+
return signal.signal(signal.SIGTERM, raise_interrupt)
|
|
307
|
+
except ValueError:
|
|
308
|
+
return None
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""Answer whether the bot container is doing its job.
|
|
2
|
+
|
|
3
|
+
`docker ps` says the process is up, which is not the same thing: the consumer
|
|
4
|
+
thread can be dead while polling continues, or Redis can be unreachable, and the
|
|
5
|
+
container stays "healthy" either way.
|
|
6
|
+
|
|
7
|
+
A wrapper, since 3.1.0, over :mod:`django_aiogram.healthcheck`. The decision
|
|
8
|
+
lives there so that ``python -m django_aiogram.healthcheck`` can make it without
|
|
9
|
+
``django.setup()`` — a management command populates the app registry and runs every
|
|
10
|
+
``AppConfig.ready()`` in the host project first, which is 17.89s in one measured
|
|
11
|
+
consumer against 0.01s of actual probing, and more than any Docker ``timeout`` can
|
|
12
|
+
honestly allow. This command is unchanged for anyone who has it in a compose file
|
|
13
|
+
today, and **Deployment** says why the other form belongs in a healthcheck.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from argparse import ArgumentParser
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from django.core.management import BaseCommand, CommandError
|
|
20
|
+
|
|
21
|
+
from django_aiogram.healthcheck import add_limit_flags, check
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Command(BaseCommand):
|
|
25
|
+
"""Check Redis, the consumer's heartbeat and the queue length, in that order."""
|
|
26
|
+
|
|
27
|
+
help = 'Exit 0 when the bot container is healthy, non-zero with a reason otherwise'
|
|
28
|
+
|
|
29
|
+
def add_arguments(self, parser: ArgumentParser) -> None:
|
|
30
|
+
"""Declare the two limits, both of which default to a setting.
|
|
31
|
+
|
|
32
|
+
Taken from the module that acts on them rather than restated here: a second copy
|
|
33
|
+
of a flag is how one form ends up with a default the other does not have.
|
|
34
|
+
"""
|
|
35
|
+
add_limit_flags(parser)
|
|
36
|
+
|
|
37
|
+
def handle(self, *args: Any, **options: Any) -> None:
|
|
38
|
+
"""Report the first thing that is wrong, or that everything is fine.
|
|
39
|
+
|
|
40
|
+
``stranded`` and ``guarantee`` are asked for explicitly, because this command's
|
|
41
|
+
output must not change for anyone who has it in a compose file — while the
|
|
42
|
+
container-facing entry point leaves both off, since neither can alter the
|
|
43
|
+
verdict and both are the expensive part of the probe.
|
|
44
|
+
"""
|
|
45
|
+
report = check(
|
|
46
|
+
max_queue=options['max_queue'],
|
|
47
|
+
max_age=options['max_age'],
|
|
48
|
+
stranded=True,
|
|
49
|
+
guarantee=True,
|
|
50
|
+
)
|
|
51
|
+
if not report.ok:
|
|
52
|
+
raise CommandError(report.message)
|
|
53
|
+
# plain when nothing was examined: a disabled process is not a healthy bot, and
|
|
54
|
+
# this command has never colored that line green
|
|
55
|
+
self.stdout.write(self.style.SUCCESS(report.message) if report.checked else report.message)
|
|
56
|
+
for warning in report.warnings:
|
|
57
|
+
self.stdout.write(self.style.WARNING(warning))
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""Delete event log rows older than the retention window.
|
|
2
|
+
|
|
3
|
+
Nothing on the write path deletes anything, so the table grows until this runs.
|
|
4
|
+
Schedule it; check ``W006`` says so while the retention is unset.
|
|
5
|
+
|
|
6
|
+
The deletion walks **primary key ranges**, not ``pk__in``. Two reasons, both
|
|
7
|
+
practical: MySQL rejects ``DELETE ... WHERE id IN (SELECT ... LIMIT n)`` against
|
|
8
|
+
the same table, which is exactly what the ORM generates for the obvious
|
|
9
|
+
formulation; and a bounded range at the cold end of the table cannot conflict
|
|
10
|
+
with the inserts still arriving at the hot end, which is what keeps InnoDB's
|
|
11
|
+
next-key locks out of the picture.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import datetime
|
|
15
|
+
import logging
|
|
16
|
+
import time
|
|
17
|
+
from argparse import ArgumentParser
|
|
18
|
+
from typing import Any, NamedTuple
|
|
19
|
+
|
|
20
|
+
from django.core.management import BaseCommand, CommandError
|
|
21
|
+
from django.db import connections, models, transaction
|
|
22
|
+
from django.utils import timezone
|
|
23
|
+
|
|
24
|
+
from django_aiogram.config.settings import conf
|
|
25
|
+
from django_aiogram.eventlog.writer import log_alias
|
|
26
|
+
from django_aiogram.models import TelegramEvent
|
|
27
|
+
|
|
28
|
+
logger = logging.getLogger('django_aiogram')
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class Window(NamedTuple):
|
|
32
|
+
"""The range one run walks: the rows, its ends, the cutoff and the alias."""
|
|
33
|
+
|
|
34
|
+
rows: models.QuerySet[TelegramEvent]
|
|
35
|
+
low: int
|
|
36
|
+
watermark: int
|
|
37
|
+
cutoff: datetime.datetime
|
|
38
|
+
alias: str
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class Command(BaseCommand):
|
|
42
|
+
"""Prune the event log in bounded chunks, one transaction each."""
|
|
43
|
+
|
|
44
|
+
help = 'Delete bot event rows older than the retention window'
|
|
45
|
+
|
|
46
|
+
def add_arguments(self, parser: ArgumentParser) -> None:
|
|
47
|
+
"""Declare the window, the chunking and the two safety valves."""
|
|
48
|
+
parser.add_argument(
|
|
49
|
+
'--days',
|
|
50
|
+
type=int,
|
|
51
|
+
default=None,
|
|
52
|
+
help="delete rows older than this. Defaults to TELEGRAM_BOT['EVENT_LOG_RETENTION_DAYS'].",
|
|
53
|
+
)
|
|
54
|
+
parser.add_argument(
|
|
55
|
+
'--chunk',
|
|
56
|
+
type=int,
|
|
57
|
+
default=1000,
|
|
58
|
+
help='width of the id range each transaction covers. Rows inside it that are still '
|
|
59
|
+
'within the window are left alone, so a chunk deletes at most this many (default 1000)',
|
|
60
|
+
)
|
|
61
|
+
parser.add_argument(
|
|
62
|
+
'--sleep',
|
|
63
|
+
type=float,
|
|
64
|
+
default=0.1,
|
|
65
|
+
help='seconds between chunks. The valve for replica lag: with row-based binlogs '
|
|
66
|
+
'every deleted row is an event (default 0.1)',
|
|
67
|
+
)
|
|
68
|
+
parser.add_argument(
|
|
69
|
+
'--max-chunks',
|
|
70
|
+
type=int,
|
|
71
|
+
default=0,
|
|
72
|
+
help='stop after this many chunks, so a nightly run has a bounded blast radius. 0 means no limit',
|
|
73
|
+
)
|
|
74
|
+
parser.add_argument('--database', default=None, help='the alias to prune; defaults to the configured one')
|
|
75
|
+
parser.add_argument('--dry-run', action='store_true', help='report what would be deleted, and delete nothing')
|
|
76
|
+
|
|
77
|
+
def handle(self, *args: Any, **options: Any) -> None:
|
|
78
|
+
"""Walk the table by primary key, deleting one bounded range per commit."""
|
|
79
|
+
days = options['days'] if options['days'] is not None else int(conf['EVENT_LOG_RETENTION_DAYS'])
|
|
80
|
+
if days <= 0:
|
|
81
|
+
self.stdout.write('Retention is not set, so nothing is pruned. See EVENT_LOG_RETENTION_DAYS.')
|
|
82
|
+
return
|
|
83
|
+
|
|
84
|
+
alias = options['database'] or log_alias()
|
|
85
|
+
if alias not in connections:
|
|
86
|
+
# E041 guards the setting; the flag bypasses it, in the one command that runs
|
|
87
|
+
# from cron — where a Django traceback is the least useful thing to wake up to
|
|
88
|
+
msg = f'no database is configured under the alias {alias!r}; DATABASES has {sorted(connections)}.'
|
|
89
|
+
raise CommandError(msg)
|
|
90
|
+
cutoff = timezone.now() - datetime.timedelta(days=days)
|
|
91
|
+
rows = TelegramEvent.objects.using(alias)
|
|
92
|
+
|
|
93
|
+
# where the walk stops: nothing older than the cutoff lives above this id.
|
|
94
|
+
# `drai_event_recent` covers the cutoff range, so neither form touches the
|
|
95
|
+
# table — but ordering by id still sorts that range, and `EXPLAIN QUERY PLAN`
|
|
96
|
+
# gives this and `Min(id)` the same two steps: the covering search and one
|
|
97
|
+
# `USE TEMP B-TREE FOR ORDER BY`. Written as a limit rather than an aggregate
|
|
98
|
+
# to read like the `low` below it, not because it measures faster
|
|
99
|
+
expired = rows.filter(created_at__lt=cutoff)
|
|
100
|
+
watermark = expired.order_by('-id').values_list('id', flat=True).first()
|
|
101
|
+
if watermark is None:
|
|
102
|
+
self.stdout.write(f'Nothing older than {cutoff.isoformat()}.')
|
|
103
|
+
return
|
|
104
|
+
# filtered by the cutoff like the watermark is. Taking the table's lowest
|
|
105
|
+
# id instead meant one surviving row down there pinned the walk to
|
|
106
|
+
# restart from it every night, and a --max-chunks run never got past it
|
|
107
|
+
low = expired.order_by('id').values_list('id', flat=True).first() or 1
|
|
108
|
+
|
|
109
|
+
deleted = self._walk(Window(rows, low, watermark, cutoff, alias), options)
|
|
110
|
+
verb = 'would delete' if options['dry_run'] else 'deleted'
|
|
111
|
+
self.stdout.write(f'{verb} {deleted} events older than {cutoff.isoformat()} from {alias!r}.')
|
|
112
|
+
|
|
113
|
+
def _walk(self, window: 'Window', options: dict[str, Any]) -> int:
|
|
114
|
+
"""Delete each id range in turn, pausing between them."""
|
|
115
|
+
rows, low, watermark, cutoff, alias = window
|
|
116
|
+
chunk = max(1, int(options['chunk']))
|
|
117
|
+
limit = max(0, int(options['max_chunks']))
|
|
118
|
+
pause = max(0.0, float(options['sleep']))
|
|
119
|
+
deleted = 0
|
|
120
|
+
rounds = 0
|
|
121
|
+
|
|
122
|
+
while low <= watermark:
|
|
123
|
+
high = min(low + chunk - 1, watermark)
|
|
124
|
+
# the id range is the access path; created_at is the correctness
|
|
125
|
+
# condition, because id order only approximates time order once
|
|
126
|
+
# several processes and a buffered writer are involved
|
|
127
|
+
batch = rows.filter(id__gte=low, id__lte=high, created_at__lt=cutoff)
|
|
128
|
+
if options['dry_run']:
|
|
129
|
+
removed_here = batch.count()
|
|
130
|
+
else:
|
|
131
|
+
with transaction.atomic(using=alias):
|
|
132
|
+
removed_here, _ = batch.delete()
|
|
133
|
+
deleted += removed_here
|
|
134
|
+
low = high + 1
|
|
135
|
+
rounds += 1
|
|
136
|
+
if limit and rounds >= limit:
|
|
137
|
+
self.stdout.write(f'Stopped after {rounds} chunks; rerun to continue.')
|
|
138
|
+
break
|
|
139
|
+
# nothing was deleted, so there is nothing for a replica to catch up
|
|
140
|
+
# on and nothing to vacuum; a dry run deletes nothing at all
|
|
141
|
+
if pause and removed_here and low <= watermark and not options['dry_run']:
|
|
142
|
+
# replicas and autovacuum both need the gaps
|
|
143
|
+
time.sleep(pause)
|
|
144
|
+
return deleted
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Put a dead worker's in-flight messages back on the queue.
|
|
2
|
+
|
|
3
|
+
A message being sent lives in ``<queue>:processing:<worker>`` until the send
|
|
4
|
+
finishes, and the worker that put it there reclaims it on its next start. That
|
|
5
|
+
only works while the name is stable — a container with no ``hostname:`` and no
|
|
6
|
+
``WORKER_NAME`` gets a fresh one every time, and its messages are stranded where
|
|
7
|
+
nothing will ever look for them again.
|
|
8
|
+
|
|
9
|
+
This is the way out, and it is deliberately manual: naming the dead worker is a
|
|
10
|
+
human saying it is dead. Nothing here probes for liveness, because a worker that
|
|
11
|
+
is merely slow looks exactly like one that is gone, and taking a message back
|
|
12
|
+
from a live sender is how you send it twice.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import logging
|
|
16
|
+
from argparse import ArgumentParser
|
|
17
|
+
from typing import Any
|
|
18
|
+
|
|
19
|
+
from django.core.management import BaseCommand, CommandError
|
|
20
|
+
from redis import Redis
|
|
21
|
+
from redis.exceptions import ResponseError
|
|
22
|
+
|
|
23
|
+
from django_aiogram.eventlog.events import worker_identity
|
|
24
|
+
from django_aiogram.redis import get_redis, processing_key, queue_key
|
|
25
|
+
|
|
26
|
+
logger = logging.getLogger('django_aiogram')
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Command(BaseCommand):
|
|
30
|
+
"""Move one worker's in-flight messages back to the queue."""
|
|
31
|
+
|
|
32
|
+
help = 'Requeue the messages a dead worker left in flight'
|
|
33
|
+
|
|
34
|
+
def add_arguments(self, parser: ArgumentParser) -> None:
|
|
35
|
+
"""Declare the worker to reclaim from, and the safety valves."""
|
|
36
|
+
parser.add_argument(
|
|
37
|
+
'--worker',
|
|
38
|
+
required=True,
|
|
39
|
+
help='the WORKER_NAME (or hostname) whose in-flight list to drain. Naming it is you '
|
|
40
|
+
'saying that worker is gone: reclaiming from a live one sends its message twice.',
|
|
41
|
+
)
|
|
42
|
+
parser.add_argument(
|
|
43
|
+
'--limit',
|
|
44
|
+
type=int,
|
|
45
|
+
default=0,
|
|
46
|
+
help='stop after this many messages, so one run has a bounded blast radius. 0 means no '
|
|
47
|
+
'limit. A bounded run takes the newest in flight first, because that is the end of the '
|
|
48
|
+
'list a reclaim pops from',
|
|
49
|
+
)
|
|
50
|
+
parser.add_argument('--dry-run', action='store_true', help='report what is there, and move nothing')
|
|
51
|
+
|
|
52
|
+
@staticmethod
|
|
53
|
+
def _move_one(connection: Redis, source: str, destination: str) -> object:
|
|
54
|
+
"""Move the newest in-flight message back to the front of the queue.
|
|
55
|
+
|
|
56
|
+
Newest, because a message is taken with ``LEFT`` → ``RIGHT`` and so the tail of the
|
|
57
|
+
in-flight list is the most recent one — and this pops that tail. Draining the whole
|
|
58
|
+
list therefore restores the original order, the oldest ending up at the front; a run
|
|
59
|
+
stopped by ``--limit`` has reclaimed the newest and left the older ones in place.
|
|
60
|
+
|
|
61
|
+
``LMOVE ... RIGHT LEFT`` is what ``reclaim()`` uses, so the order a real run
|
|
62
|
+
produces is the order a dry run promises. On a Redis older than 6.2 that command
|
|
63
|
+
does not exist, and this is the one path that can still recover those messages:
|
|
64
|
+
the consumer's own ``reclaim()`` gives up there and runs at-most-once, so a list
|
|
65
|
+
stranded before the downgrade would have nothing else to come back through.
|
|
66
|
+
``RPOPLPUSH`` is the same move, and has been there since 1.2.
|
|
67
|
+
"""
|
|
68
|
+
try:
|
|
69
|
+
return connection.lmove(source, destination, 'RIGHT', 'LEFT')
|
|
70
|
+
except ResponseError as error:
|
|
71
|
+
if 'unknown command' not in str(error).lower():
|
|
72
|
+
raise
|
|
73
|
+
return connection.rpoplpush(source, destination)
|
|
74
|
+
|
|
75
|
+
def handle(self, *args: Any, **options: Any) -> None:
|
|
76
|
+
"""Walk the named worker's in-flight list back onto the queue."""
|
|
77
|
+
worker = str(options['worker']).strip()
|
|
78
|
+
if not worker:
|
|
79
|
+
msg = '--worker cannot be empty.'
|
|
80
|
+
raise CommandError(msg)
|
|
81
|
+
limit = int(options['limit'])
|
|
82
|
+
if limit < 0:
|
|
83
|
+
# max(0, ...) would have read this as "no limit", which is the
|
|
84
|
+
# opposite of what someone typing a limit is asking for. Judged with
|
|
85
|
+
# the other arguments, so --dry-run reports the mistake rather than
|
|
86
|
+
# returning happily on a run that would have been refused
|
|
87
|
+
msg = f'--limit cannot be negative, got {limit}. Use 0 for no limit.'
|
|
88
|
+
raise CommandError(msg)
|
|
89
|
+
if worker == worker_identity():
|
|
90
|
+
# this process would be reclaiming from whatever is running here now,
|
|
91
|
+
# which on a bot container is the consumer that is mid-send
|
|
92
|
+
msg = (
|
|
93
|
+
f"{worker!r} is this process's own worker name. A running consumer reclaims its own "
|
|
94
|
+
'messages when it starts; taking them from underneath it sends them twice.'
|
|
95
|
+
)
|
|
96
|
+
raise CommandError(msg)
|
|
97
|
+
|
|
98
|
+
source, destination = processing_key(worker), queue_key()
|
|
99
|
+
connection = get_redis()
|
|
100
|
+
try:
|
|
101
|
+
waiting = int(connection.llen(source) or 0)
|
|
102
|
+
except Exception as error:
|
|
103
|
+
msg = f'could not read {source}: {error}'
|
|
104
|
+
raise CommandError(msg) from error
|
|
105
|
+
|
|
106
|
+
if not waiting:
|
|
107
|
+
self.stdout.write(f'Nothing in flight for {worker!r}.')
|
|
108
|
+
return
|
|
109
|
+
if options['dry_run']:
|
|
110
|
+
# through the same limit the real run applies. A rehearsal that
|
|
111
|
+
# promises to move two and then moves one is worse than none: it is
|
|
112
|
+
# read as the plan, and the difference shows up as messages left
|
|
113
|
+
# behind that nobody went looking for
|
|
114
|
+
would_move = min(waiting, limit) if limit else waiting
|
|
115
|
+
self.stdout.write(
|
|
116
|
+
f'{waiting} message(s) in flight for {worker!r}; would requeue {would_move} of them.'
|
|
117
|
+
if would_move != waiting
|
|
118
|
+
else f'{waiting} message(s) in flight for {worker!r}; would requeue them.'
|
|
119
|
+
)
|
|
120
|
+
return
|
|
121
|
+
|
|
122
|
+
moved = 0
|
|
123
|
+
while not limit or moved < limit:
|
|
124
|
+
try:
|
|
125
|
+
# `is None`, not falsy: nil is how both commands say the source list is
|
|
126
|
+
# empty, and a payload that happens to be empty is still a message
|
|
127
|
+
if self._move_one(connection, source, destination) is None:
|
|
128
|
+
break
|
|
129
|
+
except Exception as error:
|
|
130
|
+
msg = f'moved {moved} message(s), then failed: {error}'
|
|
131
|
+
raise CommandError(msg) from error
|
|
132
|
+
moved += 1
|
|
133
|
+
|
|
134
|
+
logger.info('reclaimed a dead worker', extra={'tg_key': source, 'tg_count': moved})
|
|
135
|
+
self.stdout.write(self.style.SUCCESS(f'Requeued {moved} message(s) from {worker!r}.'))
|