django-aiogram 4.0.0.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- django_aiogram/__init__.py +64 -0
- django_aiogram/_singleton.py +22 -0
- django_aiogram/admin.py +380 -0
- django_aiogram/api.py +38 -0
- django_aiogram/apps.py +56 -0
- django_aiogram/config/__init__.py +15 -0
- django_aiogram/config/checks.py +759 -0
- django_aiogram/config/defaults.py +102 -0
- django_aiogram/config/enums.py +108 -0
- django_aiogram/config/settings.py +266 -0
- django_aiogram/consumer/__init__.py +13 -0
- django_aiogram/consumer/delivery.py +625 -0
- django_aiogram/consumer/routers.py +19 -0
- django_aiogram/consumer/webhook.py +154 -0
- django_aiogram/context.py +34 -0
- django_aiogram/eventlog/__init__.py +16 -0
- django_aiogram/eventlog/dbrouter.py +55 -0
- django_aiogram/eventlog/events.py +120 -0
- django_aiogram/eventlog/instrumentation.py +231 -0
- django_aiogram/eventlog/recorder.py +922 -0
- django_aiogram/eventlog/signals.py +84 -0
- django_aiogram/eventlog/writer.py +231 -0
- django_aiogram/exceptions.py +60 -0
- django_aiogram/healthcheck.py +412 -0
- django_aiogram/management/__init__.py +1 -0
- django_aiogram/management/commands/__init__.py +1 -0
- django_aiogram/management/commands/start_tgbot.py +308 -0
- django_aiogram/management/commands/tgbot_healthcheck.py +57 -0
- django_aiogram/management/commands/tgbot_prune_events.py +144 -0
- django_aiogram/management/commands/tgbot_reclaim.py +135 -0
- django_aiogram/management/commands/tgbot_webhook.py +87 -0
- django_aiogram/migrations/0001_initial.py +50 -0
- django_aiogram/migrations/0002_kind_id_index.py +32 -0
- django_aiogram/migrations/__init__.py +1 -0
- django_aiogram/models.py +79 -0
- django_aiogram/producer/__init__.py +13 -0
- django_aiogram/producer/client.py +1540 -0
- django_aiogram/producer/throttling.py +336 -0
- django_aiogram/py.typed +0 -0
- django_aiogram/redis.py +394 -0
- django_aiogram/wire/__init__.py +14 -0
- django_aiogram/wire/envelope.py +146 -0
- django_aiogram/wire/payloads.py +195 -0
- django_aiogram/wire/serializers.py +533 -0
- django_aiogram-4.0.0.dev0.dist-info/METADATA +145 -0
- django_aiogram-4.0.0.dev0.dist-info/RECORD +48 -0
- django_aiogram-4.0.0.dev0.dist-info/WHEEL +4 -0
- django_aiogram-4.0.0.dev0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,625 @@
|
|
|
1
|
+
"""The backend that moves queued messages from Redis to Telegram.
|
|
2
|
+
|
|
3
|
+
``blpop`` is the only consumer: a blocking pop needs no server configuration,
|
|
4
|
+
works on any database index, delivers immediately and leaves messages in the
|
|
5
|
+
list while the worker is down. The keyspace consumer 1.x used was removed in
|
|
6
|
+
3.0 — it needed ``CONFIG SET notify-keyspace-events``, which managed Redis
|
|
7
|
+
providers usually refuse, and it could not deliver before the TTL elapsed.
|
|
8
|
+
|
|
9
|
+
It consumes crash-safely where the server allows it: a message is moved to a
|
|
10
|
+
processing list while it is being sent and removed once the send has actually
|
|
11
|
+
finished, so a worker killed mid-send leaves it behind to be reclaimed on the
|
|
12
|
+
next start. That makes delivery at-least-once — after a crash a message may be
|
|
13
|
+
sent twice. Servers older than Redis 6.2 lack ``LMOVE``; there the consumer
|
|
14
|
+
falls back to plain pops, which is the 1.x at-most-once behavior, and says so
|
|
15
|
+
in the log.
|
|
16
|
+
|
|
17
|
+
"Once the send has finished" is doing real work in that sentence. Until 3.1.0 the
|
|
18
|
+
message was acknowledged when the handler *returned*, and ``send_raw`` returns as
|
|
19
|
+
soon as the coroutine is scheduled — so in polling mode the message left the
|
|
20
|
+
in-flight list before Telegram had seen anything, and the guarantee above was
|
|
21
|
+
false. A handler that takes an ``on_complete`` keyword is now handed one and the
|
|
22
|
+
message waits for it; one that does not keeps the old semantics exactly.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
import asyncio
|
|
26
|
+
import hashlib
|
|
27
|
+
import inspect
|
|
28
|
+
import logging
|
|
29
|
+
import queue
|
|
30
|
+
import threading
|
|
31
|
+
import time
|
|
32
|
+
from abc import ABC, abstractmethod
|
|
33
|
+
from collections.abc import Callable
|
|
34
|
+
from typing import Any
|
|
35
|
+
|
|
36
|
+
from redis.exceptions import ResponseError
|
|
37
|
+
|
|
38
|
+
from django_aiogram.api import check_function
|
|
39
|
+
from django_aiogram.config.enums import DeliveryKind, EventKind
|
|
40
|
+
from django_aiogram.config.settings import blpop_ceiling, conf
|
|
41
|
+
from django_aiogram.eventlog.events import new_correlation_id, worker_identity
|
|
42
|
+
from django_aiogram.eventlog.recorder import Event, as_identifier, recorder
|
|
43
|
+
from django_aiogram.redis import (
|
|
44
|
+
as_bytes,
|
|
45
|
+
get_redis,
|
|
46
|
+
heartbeat_interval,
|
|
47
|
+
heartbeat_key,
|
|
48
|
+
heartbeat_ttl,
|
|
49
|
+
processing_key,
|
|
50
|
+
queue_key,
|
|
51
|
+
)
|
|
52
|
+
from django_aiogram.wire.envelope import Envelope, UnknownEnvelopeVersionError, unpack
|
|
53
|
+
from django_aiogram.wire.serializers import PickleReadRefusedError, SerializationError, loads
|
|
54
|
+
|
|
55
|
+
logger = logging.getLogger('django_aiogram')
|
|
56
|
+
|
|
57
|
+
Handler = Callable[..., Any]
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def defers_completion(handler: Handler) -> bool:
|
|
61
|
+
"""Whether ``handler`` will take the callback that says a send has finished.
|
|
62
|
+
|
|
63
|
+
An explicit parameter only. Every documented recipe takes ``**kwargs`` — and
|
|
64
|
+
so does ``TelegramBot.send_raw`` — so treating that as acceptance would hand
|
|
65
|
+
the callback to handlers that never call it, and their messages would sit in
|
|
66
|
+
the in-flight list until a restart reclaimed them.
|
|
67
|
+
|
|
68
|
+
It also has to be a parameter the keyword call can reach. A positional-only
|
|
69
|
+
``on_complete`` reads as acceptance but refuses ``on_complete=...`` with a
|
|
70
|
+
``TypeError``, and that lands in the handler-failed branch — acknowledging a
|
|
71
|
+
message nothing ever sent.
|
|
72
|
+
"""
|
|
73
|
+
return accepts_keyword(handler, 'on_complete')
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def accepts_keyword(handler: Handler, name: str) -> bool:
|
|
77
|
+
"""Whether ``handler`` has a parameter of that name a keyword call can reach.
|
|
78
|
+
|
|
79
|
+
Asked once per callback rather than for the pair together: a handler written to the
|
|
80
|
+
documented recipe takes ``on_complete`` and nothing else, and handing it
|
|
81
|
+
``on_refused`` would be an unexpected keyword — a ``TypeError`` landing in the
|
|
82
|
+
handler-failed branch, acknowledging a message nothing sent. ``send_raw`` takes both,
|
|
83
|
+
so the consumer's real handler gets both.
|
|
84
|
+
"""
|
|
85
|
+
takes_keyword = (inspect.Parameter.POSITIONAL_OR_KEYWORD, inspect.Parameter.KEYWORD_ONLY)
|
|
86
|
+
try:
|
|
87
|
+
parameter = inspect.signature(handler).parameters.get(name)
|
|
88
|
+
except (TypeError, ValueError):
|
|
89
|
+
# a callable signature cannot always be read; the old semantics are safe
|
|
90
|
+
return False
|
|
91
|
+
return parameter is not None and parameter.kind in takes_keyword
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class Delivery(ABC):
|
|
95
|
+
"""Consumes the Redis queue until stopped."""
|
|
96
|
+
|
|
97
|
+
def __init__(self, handler: Handler) -> None:
|
|
98
|
+
"""Take what each decoded message is handed to once it arrives."""
|
|
99
|
+
self.handler = handler
|
|
100
|
+
self._stop = threading.Event()
|
|
101
|
+
self._reliable = True
|
|
102
|
+
self._beat_at = 0.0
|
|
103
|
+
#: messages whose send is over, and whether each may leave the in-flight list.
|
|
104
|
+
#: `(handle, True)` is a finished send and gets acknowledged; `(handle, False)` is
|
|
105
|
+
#: one the producer refused outright, which releases the slot and leaves the
|
|
106
|
+
#: message for a redelivery. Filled from the bot's event loop, drained on this
|
|
107
|
+
#: thread, because every Redis call in this class belongs to the consumer
|
|
108
|
+
self._finished: queue.SimpleQueue[tuple[bytes | str, bool]] = queue.SimpleQueue()
|
|
109
|
+
self._in_flight = 0
|
|
110
|
+
# asked once: a handler that cannot take the callback is acknowledged the
|
|
111
|
+
# moment it returns, which is the behavior every existing caller has
|
|
112
|
+
self._defers = defers_completion(handler)
|
|
113
|
+
# asked separately: a handler may take the completion callback and not its pair
|
|
114
|
+
self._releases = accepts_keyword(handler, 'on_refused')
|
|
115
|
+
# read here rather than per message: `at_capacity` runs inside `run`'s loop, where
|
|
116
|
+
# an unreadable value would raise out of the consumer thread and end delivery for
|
|
117
|
+
# the life of the container. `run` resolves `BLPOP_TIMEOUT` once for the same reason
|
|
118
|
+
self._limit = max(0, int(conf['MAX_IN_FLIGHT']))
|
|
119
|
+
|
|
120
|
+
@property
|
|
121
|
+
def crash_safe(self) -> bool:
|
|
122
|
+
"""Whether a message survives this worker being killed mid-send.
|
|
123
|
+
|
|
124
|
+
False on a Redis without ``LMOVE``, where the pop and the send cannot be
|
|
125
|
+
made one step; ``REQUIRE_CRASH_SAFE`` is how a deployment refuses to run
|
|
126
|
+
that way.
|
|
127
|
+
"""
|
|
128
|
+
return self._reliable
|
|
129
|
+
|
|
130
|
+
@property
|
|
131
|
+
def queue_key(self) -> str:
|
|
132
|
+
"""The list queued messages are written to and read from."""
|
|
133
|
+
return queue_key()
|
|
134
|
+
|
|
135
|
+
@property
|
|
136
|
+
def processing_key(self) -> str:
|
|
137
|
+
"""Per-worker, so a restarting worker reclaims only its own messages.
|
|
138
|
+
|
|
139
|
+
A shared list would let a starting worker pull a message back out from
|
|
140
|
+
under another worker that is still sending it.
|
|
141
|
+
"""
|
|
142
|
+
return processing_key()
|
|
143
|
+
|
|
144
|
+
@abstractmethod
|
|
145
|
+
def run(self) -> None:
|
|
146
|
+
"""Block, consuming messages, until :meth:`stop` is called."""
|
|
147
|
+
|
|
148
|
+
def stop(self) -> None:
|
|
149
|
+
"""Ask :meth:`run` to return after its current read."""
|
|
150
|
+
self._stop.set()
|
|
151
|
+
|
|
152
|
+
def start_thread(self) -> threading.Thread:
|
|
153
|
+
"""Run the consumer on a daemon thread and return it."""
|
|
154
|
+
thread = threading.Thread(target=self.run, name='tgbot-delivery', daemon=True)
|
|
155
|
+
thread.start()
|
|
156
|
+
return thread
|
|
157
|
+
|
|
158
|
+
def reclaim(self) -> bool:
|
|
159
|
+
"""Requeue messages a crashed worker left in the processing list.
|
|
160
|
+
|
|
161
|
+
Also the probe for crash-safe mode: on a server without LMOVE the very
|
|
162
|
+
first call fails, and the consumer downgrades to plain pops.
|
|
163
|
+
|
|
164
|
+
Returns whether the list is settled; False means the caller should try
|
|
165
|
+
again, because a Redis that was unreachable at startup left messages
|
|
166
|
+
stranded there.
|
|
167
|
+
"""
|
|
168
|
+
connection = get_redis()
|
|
169
|
+
count = 0
|
|
170
|
+
try:
|
|
171
|
+
# RIGHT->LEFT keeps the original order at the front of the queue
|
|
172
|
+
while connection.lmove(self.processing_key, self.queue_key, 'RIGHT', 'LEFT'):
|
|
173
|
+
count += 1
|
|
174
|
+
except ResponseError as error:
|
|
175
|
+
if not self._downgrade_without_lmove(error):
|
|
176
|
+
# WRONGTYPE, NOPERM and friends say nothing about LMOVE support
|
|
177
|
+
logger.exception(
|
|
178
|
+
'could not reclaim previous messages, will retry',
|
|
179
|
+
extra={'tg_key': self.processing_key},
|
|
180
|
+
)
|
|
181
|
+
return False
|
|
182
|
+
return True
|
|
183
|
+
except Exception:
|
|
184
|
+
# run() is the thread target, so anything escaping here — a Redis
|
|
185
|
+
# that is not up yet, for one — would end the consumer for good
|
|
186
|
+
logger.exception(
|
|
187
|
+
'could not reclaim previous messages, will retry',
|
|
188
|
+
extra={'tg_key': self.processing_key},
|
|
189
|
+
)
|
|
190
|
+
return False
|
|
191
|
+
if count:
|
|
192
|
+
logger.info(
|
|
193
|
+
'reclaimed messages from a previous run',
|
|
194
|
+
extra={'tg_key': self.queue_key, 'tg_count': count},
|
|
195
|
+
)
|
|
196
|
+
return True
|
|
197
|
+
|
|
198
|
+
def _downgrade_without_lmove(self, error: ResponseError) -> bool:
|
|
199
|
+
"""Fall back to plain pops when the server has no LMOVE, and say whether it did.
|
|
200
|
+
|
|
201
|
+
Shared by the two places that reach for LMOVE first, so a caller draining by
|
|
202
|
+
hand gets the same downgrade the consumer loop gets rather than the raw error.
|
|
203
|
+
"""
|
|
204
|
+
if 'unknown command' not in str(error).lower():
|
|
205
|
+
return False
|
|
206
|
+
if self._reliable:
|
|
207
|
+
self._reliable = False
|
|
208
|
+
logger.warning(
|
|
209
|
+
'crash-safe delivery unavailable: this Redis predates LMOVE (6.2); '
|
|
210
|
+
'a worker killed mid-send may lose that one message',
|
|
211
|
+
extra={'tg_key': self.queue_key},
|
|
212
|
+
)
|
|
213
|
+
return True
|
|
214
|
+
|
|
215
|
+
@property
|
|
216
|
+
def heartbeat_key(self) -> str:
|
|
217
|
+
"""Per worker, like the in-flight list: each one answers for itself."""
|
|
218
|
+
return heartbeat_key()
|
|
219
|
+
|
|
220
|
+
def heartbeat(self) -> None:
|
|
221
|
+
"""Say the loop is still turning, at most once per HEARTBEAT_INTERVAL.
|
|
222
|
+
|
|
223
|
+
A container cannot see a thread in another process. This key is what the
|
|
224
|
+
healthcheck reads — ``python -m django_aiogram.healthcheck`` in a container,
|
|
225
|
+
``tgbot_healthcheck`` by hand — and refreshing it per message would be a write per
|
|
226
|
+
message, so it is paced.
|
|
227
|
+
"""
|
|
228
|
+
now = time.monotonic()
|
|
229
|
+
if now - self._beat_at < heartbeat_interval():
|
|
230
|
+
return
|
|
231
|
+
self._beat_at = now
|
|
232
|
+
try:
|
|
233
|
+
get_redis().set(self.heartbeat_key, str(int(time.time())), ex=heartbeat_ttl())
|
|
234
|
+
except Exception:
|
|
235
|
+
# the loop must keep consuming even when it cannot say so
|
|
236
|
+
logger.exception('could not write the heartbeat', extra={'tg_key': self.heartbeat_key})
|
|
237
|
+
|
|
238
|
+
def collect(self) -> None:
|
|
239
|
+
"""Take every finished send off the in-flight list.
|
|
240
|
+
|
|
241
|
+
Called between reads rather than inside one, so every Redis call this
|
|
242
|
+
class makes still happens on this thread.
|
|
243
|
+
"""
|
|
244
|
+
while True:
|
|
245
|
+
try:
|
|
246
|
+
raw, delivered = self._finished.get_nowait()
|
|
247
|
+
except queue.Empty:
|
|
248
|
+
return
|
|
249
|
+
self._in_flight -= 1
|
|
250
|
+
if delivered:
|
|
251
|
+
self.acknowledge(raw)
|
|
252
|
+
|
|
253
|
+
def _release_for(self, handle: bytes | str) -> Callable[[], None]:
|
|
254
|
+
"""Give back the slot a refused send took, without acknowledging the message.
|
|
255
|
+
|
|
256
|
+
The slot has to come back — `_hand_over` took one before the handler ran — but
|
|
257
|
+
the message must **not** be acknowledged: nothing sent it, so leaving it in the
|
|
258
|
+
in-flight list is what lets the next start pick it up. Without this a refusal
|
|
259
|
+
held its slot for the life of the process, and under ``MAX_IN_FLIGHT`` the
|
|
260
|
+
consumer stopped taking messages entirely once enough had piled up.
|
|
261
|
+
|
|
262
|
+
Latched like its pair, and for the same reason: two reports would take another
|
|
263
|
+
message's place in the count.
|
|
264
|
+
"""
|
|
265
|
+
latch = threading.Lock()
|
|
266
|
+
|
|
267
|
+
def once() -> None:
|
|
268
|
+
"""Give the slot back, once."""
|
|
269
|
+
if latch.acquire(blocking=False):
|
|
270
|
+
self._finished.put((handle, False))
|
|
271
|
+
|
|
272
|
+
return once
|
|
273
|
+
|
|
274
|
+
def _completion_for(self, handle: bytes | str) -> Callable[[], None]:
|
|
275
|
+
"""One report per message, however many times the send says it finished.
|
|
276
|
+
|
|
277
|
+
A latch rather than a flag: two threads can both read an unset flag and
|
|
278
|
+
both report. A second report is not harmless — it takes another message's
|
|
279
|
+
place in the in-flight count, drives it below zero and quietly widens the
|
|
280
|
+
bound ``MAX_IN_FLIGHT`` exists to hold.
|
|
281
|
+
"""
|
|
282
|
+
latch = threading.Lock()
|
|
283
|
+
|
|
284
|
+
def once() -> None:
|
|
285
|
+
"""Report the first finish and drop every later one.
|
|
286
|
+
|
|
287
|
+
The acquire is never released: the lock is a one-way latch here, not a
|
|
288
|
+
critical section, and the first caller through it is the only one that
|
|
289
|
+
should reach the in-flight count.
|
|
290
|
+
"""
|
|
291
|
+
if latch.acquire(blocking=False):
|
|
292
|
+
self._finished.put((handle, True))
|
|
293
|
+
|
|
294
|
+
return once
|
|
295
|
+
|
|
296
|
+
def at_capacity(self) -> bool:
|
|
297
|
+
"""Whether this consumer is already holding as many sends as it may."""
|
|
298
|
+
return bool(self._limit) and self._in_flight >= self._limit
|
|
299
|
+
|
|
300
|
+
def hold_for_capacity(self) -> None:
|
|
301
|
+
"""Stop taking messages while too many are still in flight.
|
|
302
|
+
|
|
303
|
+
The bound is on the in-flight list as much as on memory: acknowledging is
|
|
304
|
+
an ``LREM``, which scans that list, so letting a backlog accumulate there
|
|
305
|
+
turns draining it into quadratic work. Zero, the default, is the
|
|
306
|
+
behavior that shipped before deferred acknowledgement existed.
|
|
307
|
+
|
|
308
|
+
The wait keeps writing the heartbeat, for the same reason ``run()`` caps
|
|
309
|
+
the blocking pop at ``HEARTBEAT_INTERVAL``: a worker at its limit is busy,
|
|
310
|
+
not dead. Held silently past the key's :func:`heartbeat_ttl` it would be
|
|
311
|
+
restarted while healthy, and the messages it was still sending reclaimed
|
|
312
|
+
and sent again.
|
|
313
|
+
"""
|
|
314
|
+
while self.at_capacity() and not self._stop.is_set():
|
|
315
|
+
self.heartbeat()
|
|
316
|
+
try:
|
|
317
|
+
raw, delivered = self._finished.get(timeout=1)
|
|
318
|
+
except queue.Empty:
|
|
319
|
+
continue
|
|
320
|
+
self._in_flight -= 1
|
|
321
|
+
if delivered:
|
|
322
|
+
self.acknowledge(raw)
|
|
323
|
+
|
|
324
|
+
def acknowledge(self, raw: bytes | str) -> None:
|
|
325
|
+
"""Drop a delivered message from the processing list."""
|
|
326
|
+
if not self._reliable:
|
|
327
|
+
return
|
|
328
|
+
try:
|
|
329
|
+
# redis-py's stubs say str, but bytes round-trip identically
|
|
330
|
+
get_redis().lrem(self.processing_key, 1, raw) # type: ignore[arg-type]
|
|
331
|
+
except Exception:
|
|
332
|
+
# worst case the message is redelivered on the next start
|
|
333
|
+
logger.exception(
|
|
334
|
+
'failed to acknowledge a delivered message',
|
|
335
|
+
extra={'tg_key': self.processing_key},
|
|
336
|
+
)
|
|
337
|
+
|
|
338
|
+
def _read(self, raw: bytes) -> tuple['Envelope | None', bool]:
|
|
339
|
+
"""Turn one message off the queue into an envelope, or into a verdict.
|
|
340
|
+
|
|
341
|
+
Everything here is untrusted input, so no failure may escape: what comes
|
|
342
|
+
back is either the envelope or `None` plus whether to acknowledge the
|
|
343
|
+
message that never became one.
|
|
344
|
+
"""
|
|
345
|
+
try:
|
|
346
|
+
payload = loads(raw)
|
|
347
|
+
except PickleReadRefusedError:
|
|
348
|
+
logger.exception(
|
|
349
|
+
'leaving a refused pickle message in flight; set ALLOW_PICKLE to deliver it',
|
|
350
|
+
extra={'tg_key': self.processing_key},
|
|
351
|
+
)
|
|
352
|
+
return None, False
|
|
353
|
+
except SerializationError:
|
|
354
|
+
self._record_undecodable(raw, 'serialization')
|
|
355
|
+
logger.exception('dropping undecodable queued message')
|
|
356
|
+
return None, True
|
|
357
|
+
except Exception:
|
|
358
|
+
self._record_undecodable(raw, 'unknown')
|
|
359
|
+
logger.exception('dropping queued message that failed to decode')
|
|
360
|
+
return None, True
|
|
361
|
+
try:
|
|
362
|
+
return unpack(payload), True
|
|
363
|
+
except UnknownEnvelopeVersionError:
|
|
364
|
+
# written by a newer producer than this consumer understands, so
|
|
365
|
+
# leaving it in flight is what lets an upgrade deliver it
|
|
366
|
+
logger.exception('leaving a message from a newer version in flight')
|
|
367
|
+
return None, False
|
|
368
|
+
except Exception:
|
|
369
|
+
# MalformedEnvelopeError and whatever else a hostile payload can
|
|
370
|
+
# provoke: nothing will ever make sense of it, so it is
|
|
371
|
+
# acknowledged rather than left to come back for ever — and this
|
|
372
|
+
# reader is on the far side of a trust boundary, where an escaping
|
|
373
|
+
# exception would end the consumer for the life of the container
|
|
374
|
+
self._record_undecodable(raw, 'envelope')
|
|
375
|
+
logger.exception('dropping a queued message whose envelope cannot be read')
|
|
376
|
+
return None, True
|
|
377
|
+
|
|
378
|
+
def dispatch(self, raw: bytes, handle: bytes | str | None = None) -> bool:
|
|
379
|
+
"""Decode one message and hand it to the handler.
|
|
380
|
+
|
|
381
|
+
A bad payload is one message's problem, so everything short of a kill is
|
|
382
|
+
logged and dropped: the consumer has to survive it to deliver the rest.
|
|
383
|
+
|
|
384
|
+
Returns whether the message should be acknowledged. Four paths say no, in
|
|
385
|
+
two kinds. Three are refusals that leave a valid payload for somebody else:
|
|
386
|
+
a pickle the configuration refuses, an envelope from a newer version, and a
|
|
387
|
+
handler raising ``CancelledError``, whose outcome is *unknown* rather than
|
|
388
|
+
nothing — a send can be cancelled after Telegram has taken the request — at
|
|
389
|
+
shutdown usually, but the ``except`` is unqualified, so any cancellation counts.
|
|
390
|
+
Acknowledging any of the three would destroy a message over a setting, a deploy
|
|
391
|
+
order or a restart. The fourth is
|
|
392
|
+
:meth:`_hand_over` returning ``not deferring``, which is not a refusal: a handler
|
|
393
|
+
that took ``on_complete`` *signals* completion through it, the handle goes into a
|
|
394
|
+
queue, and :meth:`collect` takes the message off the in-flight list on the
|
|
395
|
+
consumer's next turn. That is what makes at-least-once true — **where there is an
|
|
396
|
+
in-flight list**. Without ``LMOVE`` the plain pop has already removed the message
|
|
397
|
+
and :meth:`acknowledge` is a no-op, so deferring the acknowledgement defers
|
|
398
|
+
nothing: that server is at-most-once whatever the handler does.
|
|
399
|
+
|
|
400
|
+
Those three refusals save the message only where there *is* an in-flight list.
|
|
401
|
+
Against a server without ``LMOVE`` the consumer falls back to a plain pop, so the
|
|
402
|
+
message is gone before the refusal happens and ``False`` buys nothing: what they
|
|
403
|
+
avoid there is a second delete, not a loss.
|
|
404
|
+
"""
|
|
405
|
+
if handle is None:
|
|
406
|
+
handle = raw
|
|
407
|
+
envelope, acknowledge = self._read(raw)
|
|
408
|
+
if envelope is None:
|
|
409
|
+
return acknowledge
|
|
410
|
+
try:
|
|
411
|
+
check_function(envelope.function)
|
|
412
|
+
except ValueError:
|
|
413
|
+
self._record(
|
|
414
|
+
EventKind.QUEUE_REJECTED,
|
|
415
|
+
envelope,
|
|
416
|
+
error='not a Telegram API method',
|
|
417
|
+
)
|
|
418
|
+
logger.exception(
|
|
419
|
+
'dropping queued message naming a method that is not Telegram API',
|
|
420
|
+
extra={'tg_function': envelope.function},
|
|
421
|
+
)
|
|
422
|
+
return True
|
|
423
|
+
self._record(EventKind.OUTBOUND_CONSUMED, envelope)
|
|
424
|
+
# by keyword, the way 2.x splatted it: a handler taking **kwargs
|
|
425
|
+
# only — which every documented recipe does — refuses a positional.
|
|
426
|
+
#
|
|
427
|
+
# The envelope's own fields go in *after* the payload, for the reason
|
|
428
|
+
# `_hand_over` gives about `on_complete`: the queue is a trust boundary, and
|
|
429
|
+
# spreading last let a payload carrying `function` replace the name
|
|
430
|
+
# `check_function` had just validated. `send_raw` validates again and so refuses
|
|
431
|
+
# an unknown one, but a handler taking only `**kwargs` does not — and
|
|
432
|
+
# `correlation_id` and `queued_at` were replaceable either way, which is the
|
|
433
|
+
# event log's correlation and its queue latency
|
|
434
|
+
call: dict[str, Any] = {
|
|
435
|
+
**envelope.kwargs,
|
|
436
|
+
'function': envelope.function,
|
|
437
|
+
'correlation_id': envelope.correlation_id,
|
|
438
|
+
'queued_at': envelope.queued_at,
|
|
439
|
+
}
|
|
440
|
+
return self._hand_over(envelope, call, handle)
|
|
441
|
+
|
|
442
|
+
def _hand_over(self, envelope: Envelope, call: dict[str, Any], handle: bytes | str) -> bool:
|
|
443
|
+
"""Call the handler, and say whether the message may be acknowledged.
|
|
444
|
+
|
|
445
|
+
Cancellation is the reason this is not one ``except``: it is a
|
|
446
|
+
``BaseException``, so letting it through would leave :meth:`run` and end
|
|
447
|
+
the consumer for the life of the container. The message stays in flight because
|
|
448
|
+
the outcome is *unknown*: a send can be cancelled after Telegram has taken the
|
|
449
|
+
request, so leaving it risks a duplicate rather than a loss. This worker has to
|
|
450
|
+
keep reading either way.
|
|
451
|
+
"""
|
|
452
|
+
deferring = self._defers
|
|
453
|
+
if deferring:
|
|
454
|
+
# into the dict, never alongside it as a second keyword. The queue is
|
|
455
|
+
# a trust boundary and send() forwards whatever it was given, so a
|
|
456
|
+
# payload can carry this name — as a keyword that is "got multiple
|
|
457
|
+
# values", a TypeError landing in the failure branch below, which
|
|
458
|
+
# acknowledges a message nothing sent. Assigning simply wins
|
|
459
|
+
call['on_complete'] = self._completion_for(handle)
|
|
460
|
+
if self._releases:
|
|
461
|
+
# its pair, so a producer that refuses the send gives the slot back
|
|
462
|
+
call['on_refused'] = self._release_for(handle)
|
|
463
|
+
self._in_flight += 1
|
|
464
|
+
try:
|
|
465
|
+
self.handler(**call)
|
|
466
|
+
except asyncio.CancelledError:
|
|
467
|
+
if deferring:
|
|
468
|
+
self._in_flight -= 1
|
|
469
|
+
logger.warning(
|
|
470
|
+
'a queued send was cancelled; leaving it in flight',
|
|
471
|
+
extra={'tg_function': envelope.function},
|
|
472
|
+
)
|
|
473
|
+
return False
|
|
474
|
+
except Exception:
|
|
475
|
+
if deferring:
|
|
476
|
+
self._in_flight -= 1
|
|
477
|
+
logger.exception(
|
|
478
|
+
'handler failed for queued message',
|
|
479
|
+
extra={'tg_function': envelope.function},
|
|
480
|
+
)
|
|
481
|
+
return True
|
|
482
|
+
# a deferring handler decides when this message is done. Returning True
|
|
483
|
+
# here is what made the at-least-once promise false: send_raw returns as
|
|
484
|
+
# soon as the coroutine is scheduled, long before Telegram has seen it
|
|
485
|
+
return not deferring
|
|
486
|
+
|
|
487
|
+
def _record(self, kind: EventKind, envelope: Envelope, error: str = '') -> None:
|
|
488
|
+
"""Record what the consumer did with one message."""
|
|
489
|
+
chat_id = envelope.kwargs.get('chat_id')
|
|
490
|
+
recorder.record(
|
|
491
|
+
Event(
|
|
492
|
+
kind=kind.value,
|
|
493
|
+
correlation_id=envelope.correlation_id or new_correlation_id(),
|
|
494
|
+
function=envelope.function,
|
|
495
|
+
chat_id=as_identifier(chat_id),
|
|
496
|
+
worker=worker_identity(),
|
|
497
|
+
error=error,
|
|
498
|
+
detail=self._queue_latency(envelope),
|
|
499
|
+
)
|
|
500
|
+
)
|
|
501
|
+
|
|
502
|
+
@staticmethod
|
|
503
|
+
def _queue_latency(envelope: Envelope) -> dict[str, Any]:
|
|
504
|
+
"""How long the message waited, when the producer said when it was queued."""
|
|
505
|
+
if not envelope.queued_at:
|
|
506
|
+
return {}
|
|
507
|
+
return {'queue_ms': int((time.time() - envelope.queued_at) * 1000)}
|
|
508
|
+
|
|
509
|
+
def _record_undecodable(self, raw: bytes, reason: str) -> None:
|
|
510
|
+
"""Record a payload nothing could read.
|
|
511
|
+
|
|
512
|
+
A fingerprint, never the bytes: an undecodable payload is by definition
|
|
513
|
+
untrusted input and may be a pickle, so putting it in a JSON column
|
|
514
|
+
would spread it into every log shipper and admin page downstream.
|
|
515
|
+
"""
|
|
516
|
+
recorder.record(
|
|
517
|
+
Event(
|
|
518
|
+
kind=EventKind.QUEUE_UNDECODABLE.value,
|
|
519
|
+
worker=worker_identity(),
|
|
520
|
+
error=reason,
|
|
521
|
+
detail={'bytes': len(raw), 'sha256': hashlib.sha256(raw).hexdigest()[:16]},
|
|
522
|
+
)
|
|
523
|
+
)
|
|
524
|
+
|
|
525
|
+
def consume_pending(self) -> None:
|
|
526
|
+
"""Drain the queue without blocking, acknowledging each message."""
|
|
527
|
+
connection = get_redis()
|
|
528
|
+
raw: bytes | str | None
|
|
529
|
+
while not self._stop.is_set():
|
|
530
|
+
self.collect()
|
|
531
|
+
if self.at_capacity():
|
|
532
|
+
# the blocking loop waits here; a drain has no thread to wait on,
|
|
533
|
+
# so it stops instead of scheduling past the bound
|
|
534
|
+
return
|
|
535
|
+
if self._reliable:
|
|
536
|
+
try:
|
|
537
|
+
raw = connection.lmove(self.queue_key, self.processing_key, 'LEFT', 'RIGHT')
|
|
538
|
+
except ResponseError as error:
|
|
539
|
+
# run() learns this from reclaim(); nothing probes for a caller
|
|
540
|
+
# draining by hand, so without this the first pop against a
|
|
541
|
+
# pre-6.2 server raises out of a documented helper
|
|
542
|
+
if not self._downgrade_without_lmove(error):
|
|
543
|
+
raise
|
|
544
|
+
continue
|
|
545
|
+
else:
|
|
546
|
+
# lpop only widens to a list when given a count
|
|
547
|
+
raw = connection.lpop(self.queue_key) # type: ignore[assignment]
|
|
548
|
+
if raw is None:
|
|
549
|
+
self.collect()
|
|
550
|
+
return
|
|
551
|
+
if self.dispatch(as_bytes(raw), raw):
|
|
552
|
+
self.acknowledge(raw)
|
|
553
|
+
self.collect()
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
class BlpopDelivery(Delivery):
|
|
557
|
+
"""Blocks on the queue itself, so a message is delivered as it arrives."""
|
|
558
|
+
|
|
559
|
+
def run(self) -> None:
|
|
560
|
+
"""Block on the queue until :meth:`stop` is called."""
|
|
561
|
+
# 0 means "block for ever" in Redis, which would swallow stop(); the
|
|
562
|
+
# heartbeat would expire under a consumer that is doing fine; and a pop asked
|
|
563
|
+
# to wait longer than the socket will turns an idle round into an error.
|
|
564
|
+
# `blpop_ceiling()` weighs the last two and this line applies the first against
|
|
565
|
+
# them, which is why `bound_by` never names `BLPOP_TIMEOUT`. `W004` reports on the
|
|
566
|
+
# same helper — one place, so the check cannot describe a cap the consumer does
|
|
567
|
+
# not use
|
|
568
|
+
timeout = max(1, min(int(conf['BLPOP_TIMEOUT']), blpop_ceiling().seconds))
|
|
569
|
+
connection = get_redis()
|
|
570
|
+
reclaimed = self.reclaim()
|
|
571
|
+
logger.info(
|
|
572
|
+
'delivery started',
|
|
573
|
+
extra={
|
|
574
|
+
'tg_delivery': DeliveryKind.BLPOP.value,
|
|
575
|
+
'tg_key': self.queue_key,
|
|
576
|
+
'tg_timeout': timeout,
|
|
577
|
+
'tg_crash_safe': self._reliable,
|
|
578
|
+
},
|
|
579
|
+
)
|
|
580
|
+
raw: bytes | str | None
|
|
581
|
+
while not self._stop.is_set():
|
|
582
|
+
self.heartbeat()
|
|
583
|
+
self.collect()
|
|
584
|
+
self.hold_for_capacity()
|
|
585
|
+
if self._stop.is_set():
|
|
586
|
+
# the gate above releases on shutdown as well as on capacity, and
|
|
587
|
+
# without this the loop would go on to take one more message it
|
|
588
|
+
# has no intention of sending
|
|
589
|
+
break
|
|
590
|
+
if not reclaimed:
|
|
591
|
+
reclaimed = self.reclaim()
|
|
592
|
+
try:
|
|
593
|
+
if self._reliable:
|
|
594
|
+
raw = connection.blmove(self.queue_key, self.processing_key, timeout, 'LEFT', 'RIGHT')
|
|
595
|
+
else:
|
|
596
|
+
item = connection.blpop([self.queue_key], timeout=timeout)
|
|
597
|
+
raw = None if item is None else item[1]
|
|
598
|
+
except Exception:
|
|
599
|
+
# a dropped connection must not kill the worker thread
|
|
600
|
+
logger.exception('blocking pop failed, retrying', extra={'tg_key': self.queue_key})
|
|
601
|
+
self._stop.wait(timeout)
|
|
602
|
+
continue
|
|
603
|
+
if raw is None:
|
|
604
|
+
continue
|
|
605
|
+
if self.dispatch(as_bytes(raw), raw):
|
|
606
|
+
self.acknowledge(raw)
|
|
607
|
+
# sends that finished while the last read was blocking still have to
|
|
608
|
+
# leave the in-flight list, or every stop redelivers them
|
|
609
|
+
self.collect()
|
|
610
|
+
|
|
611
|
+
|
|
612
|
+
# keyed by the enum's value, so the keys are the plain strings the setting holds
|
|
613
|
+
DELIVERIES: dict[str, type[Delivery]] = {
|
|
614
|
+
DeliveryKind.BLPOP.value: BlpopDelivery,
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
|
|
618
|
+
def get_delivery(handler: Handler) -> Delivery:
|
|
619
|
+
"""Build the consumer the DELIVERY setting names."""
|
|
620
|
+
name = conf['DELIVERY']
|
|
621
|
+
try:
|
|
622
|
+
return DELIVERIES[name](handler)
|
|
623
|
+
except KeyError:
|
|
624
|
+
msg = f'Unknown delivery {name!r}, expected one of {sorted(DELIVERIES)}.'
|
|
625
|
+
raise ValueError(msg) from None
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Collect the handlers each installed app contributes.
|
|
2
|
+
|
|
3
|
+
Handlers register themselves on the shared bot as a side effect of being
|
|
4
|
+
imported, so discovery is nothing more than importing one module per app.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from django.utils.module_loading import autodiscover_modules
|
|
8
|
+
|
|
9
|
+
from django_aiogram.config.settings import conf
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def autodiscover_tg_routers() -> None:
|
|
13
|
+
"""Import <app>.<MODULE_NAME> for every installed app.
|
|
14
|
+
|
|
15
|
+
Django's helper tells "the app has no such module" apart from "the module
|
|
16
|
+
exists but raised on import", so a broken router surfaces instead of being
|
|
17
|
+
silently swallowed.
|
|
18
|
+
"""
|
|
19
|
+
autodiscover_modules(conf['MODULE_NAME'])
|