librus-python-api 1.0.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- librus_python_api/__init__.py +243 -0
- librus_python_api/_notification_codec.py +310 -0
- librus_python_api/_storage.py +367 -0
- librus_python_api/announcements.py +159 -0
- librus_python_api/attachment_routes.py +114 -0
- librus_python_api/attachments.py +297 -0
- librus_python_api/attendance.py +182 -0
- librus_python_api/attendance_frequency.py +112 -0
- librus_python_api/budget.py +79 -0
- librus_python_api/checkpoint.py +61 -0
- librus_python_api/completed_lessons.py +216 -0
- librus_python_api/config.py +1410 -0
- librus_python_api/detail_fields.py +50 -0
- librus_python_api/diagnostics.py +25 -0
- librus_python_api/exceptions.py +172 -0
- librus_python_api/files.py +156 -0
- librus_python_api/grade_parsers.py +169 -0
- librus_python_api/grade_records.py +454 -0
- librus_python_api/homework_range.py +41 -0
- librus_python_api/lifecycle.py +24 -0
- librus_python_api/markup.py +147 -0
- librus_python_api/message_content.py +230 -0
- librus_python_api/messages.py +288 -0
- librus_python_api/models.py +1160 -0
- librus_python_api/modern_body.py +75 -0
- librus_python_api/modern_mailbox.py +459 -0
- librus_python_api/modern_messages.py +276 -0
- librus_python_api/notification_models.py +67 -0
- librus_python_api/notification_persistence.py +1044 -0
- librus_python_api/notification_workflow.py +303 -0
- librus_python_api/notifications.py +216 -0
- librus_python_api/parsers.py +232 -0
- librus_python_api/parsing.py +49 -0
- librus_python_api/persistence.py +405 -0
- librus_python_api/py.typed +0 -0
- librus_python_api/recipients.py +271 -0
- librus_python_api/scheduler.py +287 -0
- librus_python_api/school_reads.py +400 -0
- librus_python_api/sending.py +125 -0
- librus_python_api/service.py +2285 -0
- librus_python_api/timetable.py +261 -0
- librus_python_api/transport.py +956 -0
- librus_python_api-1.0.0rc1.dist-info/METADATA +254 -0
- librus_python_api-1.0.0rc1.dist-info/RECORD +46 -0
- librus_python_api-1.0.0rc1.dist-info/WHEEL +4 -0
- librus_python_api-1.0.0rc1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
"""Bounded inert Base64 reader layouts, without XML entity or network access."""
|
|
2
|
+
|
|
3
|
+
import base64
|
|
4
|
+
import binascii
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from lxml import etree
|
|
8
|
+
|
|
9
|
+
from librus_python_api.config import MESSAGE_MAX_CONTENT_LENGTH
|
|
10
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
11
|
+
from librus_python_api.markup import text
|
|
12
|
+
from librus_python_api.parsers import parse_html_document
|
|
13
|
+
|
|
14
|
+
_XML_PREFIX = re.compile(
|
|
15
|
+
r"\s*(?:(?:<!--.*?-->|<\?(?!xml\b).*?\?>)\s*)*"
|
|
16
|
+
r"(?:<\?xml\b|<Message(?=[\s/>]))",
|
|
17
|
+
re.DOTALL,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def render_body(encoded: str) -> str:
|
|
22
|
+
if type(encoded) is not str:
|
|
23
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
24
|
+
if len(encoded) > 4 * MESSAGE_MAX_CONTENT_LENGTH:
|
|
25
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
26
|
+
try:
|
|
27
|
+
decoded = base64.b64decode(encoded, validate=True).decode("utf-8", "strict")
|
|
28
|
+
except (ValueError, UnicodeError, binascii.Error):
|
|
29
|
+
raise LibrusError(ErrorKind.PARSE) from None
|
|
30
|
+
if len(decoded.encode()) > MESSAGE_MAX_CONTENT_LENGTH:
|
|
31
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
32
|
+
if re.search(r"<!\s*(?:DOCTYPE|ENTITY)\b", decoded, re.IGNORECASE):
|
|
33
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
34
|
+
decoded = decoded.removeprefix("\ufeff")
|
|
35
|
+
# XML comments/PIs and a UTF-8 BOM must not route a Message wrapper through
|
|
36
|
+
# the permissive HTML renderer, losing CDATA or bypassing Content checks.
|
|
37
|
+
if _XML_PREFIX.match(decoded):
|
|
38
|
+
decoded = _xml_content(decoded)
|
|
39
|
+
if not decoded:
|
|
40
|
+
return ""
|
|
41
|
+
# The official reader preserves physical line endings before HTML rendering.
|
|
42
|
+
decoded = re.sub(r"\r\n|\r|\n", "<br>", decoded)
|
|
43
|
+
return text(
|
|
44
|
+
parse_html_document(decoded.encode()),
|
|
45
|
+
MESSAGE_MAX_CONTENT_LENGTH,
|
|
46
|
+
multiline=True,
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _xml_content(decoded: str) -> str:
|
|
51
|
+
parser = etree.XMLParser(
|
|
52
|
+
no_network=True,
|
|
53
|
+
resolve_entities=False,
|
|
54
|
+
load_dtd=False,
|
|
55
|
+
recover=False,
|
|
56
|
+
huge_tree=False,
|
|
57
|
+
)
|
|
58
|
+
try:
|
|
59
|
+
root = etree.fromstring(decoded.encode(), parser=parser)
|
|
60
|
+
except (ValueError, etree.LxmlError):
|
|
61
|
+
raise LibrusError(ErrorKind.PARSE) from None
|
|
62
|
+
if root.tag != "Message" or root.getroottree().docinfo.encoding.upper() not in {
|
|
63
|
+
"UTF-8",
|
|
64
|
+
"UTF8",
|
|
65
|
+
}:
|
|
66
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
67
|
+
nodes = list(root.iter())
|
|
68
|
+
if len(nodes) > 8192 or any(
|
|
69
|
+
len(tuple(node.iterancestors())) > 32 for node in nodes
|
|
70
|
+
):
|
|
71
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
72
|
+
contents = list(root.iter("Content"))
|
|
73
|
+
if len(contents) != 1 or len(contents[0]) or contents[0].attrib:
|
|
74
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
75
|
+
return contents[0].text or ""
|
|
@@ -0,0 +1,459 @@
|
|
|
1
|
+
"""Bounded modern mailbox parsing and continuation, with distinct backend references."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import re
|
|
5
|
+
from collections.abc import Awaitable, Callable
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from datetime import datetime
|
|
8
|
+
from typing import Any, Literal
|
|
9
|
+
|
|
10
|
+
from librus_python_api.config import (
|
|
11
|
+
MESSAGE_MAX_ATTACHMENTS,
|
|
12
|
+
MESSAGE_MAX_BATCH_ITEMS,
|
|
13
|
+
MESSAGE_MAX_BATCH_PAGES,
|
|
14
|
+
MESSAGE_MAX_CURSOR_IDS,
|
|
15
|
+
MESSAGE_MAX_FIELD_LENGTH,
|
|
16
|
+
MESSAGE_MAX_TOTAL_TEXT_LENGTH,
|
|
17
|
+
modern_mailbox_query,
|
|
18
|
+
)
|
|
19
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
20
|
+
from librus_python_api.models import (
|
|
21
|
+
MessageFolder,
|
|
22
|
+
ModernMessageAttachment,
|
|
23
|
+
ModernMessageAttachmentReference,
|
|
24
|
+
ModernMessageRecipientReceipt,
|
|
25
|
+
ModernMessageReference,
|
|
26
|
+
ModernMessagesCursor,
|
|
27
|
+
ModernMessagesPage,
|
|
28
|
+
ModernMessageSummary,
|
|
29
|
+
)
|
|
30
|
+
from librus_python_api.modern_body import render_body
|
|
31
|
+
from librus_python_api.parsers import decode_json
|
|
32
|
+
|
|
33
|
+
MAX_TOTAL_MESSAGES = 50000
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def identifier(value: Any) -> str:
|
|
37
|
+
# Modern message/attachment IDs are an explicit bounded compatibility policy;
|
|
38
|
+
# directory identifiers remain independently validated strict strings.
|
|
39
|
+
if type(value) is int and 0 < value < 10**64:
|
|
40
|
+
value = str(value)
|
|
41
|
+
if type(value) is not str or re.fullmatch(r"[0-9]{1,64}", value) is None:
|
|
42
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
43
|
+
return value
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def validate_reference(reference: ModernMessageReference, account: str) -> None:
|
|
47
|
+
if (
|
|
48
|
+
not isinstance(reference, ModernMessageReference)
|
|
49
|
+
or not isinstance(reference.folder, MessageFolder)
|
|
50
|
+
or reference.account != account
|
|
51
|
+
or type(reference.identifier) is not str
|
|
52
|
+
or re.fullmatch(r"[0-9]{1,64}", reference.identifier) is None
|
|
53
|
+
):
|
|
54
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _string(value: Any, *, limit: int = MESSAGE_MAX_FIELD_LENGTH) -> str:
|
|
58
|
+
if type(value) is not str or len(value) > limit or "\x00" in value:
|
|
59
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
60
|
+
return value
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _date(value: Any) -> datetime:
|
|
64
|
+
value = _string(value)
|
|
65
|
+
if not re.fullmatch(
|
|
66
|
+
r"[0-9]{4}-[0-9]{2}-[0-9]{2}[ T][0-9]{2}:[0-9]{2}:[0-9]{2}"
|
|
67
|
+
r"(?:\.[0-9]{1,6})?(?:Z|[+-][0-9]{2}:[0-9]{2})?",
|
|
68
|
+
value,
|
|
69
|
+
):
|
|
70
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
71
|
+
try:
|
|
72
|
+
return datetime.fromisoformat(value)
|
|
73
|
+
except ValueError:
|
|
74
|
+
pass
|
|
75
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _summary(data: Any, folder: MessageFolder, account: str) -> ModernMessageSummary:
|
|
79
|
+
if not isinstance(data, dict):
|
|
80
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
81
|
+
correspondent = (
|
|
82
|
+
data.get("senderName")
|
|
83
|
+
if folder is MessageFolder.RECEIVED
|
|
84
|
+
else data.get("receiverName")
|
|
85
|
+
)
|
|
86
|
+
if folder is MessageFolder.SENT and correspondent in (None, ""):
|
|
87
|
+
correspondent = data.get("senderName")
|
|
88
|
+
name = _string(correspondent)
|
|
89
|
+
if not name.strip():
|
|
90
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
91
|
+
subject, stamp = _string(data.get("topic")), _string(data.get("sendDate"))
|
|
92
|
+
has_attachment = data.get("isAnyFileAttached")
|
|
93
|
+
if type(has_attachment) is not bool or (
|
|
94
|
+
folder is MessageFolder.RECEIVED and "readDate" not in data
|
|
95
|
+
):
|
|
96
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
97
|
+
# The observed outbox omits readDate. No recipient receipt is inferred.
|
|
98
|
+
read = data.get("readDate")
|
|
99
|
+
read_at = None if read in (None, "") else _date(read)
|
|
100
|
+
return ModernMessageSummary(
|
|
101
|
+
ModernMessageReference(folder, identifier(data.get("messageId")), account),
|
|
102
|
+
name,
|
|
103
|
+
subject,
|
|
104
|
+
_date(stamp),
|
|
105
|
+
stamp,
|
|
106
|
+
read_at,
|
|
107
|
+
read_at is None if folder is MessageFolder.RECEIVED else None,
|
|
108
|
+
has_attachment,
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def parse_page(
|
|
113
|
+
body: bytes, folder: MessageFolder, page: int, page_size: int, account: str
|
|
114
|
+
) -> tuple[tuple[ModernMessageSummary, ...], int, str]:
|
|
115
|
+
modern_mailbox_query(folder, page, page_size)
|
|
116
|
+
data = decode_json(body)
|
|
117
|
+
if not isinstance(data, dict) or not {"data", "total"} <= set(data):
|
|
118
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
119
|
+
if data.get("archivingInProgress", False) is not False:
|
|
120
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
121
|
+
entries, total = data["data"], data["total"]
|
|
122
|
+
if (
|
|
123
|
+
not isinstance(entries, list)
|
|
124
|
+
or type(total) is not int
|
|
125
|
+
or not 0 <= total <= MAX_TOTAL_MESSAGES
|
|
126
|
+
):
|
|
127
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
128
|
+
if page > max(1, (total + page_size - 1) // page_size):
|
|
129
|
+
raise LibrusError(ErrorKind.STALE_CURSOR)
|
|
130
|
+
expected = min(page_size, max(0, total - (page - 1) * page_size))
|
|
131
|
+
if len(entries) != expected:
|
|
132
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
133
|
+
items = tuple(_summary(item, folder, account) for item in entries)
|
|
134
|
+
if len({item.reference.identifier for item in items}) != len(items):
|
|
135
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
136
|
+
if (
|
|
137
|
+
sum(
|
|
138
|
+
len(item.subject) + len(item.correspondent) + len(item.raw_sent_at)
|
|
139
|
+
for item in items
|
|
140
|
+
)
|
|
141
|
+
> MESSAGE_MAX_TOTAL_TEXT_LENGTH
|
|
142
|
+
):
|
|
143
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
144
|
+
fingerprint = hashlib.sha256(
|
|
145
|
+
repr(
|
|
146
|
+
tuple(
|
|
147
|
+
(
|
|
148
|
+
i.reference.identifier,
|
|
149
|
+
i.correspondent,
|
|
150
|
+
i.subject,
|
|
151
|
+
i.raw_sent_at,
|
|
152
|
+
i.has_attachment,
|
|
153
|
+
)
|
|
154
|
+
for i in items
|
|
155
|
+
)
|
|
156
|
+
).encode()
|
|
157
|
+
).hexdigest()
|
|
158
|
+
return items, total, fingerprint
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
@dataclass(frozen=True, slots=True, repr=False)
|
|
162
|
+
class ParsedContent:
|
|
163
|
+
summary: ModernMessageSummary
|
|
164
|
+
text: str
|
|
165
|
+
attachments: tuple[ModernMessageAttachment, ...]
|
|
166
|
+
receipts: tuple[ModernMessageRecipientReceipt, ...]
|
|
167
|
+
recipient_count: int | None
|
|
168
|
+
read_count: int | None
|
|
169
|
+
receipt_source: Literal["receivers", "individualRecipients"] | None
|
|
170
|
+
archived: bool
|
|
171
|
+
withdrawn: bool
|
|
172
|
+
original_subject: str | None
|
|
173
|
+
original_text: str | None
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _flag(value: Any) -> bool:
|
|
177
|
+
if type(value) not in (bool, int, str) or value not in (
|
|
178
|
+
False,
|
|
179
|
+
True,
|
|
180
|
+
0,
|
|
181
|
+
1,
|
|
182
|
+
"0",
|
|
183
|
+
"1",
|
|
184
|
+
):
|
|
185
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
186
|
+
return value in (True, 1, "1")
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _receipts(
|
|
190
|
+
content: dict[str, Any], folder: MessageFolder
|
|
191
|
+
) -> tuple[
|
|
192
|
+
tuple[ModernMessageRecipientReceipt, ...],
|
|
193
|
+
int | None,
|
|
194
|
+
int | None,
|
|
195
|
+
Literal["receivers", "individualRecipients"] | None,
|
|
196
|
+
]:
|
|
197
|
+
if folder is MessageFolder.RECEIVED:
|
|
198
|
+
return (), None, None, None
|
|
199
|
+
counts = []
|
|
200
|
+
for key in ("receiversCount", "readedCount"):
|
|
201
|
+
value = content.get(key)
|
|
202
|
+
if value is not None and (
|
|
203
|
+
type(value) is not int or not 0 <= value <= MAX_TOTAL_MESSAGES
|
|
204
|
+
):
|
|
205
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
206
|
+
counts.append(value)
|
|
207
|
+
total, read_count = counts
|
|
208
|
+
if total is not None and read_count is not None and read_count > total:
|
|
209
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
210
|
+
source: Literal["receivers", "individualRecipients"] | None = None
|
|
211
|
+
if (
|
|
212
|
+
"individualRecipients" in content
|
|
213
|
+
and content["individualRecipients"] is not None
|
|
214
|
+
):
|
|
215
|
+
source = "individualRecipients"
|
|
216
|
+
elif "receivers" in content:
|
|
217
|
+
source = "receivers"
|
|
218
|
+
if source is None:
|
|
219
|
+
return (), total, read_count, None
|
|
220
|
+
rows = content[source]
|
|
221
|
+
if not isinstance(rows, list):
|
|
222
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
223
|
+
if len(rows) > MESSAGE_MAX_BATCH_ITEMS:
|
|
224
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
225
|
+
result = []
|
|
226
|
+
seen = set()
|
|
227
|
+
for row in rows:
|
|
228
|
+
if not isinstance(row, dict):
|
|
229
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
230
|
+
receiver_id = identifier(_string(row.get("receiverId")))
|
|
231
|
+
if receiver_id in seen:
|
|
232
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
233
|
+
seen.add(receiver_id)
|
|
234
|
+
cc, bcc = row.get("isCc"), row.get("isBcc")
|
|
235
|
+
if (
|
|
236
|
+
type(cc) is not str
|
|
237
|
+
or type(bcc) is not str
|
|
238
|
+
or cc not in ("0", "1")
|
|
239
|
+
or bcc not in ("0", "1")
|
|
240
|
+
or cc == bcc == "1"
|
|
241
|
+
):
|
|
242
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
243
|
+
stamp = row.get("readed")
|
|
244
|
+
read_at = None if stamp in (None, "") else _date(stamp)
|
|
245
|
+
name = _string(row.get("name"))
|
|
246
|
+
if not name.strip():
|
|
247
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
248
|
+
result.append(
|
|
249
|
+
ModernMessageRecipientReceipt(
|
|
250
|
+
receiver_id,
|
|
251
|
+
name,
|
|
252
|
+
"bcc" if bcc == "1" else "cc" if cc == "1" else "to",
|
|
253
|
+
read_at is not None if "readed" in row else None,
|
|
254
|
+
read_at,
|
|
255
|
+
)
|
|
256
|
+
)
|
|
257
|
+
if (
|
|
258
|
+
sum(len(r.name) + len(r.recipient_id) for r in result)
|
|
259
|
+
> MESSAGE_MAX_TOTAL_TEXT_LENGTH
|
|
260
|
+
):
|
|
261
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
262
|
+
return tuple(result), total, read_count, source
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def parse_content(body: bytes, reference: ModernMessageReference) -> ParsedContent:
|
|
266
|
+
data = decode_json(body)
|
|
267
|
+
if not isinstance(data, dict) or not isinstance(data.get("data"), dict):
|
|
268
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
269
|
+
content = data["data"]
|
|
270
|
+
archived = _flag(content.get("archive", False))
|
|
271
|
+
withdrawn = _flag(content.get("isMessageWithdrawn", False))
|
|
272
|
+
entries = content.get("attachments")
|
|
273
|
+
if not isinstance(entries, list) or len(entries) > MESSAGE_MAX_ATTACHMENTS:
|
|
274
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
275
|
+
# Detail responses do not carry the mailbox's isAnyFileAttached field.
|
|
276
|
+
has_attachment = content.get("isAnyFileAttached", bool(entries))
|
|
277
|
+
if type(has_attachment) is not bool or has_attachment != bool(entries):
|
|
278
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
279
|
+
summary = _summary(
|
|
280
|
+
content | {"isAnyFileAttached": has_attachment},
|
|
281
|
+
reference.folder,
|
|
282
|
+
reference.account,
|
|
283
|
+
)
|
|
284
|
+
if summary.reference != reference:
|
|
285
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
286
|
+
rendered = render_body(content.get("Message"))
|
|
287
|
+
attachments = []
|
|
288
|
+
seen = set()
|
|
289
|
+
for raw in entries:
|
|
290
|
+
if not isinstance(raw, dict):
|
|
291
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
292
|
+
file_id, name = identifier(raw.get("id")), _string(raw.get("filename"))
|
|
293
|
+
if not name.strip() or file_id in seen:
|
|
294
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
295
|
+
seen.add(file_id)
|
|
296
|
+
attachments.append(
|
|
297
|
+
ModernMessageAttachment(
|
|
298
|
+
ModernMessageAttachmentReference(reference, file_id, archived), name
|
|
299
|
+
)
|
|
300
|
+
)
|
|
301
|
+
receipts, total, read_count, source = _receipts(content, reference.folder)
|
|
302
|
+
original = content.get("originalMessage")
|
|
303
|
+
original_text = None if original in (None, "") else render_body(original)
|
|
304
|
+
original_subject = (
|
|
305
|
+
None if original_text is None else _string(content.get("originalTopic"))
|
|
306
|
+
)
|
|
307
|
+
return ParsedContent(
|
|
308
|
+
summary,
|
|
309
|
+
rendered,
|
|
310
|
+
tuple(attachments),
|
|
311
|
+
receipts,
|
|
312
|
+
total,
|
|
313
|
+
read_count,
|
|
314
|
+
source,
|
|
315
|
+
archived,
|
|
316
|
+
withdrawn,
|
|
317
|
+
original_subject,
|
|
318
|
+
original_text,
|
|
319
|
+
)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def validate_selection(
|
|
323
|
+
folder: MessageFolder,
|
|
324
|
+
cursor: ModernMessagesCursor | None,
|
|
325
|
+
page_size: int,
|
|
326
|
+
max_pages: int,
|
|
327
|
+
limit: int,
|
|
328
|
+
account: str,
|
|
329
|
+
) -> None:
|
|
330
|
+
modern_mailbox_query(folder, 1, page_size)
|
|
331
|
+
if type(max_pages) is not int or not 1 <= max_pages <= MESSAGE_MAX_BATCH_PAGES:
|
|
332
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
333
|
+
if type(limit) is not int or not 1 <= limit <= MESSAGE_MAX_BATCH_ITEMS:
|
|
334
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
335
|
+
if cursor is None:
|
|
336
|
+
return
|
|
337
|
+
if not isinstance(cursor, ModernMessagesCursor):
|
|
338
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
339
|
+
modern_mailbox_query(folder, cursor.page, cursor.page_size)
|
|
340
|
+
if (
|
|
341
|
+
cursor.account != account
|
|
342
|
+
or cursor.folder is not folder
|
|
343
|
+
or cursor.page_size != page_size
|
|
344
|
+
or type(cursor.offset) is not int
|
|
345
|
+
or not 0 <= cursor.offset < page_size
|
|
346
|
+
or type(cursor.total_count) is not int
|
|
347
|
+
or not 1 <= cursor.total_count <= MAX_TOTAL_MESSAGES
|
|
348
|
+
or type(cursor.fingerprint) is not str
|
|
349
|
+
or re.fullmatch(r"[0-9a-f]{64}", cursor.fingerprint) is None
|
|
350
|
+
or type(cursor.seen_ids) is not tuple
|
|
351
|
+
or not 1 <= len(cursor.seen_ids) <= MESSAGE_MAX_CURSOR_IDS
|
|
352
|
+
or any(
|
|
353
|
+
type(i) is not str or re.fullmatch(r"[0-9]{1,64}", i) is None
|
|
354
|
+
for i in cursor.seen_ids
|
|
355
|
+
)
|
|
356
|
+
or len(set(cursor.seen_ids)) != len(cursor.seen_ids)
|
|
357
|
+
or cursor.page > (cursor.total_count + page_size - 1) // page_size
|
|
358
|
+
):
|
|
359
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
async def collect(
|
|
363
|
+
fetch: Callable[[int], Awaitable[ModernMessagesPage]],
|
|
364
|
+
folder: MessageFolder,
|
|
365
|
+
account: str,
|
|
366
|
+
cursor: ModernMessagesCursor | None,
|
|
367
|
+
page_size: int,
|
|
368
|
+
max_pages: int,
|
|
369
|
+
limit: int,
|
|
370
|
+
) -> tuple[
|
|
371
|
+
tuple[ModernMessageSummary, ...],
|
|
372
|
+
int,
|
|
373
|
+
int,
|
|
374
|
+
ModernMessagesCursor | None,
|
|
375
|
+
Literal["item_limit", "page_limit"] | None,
|
|
376
|
+
]:
|
|
377
|
+
validate_selection(folder, cursor, page_size, max_pages, limit, account)
|
|
378
|
+
page = cursor.page if cursor else 1
|
|
379
|
+
offset = cursor.offset if cursor else 0
|
|
380
|
+
seen = list(cursor.seen_ids) if cursor else []
|
|
381
|
+
seen_set = set(seen)
|
|
382
|
+
items = []
|
|
383
|
+
total = cursor.total_count if cursor else None
|
|
384
|
+
pages = duplicates = 0
|
|
385
|
+
for _ in range(max_pages):
|
|
386
|
+
result = await fetch(page)
|
|
387
|
+
pages += 1
|
|
388
|
+
if (total is not None and result.total_count != total) or (
|
|
389
|
+
cursor
|
|
390
|
+
and cursor.offset
|
|
391
|
+
and pages == 1
|
|
392
|
+
and result.fingerprint != cursor.fingerprint
|
|
393
|
+
):
|
|
394
|
+
raise LibrusError(ErrorKind.STALE_CURSOR)
|
|
395
|
+
total = result.total_count
|
|
396
|
+
if offset and offset >= len(result.items):
|
|
397
|
+
raise LibrusError(ErrorKind.STALE_CURSOR)
|
|
398
|
+
if (
|
|
399
|
+
result.items
|
|
400
|
+
and offset == 0
|
|
401
|
+
and all(item.reference.identifier in seen_set for item in result.items)
|
|
402
|
+
):
|
|
403
|
+
raise LibrusError(ErrorKind.STALE_CURSOR)
|
|
404
|
+
new = 0
|
|
405
|
+
for index in range(offset, len(result.items)):
|
|
406
|
+
item = result.items[index]
|
|
407
|
+
if item.reference.identifier in seen_set:
|
|
408
|
+
duplicates += 1
|
|
409
|
+
continue
|
|
410
|
+
if len(seen) >= MESSAGE_MAX_CURSOR_IDS:
|
|
411
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
412
|
+
items.append(item)
|
|
413
|
+
new += 1
|
|
414
|
+
seen.append(item.reference.identifier)
|
|
415
|
+
seen_set.add(item.reference.identifier)
|
|
416
|
+
if len(items) == limit and index + 1 < len(result.items):
|
|
417
|
+
return (
|
|
418
|
+
tuple(items),
|
|
419
|
+
pages,
|
|
420
|
+
duplicates,
|
|
421
|
+
ModernMessagesCursor(
|
|
422
|
+
account,
|
|
423
|
+
folder,
|
|
424
|
+
page,
|
|
425
|
+
index + 1,
|
|
426
|
+
page_size,
|
|
427
|
+
total,
|
|
428
|
+
result.fingerprint,
|
|
429
|
+
tuple(seen),
|
|
430
|
+
),
|
|
431
|
+
"item_limit",
|
|
432
|
+
)
|
|
433
|
+
if result.items and not new and pages > 1:
|
|
434
|
+
raise LibrusError(ErrorKind.STALE_CURSOR)
|
|
435
|
+
if page * page_size >= total:
|
|
436
|
+
return tuple(items), pages, duplicates, None, None
|
|
437
|
+
if page >= 1000:
|
|
438
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
439
|
+
if len(items) >= limit or pages == max_pages:
|
|
440
|
+
# At a page boundary, preserve the previous fingerprint for repeated
|
|
441
|
+
# page rejection; only a mid-page cursor claims its current fingerprint.
|
|
442
|
+
return (
|
|
443
|
+
tuple(items),
|
|
444
|
+
pages,
|
|
445
|
+
duplicates,
|
|
446
|
+
ModernMessagesCursor(
|
|
447
|
+
account,
|
|
448
|
+
folder,
|
|
449
|
+
page + 1,
|
|
450
|
+
0,
|
|
451
|
+
page_size,
|
|
452
|
+
total,
|
|
453
|
+
result.fingerprint,
|
|
454
|
+
tuple(seen),
|
|
455
|
+
),
|
|
456
|
+
"item_limit" if len(items) >= limit else "page_limit",
|
|
457
|
+
)
|
|
458
|
+
page, offset = page + 1, 0
|
|
459
|
+
raise AssertionError("Bounded collector must terminate")
|