librus-python-api 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- librus_python_api/__init__.py +243 -0
- librus_python_api/_notification_bootstrap.py +124 -0
- librus_python_api/_notification_codec.py +396 -0
- librus_python_api/_storage.py +403 -0
- librus_python_api/_windows_filesystem.py +390 -0
- librus_python_api/announcements.py +159 -0
- librus_python_api/attachment_routes.py +114 -0
- librus_python_api/attachments.py +297 -0
- librus_python_api/attendance.py +182 -0
- librus_python_api/attendance_frequency.py +112 -0
- librus_python_api/budget.py +79 -0
- librus_python_api/checkpoint.py +61 -0
- librus_python_api/completed_lessons.py +216 -0
- librus_python_api/config.py +1410 -0
- librus_python_api/detail_fields.py +50 -0
- librus_python_api/diagnostics.py +25 -0
- librus_python_api/exceptions.py +172 -0
- librus_python_api/files.py +242 -0
- librus_python_api/grade_parsers.py +169 -0
- librus_python_api/grade_records.py +454 -0
- librus_python_api/homework_range.py +41 -0
- librus_python_api/lifecycle.py +24 -0
- librus_python_api/markup.py +147 -0
- librus_python_api/message_content.py +230 -0
- librus_python_api/messages.py +288 -0
- librus_python_api/models.py +1160 -0
- librus_python_api/modern_body.py +75 -0
- librus_python_api/modern_mailbox.py +459 -0
- librus_python_api/modern_messages.py +276 -0
- librus_python_api/notification_models.py +145 -0
- librus_python_api/notification_persistence.py +1243 -0
- librus_python_api/notification_workflow.py +337 -0
- librus_python_api/notifications.py +216 -0
- librus_python_api/parsers.py +232 -0
- librus_python_api/parsing.py +49 -0
- librus_python_api/persistence.py +419 -0
- librus_python_api/py.typed +0 -0
- librus_python_api/recipients.py +271 -0
- librus_python_api/scheduler.py +287 -0
- librus_python_api/school_reads.py +400 -0
- librus_python_api/sending.py +125 -0
- librus_python_api/service.py +2285 -0
- librus_python_api/timetable.py +261 -0
- librus_python_api/transport.py +956 -0
- librus_python_api-1.0.0.dist-info/METADATA +262 -0
- librus_python_api-1.0.0.dist-info/RECORD +48 -0
- librus_python_api-1.0.0.dist-info/WHEEL +4 -0
- librus_python_api-1.0.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,230 @@
|
|
|
1
|
+
"""Pure bounded message content and inert attachment metadata extraction."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
from lxml import html
|
|
6
|
+
|
|
7
|
+
from librus_python_api.config import (
|
|
8
|
+
MESSAGE_ATTACHMENT_PATH_PREFIX,
|
|
9
|
+
MESSAGE_INFORMATION_NOTICES,
|
|
10
|
+
MESSAGE_MAX_ATTACHMENTS,
|
|
11
|
+
MESSAGE_MAX_CONTENT_LENGTH,
|
|
12
|
+
MESSAGE_MAX_FIELD_LENGTH,
|
|
13
|
+
MESSAGE_MAX_RECEIPT_TEXT_LENGTH,
|
|
14
|
+
MESSAGE_MAX_RECIPIENT_RECEIPTS,
|
|
15
|
+
)
|
|
16
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
17
|
+
from librus_python_api.markup import cells, rows, text
|
|
18
|
+
from librus_python_api.messages import _timestamp
|
|
19
|
+
from librus_python_api.models import (
|
|
20
|
+
MessageAttachment,
|
|
21
|
+
MessageAttachmentReference,
|
|
22
|
+
MessageContentData,
|
|
23
|
+
MessageFolder,
|
|
24
|
+
MessageRecipientReceipt,
|
|
25
|
+
MessageReference,
|
|
26
|
+
)
|
|
27
|
+
from librus_python_api.parsers import page_notices, parse_page
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def validate_reference(reference: MessageReference, account: str) -> None:
|
|
31
|
+
if (
|
|
32
|
+
not isinstance(reference, MessageReference)
|
|
33
|
+
or not isinstance(reference.folder, MessageFolder)
|
|
34
|
+
or reference.account != account
|
|
35
|
+
or type(reference.identifier) is not str
|
|
36
|
+
or re.fullmatch(r"[0-9]{1,64}", reference.identifier) is None
|
|
37
|
+
):
|
|
38
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _recipient_receipts(
|
|
42
|
+
table_rows: list[html.HtmlElement],
|
|
43
|
+
) -> tuple[MessageRecipientReceipt, ...]:
|
|
44
|
+
if not 2 <= len(table_rows) <= MESSAGE_MAX_RECIPIENT_RECEIPTS + 1:
|
|
45
|
+
raise LibrusError(
|
|
46
|
+
ErrorKind.LIMIT
|
|
47
|
+
if len(table_rows) > MESSAGE_MAX_RECIPIENT_RECEIPTS + 1
|
|
48
|
+
else ErrorKind.PARSE
|
|
49
|
+
)
|
|
50
|
+
heading = cells(table_rows[0])
|
|
51
|
+
if (
|
|
52
|
+
len(heading) != 1
|
|
53
|
+
or text(heading[0]).rstrip(":") != "Przeczytano"
|
|
54
|
+
or heading[0].get("colspan") != "3"
|
|
55
|
+
or heading[0].get("rowspan", "1") != "1"
|
|
56
|
+
):
|
|
57
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
58
|
+
result = []
|
|
59
|
+
total = 0
|
|
60
|
+
for row in table_rows[1:]:
|
|
61
|
+
columns = cells(row)
|
|
62
|
+
if len(columns) != 2 or any(
|
|
63
|
+
c.tag != "td"
|
|
64
|
+
or c.get("colspan", "1") != "1"
|
|
65
|
+
or c.get("rowspan", "1") != "1"
|
|
66
|
+
or next(c.iterdescendants("table"), None) is not None
|
|
67
|
+
for c in columns
|
|
68
|
+
):
|
|
69
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
70
|
+
recipient, status = [
|
|
71
|
+
text(c, MESSAGE_MAX_FIELD_LENGTH, multiline=True) for c in columns
|
|
72
|
+
]
|
|
73
|
+
if not recipient or not status:
|
|
74
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
75
|
+
total += len(recipient) + len(status)
|
|
76
|
+
if total > MESSAGE_MAX_RECEIPT_TEXT_LENGTH:
|
|
77
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
78
|
+
stamp = None if status == "NIE" else _timestamp(status)
|
|
79
|
+
result.append(MessageRecipientReceipt(recipient, status, stamp))
|
|
80
|
+
return tuple(result)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _metadata(
|
|
84
|
+
document: html.HtmlElement, folder: MessageFolder
|
|
85
|
+
) -> tuple[list[str | None], str | None, tuple[MessageRecipientReceipt, ...]]:
|
|
86
|
+
candidates = [
|
|
87
|
+
t for t in document.iter("table") if t.get("class", "").split() == ["stretch"]
|
|
88
|
+
]
|
|
89
|
+
main = []
|
|
90
|
+
receipt: str | None = None
|
|
91
|
+
individual: tuple[MessageRecipientReceipt, ...] = ()
|
|
92
|
+
for table in candidates:
|
|
93
|
+
table_rows = list(rows(table))
|
|
94
|
+
first = cells(table_rows[0]) if table_rows else []
|
|
95
|
+
if len(first) == 1 and text(first[0]).rstrip(":") == "Przeczytano":
|
|
96
|
+
if folder is not MessageFolder.SENT or individual or receipt is not None:
|
|
97
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
98
|
+
individual = _recipient_receipts(table_rows)
|
|
99
|
+
continue
|
|
100
|
+
if len(table_rows) == 1:
|
|
101
|
+
columns = cells(table_rows[0])
|
|
102
|
+
if receipt is not None or individual:
|
|
103
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
104
|
+
if (
|
|
105
|
+
len(columns) != 2
|
|
106
|
+
or text(columns[0]).rstrip(":") != "Przeczytano"
|
|
107
|
+
or any(
|
|
108
|
+
c.get("colspan", "1") != "1" or c.get("rowspan", "1") != "1"
|
|
109
|
+
for c in columns
|
|
110
|
+
)
|
|
111
|
+
):
|
|
112
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
113
|
+
receipt = text(columns[1], MESSAGE_MAX_FIELD_LENGTH)
|
|
114
|
+
else:
|
|
115
|
+
main.append(table_rows)
|
|
116
|
+
if len(main) != 1:
|
|
117
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
118
|
+
return _main_metadata(main[0], folder), receipt, individual
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _main_metadata(
|
|
122
|
+
table_rows: list[html.HtmlElement], folder: MessageFolder
|
|
123
|
+
) -> list[str | None]:
|
|
124
|
+
short_sent = folder is MessageFolder.SENT and len(table_rows) == 2
|
|
125
|
+
if len(table_rows) != 3 and not short_sent:
|
|
126
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
127
|
+
expected: tuple[str, ...] = (
|
|
128
|
+
"Nadawca" if folder is MessageFolder.RECEIVED else "Adresat",
|
|
129
|
+
"Temat",
|
|
130
|
+
"Wysłano",
|
|
131
|
+
)
|
|
132
|
+
if short_sent:
|
|
133
|
+
expected = expected[1:]
|
|
134
|
+
result: list[str | None] = [None] if short_sent else []
|
|
135
|
+
for row, label in zip(table_rows, expected, strict=True):
|
|
136
|
+
columns = cells(row)
|
|
137
|
+
if (
|
|
138
|
+
len(columns) != 2
|
|
139
|
+
or text(columns[0]).rstrip(":") != label
|
|
140
|
+
or any(
|
|
141
|
+
c.get("colspan", "1") != "1" or c.get("rowspan", "1") != "1"
|
|
142
|
+
for c in columns
|
|
143
|
+
)
|
|
144
|
+
or any(next(c.iterdescendants("table"), None) is not None for c in columns)
|
|
145
|
+
):
|
|
146
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
147
|
+
result.append(text(columns[1], MESSAGE_MAX_FIELD_LENGTH, multiline=True))
|
|
148
|
+
if not short_sent and not result[0]:
|
|
149
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
150
|
+
return result
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _attachments(
|
|
154
|
+
document: html.HtmlElement, reference: MessageReference
|
|
155
|
+
) -> tuple[MessageAttachment, ...]:
|
|
156
|
+
result: list[MessageAttachment] = []
|
|
157
|
+
seen: set[str] = set()
|
|
158
|
+
for element in document.iter():
|
|
159
|
+
if not isinstance(element.tag, str):
|
|
160
|
+
continue
|
|
161
|
+
if "pobierz_zalacznik" in element.get("href", ""):
|
|
162
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
163
|
+
# Markers live in download-icon handlers. They are parsed as data only,
|
|
164
|
+
# never evaluated, requested or exposed as arbitrary authenticated URLs.
|
|
165
|
+
handler = element.get("onclick", "")
|
|
166
|
+
if "pobierz_zalacznik" not in handler:
|
|
167
|
+
continue
|
|
168
|
+
if len(handler) > MESSAGE_MAX_FIELD_LENGTH:
|
|
169
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
170
|
+
literals = re.findall(r"(['\"])([^'\"]+)\1", handler.replace(r"\/", "/"))
|
|
171
|
+
targets = [
|
|
172
|
+
re.fullmatch(
|
|
173
|
+
re.escape(MESSAGE_ATTACHMENT_PATH_PREFIX)
|
|
174
|
+
+ r"([0-9]{1,64})/([0-9]{1,64})",
|
|
175
|
+
value,
|
|
176
|
+
)
|
|
177
|
+
for _, value in literals
|
|
178
|
+
]
|
|
179
|
+
matches = [m for m in targets if m is not None]
|
|
180
|
+
if (
|
|
181
|
+
len(matches) != 1
|
|
182
|
+
or handler.count("pobierz_zalacznik") != 1
|
|
183
|
+
or element.tag != "img"
|
|
184
|
+
):
|
|
185
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
186
|
+
message_id, file_id = matches[0].groups()
|
|
187
|
+
if message_id != reference.identifier or file_id in seen:
|
|
188
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
189
|
+
row = next(element.iterancestors("tr"), None)
|
|
190
|
+
columns = cells(row) if row is not None else []
|
|
191
|
+
if not columns:
|
|
192
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
193
|
+
name = text(columns[0], MESSAGE_MAX_FIELD_LENGTH)
|
|
194
|
+
if not name:
|
|
195
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
196
|
+
if len(result) >= MESSAGE_MAX_ATTACHMENTS:
|
|
197
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
198
|
+
seen.add(file_id)
|
|
199
|
+
result.append(
|
|
200
|
+
MessageAttachment(MessageAttachmentReference(reference, file_id), name)
|
|
201
|
+
)
|
|
202
|
+
return tuple(result)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def parse_message_content(
|
|
206
|
+
body: bytes, reference: MessageReference
|
|
207
|
+
) -> MessageContentData:
|
|
208
|
+
validate_reference(reference, reference.account)
|
|
209
|
+
document = parse_page(body)
|
|
210
|
+
if any(n not in MESSAGE_INFORMATION_NOTICES for n in page_notices(document)):
|
|
211
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
212
|
+
metadata, receipt, individual = _metadata(document, reference.folder)
|
|
213
|
+
correspondent, subject, stamp = metadata
|
|
214
|
+
assert subject is not None and stamp is not None
|
|
215
|
+
containers = document.xpath(
|
|
216
|
+
'//*[contains(concat(" ",normalize-space(@class)," "),'
|
|
217
|
+
'" container-message-content ")]'
|
|
218
|
+
)
|
|
219
|
+
if len(containers) != 1:
|
|
220
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
221
|
+
return MessageContentData(
|
|
222
|
+
reference,
|
|
223
|
+
correspondent,
|
|
224
|
+
subject,
|
|
225
|
+
_timestamp(stamp),
|
|
226
|
+
_timestamp(receipt) if receipt is not None else None,
|
|
227
|
+
text(containers[0], MESSAGE_MAX_CONTENT_LENGTH, multiline=True),
|
|
228
|
+
_attachments(document, reference),
|
|
229
|
+
individual,
|
|
230
|
+
)
|
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
"""Bounded ordinary mailbox summaries; no body opens, mark-read, sends or deletes."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import astuple
|
|
6
|
+
from datetime import datetime
|
|
7
|
+
|
|
8
|
+
from lxml import html
|
|
9
|
+
|
|
10
|
+
from librus_python_api.config import (
|
|
11
|
+
MESSAGE_EMPTY_TEXT,
|
|
12
|
+
MESSAGE_HEADER_LABELS,
|
|
13
|
+
MESSAGE_INFORMATION_NOTICES,
|
|
14
|
+
MESSAGE_MAX_BATCH_ITEMS,
|
|
15
|
+
MESSAGE_MAX_BATCH_PAGES,
|
|
16
|
+
MESSAGE_MAX_CURSOR_IDS,
|
|
17
|
+
MESSAGE_MAX_FIELD_LENGTH,
|
|
18
|
+
MESSAGE_MAX_PAGE_COUNT,
|
|
19
|
+
MESSAGE_MAX_PAGE_ITEMS,
|
|
20
|
+
MESSAGE_MAX_TOTAL_TEXT_LENGTH,
|
|
21
|
+
MESSAGE_REFERENCE_PREFIXES,
|
|
22
|
+
message_page_form,
|
|
23
|
+
)
|
|
24
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
25
|
+
from librus_python_api.markup import cells, in_header, rows, text
|
|
26
|
+
from librus_python_api.models import (
|
|
27
|
+
MessageFolder,
|
|
28
|
+
MessageReference,
|
|
29
|
+
MessagesCursor,
|
|
30
|
+
MessageSummary,
|
|
31
|
+
MessageTimestamp,
|
|
32
|
+
)
|
|
33
|
+
from librus_python_api.parsers import page_notices, parse_page
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def validate_selection(
|
|
37
|
+
folder: MessageFolder,
|
|
38
|
+
cursor: MessagesCursor | None,
|
|
39
|
+
max_pages: int,
|
|
40
|
+
limit: int,
|
|
41
|
+
account: str,
|
|
42
|
+
) -> None:
|
|
43
|
+
message_page_form(folder, 0)
|
|
44
|
+
if type(max_pages) is not int or not 1 <= max_pages <= MESSAGE_MAX_BATCH_PAGES:
|
|
45
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
46
|
+
if type(limit) is not int or not 1 <= limit <= MESSAGE_MAX_BATCH_ITEMS:
|
|
47
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
48
|
+
if cursor is None:
|
|
49
|
+
return
|
|
50
|
+
if not isinstance(cursor, MessagesCursor):
|
|
51
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
52
|
+
if (
|
|
53
|
+
cursor.account != account
|
|
54
|
+
or cursor.folder is not folder
|
|
55
|
+
or type(cursor.page_count) is not int
|
|
56
|
+
or not 1 <= cursor.page_count <= MESSAGE_MAX_PAGE_COUNT
|
|
57
|
+
or type(cursor.page) is not int
|
|
58
|
+
or not 0 <= cursor.page < cursor.page_count
|
|
59
|
+
or type(cursor.offset) is not int
|
|
60
|
+
or not 0 <= cursor.offset < MESSAGE_MAX_PAGE_ITEMS
|
|
61
|
+
or (cursor.page == 0 and cursor.offset == 0)
|
|
62
|
+
or type(cursor.fingerprint) is not str
|
|
63
|
+
or re.fullmatch(r"[0-9a-f]{64}", cursor.fingerprint) is None
|
|
64
|
+
or type(cursor.seen_ids) is not tuple
|
|
65
|
+
or not 1 <= len(cursor.seen_ids) <= MESSAGE_MAX_CURSOR_IDS
|
|
66
|
+
or any(
|
|
67
|
+
type(i) is not str or re.fullmatch(r"[0-9]{1,64}", i) is None
|
|
68
|
+
for i in cursor.seen_ids
|
|
69
|
+
)
|
|
70
|
+
or len(set(cursor.seen_ids)) != len(cursor.seen_ids)
|
|
71
|
+
):
|
|
72
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _pagination(document: html.HtmlElement, page: int) -> int:
|
|
76
|
+
markers = document.xpath(
|
|
77
|
+
'//div[contains(concat(" ",normalize-space(@class)," ")," pagination ")]'
|
|
78
|
+
)
|
|
79
|
+
if not markers:
|
|
80
|
+
if page:
|
|
81
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
82
|
+
return 1
|
|
83
|
+
if len(markers) != 1:
|
|
84
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
85
|
+
spans = [s for s in markers[0] if s.tag == "span"]
|
|
86
|
+
if len(spans) != 1:
|
|
87
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
88
|
+
label = text(spans[0])
|
|
89
|
+
match = re.fullmatch(
|
|
90
|
+
r"(?:Strona\s*:?\s*)?([0-9]{1,4})\s+z\s+([0-9]{1,4})", label, re.I
|
|
91
|
+
)
|
|
92
|
+
if match is None:
|
|
93
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
94
|
+
current, count = int(match[1]), int(match[2])
|
|
95
|
+
if count > MESSAGE_MAX_PAGE_COUNT:
|
|
96
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
97
|
+
if not 1 <= current <= count or current != page + 1:
|
|
98
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
99
|
+
return count
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _header(table: html.HtmlElement, folder: MessageFolder) -> int:
|
|
103
|
+
width = 6 if folder is MessageFolder.RECEIVED else 7
|
|
104
|
+
headers = [r for r in rows(table) if in_header(r)]
|
|
105
|
+
if len(headers) != 1:
|
|
106
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
107
|
+
columns = cells(headers[0])
|
|
108
|
+
if len(columns) != width or any(
|
|
109
|
+
c.get("colspan", "1") != "1" or c.get("rowspan", "1") != "1" for c in columns
|
|
110
|
+
):
|
|
111
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
112
|
+
labels = [text(c) for c in columns]
|
|
113
|
+
if (
|
|
114
|
+
labels[2] != MESSAGE_HEADER_LABELS[folder]
|
|
115
|
+
or labels[3] != "Temat"
|
|
116
|
+
or re.fullmatch(r"Wysłano(?: \[[↑↓]\])?", labels[4]) is None
|
|
117
|
+
):
|
|
118
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
119
|
+
if folder is MessageFolder.SENT and labels[5] != "Przeczytano":
|
|
120
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
121
|
+
return width
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _reference(
|
|
125
|
+
correspondent: html.HtmlElement,
|
|
126
|
+
subject: html.HtmlElement,
|
|
127
|
+
folder: MessageFolder,
|
|
128
|
+
account: str,
|
|
129
|
+
) -> MessageReference:
|
|
130
|
+
identifiers = []
|
|
131
|
+
for cell in (correspondent, subject):
|
|
132
|
+
anchors = list(cell.iter("a"))
|
|
133
|
+
if len(anchors) != 1:
|
|
134
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
135
|
+
match = re.fullmatch(
|
|
136
|
+
re.escape(MESSAGE_REFERENCE_PREFIXES[folder]) + r"([0-9]{1,64})(?:/f0)?",
|
|
137
|
+
anchors[0].get("href", ""),
|
|
138
|
+
)
|
|
139
|
+
if match is None:
|
|
140
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
141
|
+
identifiers.append(match[1])
|
|
142
|
+
if identifiers[0] != identifiers[1]:
|
|
143
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
144
|
+
return MessageReference(folder, identifiers[0], account)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _timestamp(value: str) -> MessageTimestamp:
|
|
148
|
+
if (
|
|
149
|
+
re.fullmatch(
|
|
150
|
+
r"[0-9]{4}-[0-9]{2}-[0-9]{2} [0-9]{2}:[0-9]{2}(?::[0-9]{2})?", value
|
|
151
|
+
)
|
|
152
|
+
is None
|
|
153
|
+
):
|
|
154
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
155
|
+
try:
|
|
156
|
+
local = datetime.fromisoformat(value)
|
|
157
|
+
except ValueError:
|
|
158
|
+
local = None
|
|
159
|
+
if local is None:
|
|
160
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
161
|
+
# The page carries no UTC offset/fold. Retain school civil time, including
|
|
162
|
+
# ambiguous DST wall times, rather than inventing a specific UTC instant.
|
|
163
|
+
return MessageTimestamp(local, value)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _unread(subject: html.HtmlElement) -> bool:
|
|
167
|
+
weights = []
|
|
168
|
+
for declaration in subject.get("style", "").split(";"):
|
|
169
|
+
key, separator, value = declaration.partition(":")
|
|
170
|
+
if separator and key.strip().casefold() == "font-weight":
|
|
171
|
+
weights.append(value.strip().casefold().removesuffix("!important").strip())
|
|
172
|
+
if not weights:
|
|
173
|
+
return False
|
|
174
|
+
if len(weights) != 1 or weights[0] not in {"bold", "normal", "400", "700"}:
|
|
175
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
176
|
+
return weights[0] in {"bold", "700"}
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _summary(
|
|
180
|
+
row: html.HtmlElement, width: int, folder: MessageFolder, account: str
|
|
181
|
+
) -> MessageSummary:
|
|
182
|
+
columns = cells(row)
|
|
183
|
+
if len(columns) != width or any(
|
|
184
|
+
c.tag != "td" or c.get("rowspan", "1") != "1" or c.get("colspan", "1") != "1"
|
|
185
|
+
for c in columns
|
|
186
|
+
):
|
|
187
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
188
|
+
correspondent, subject, stamp = [
|
|
189
|
+
text(c, MESSAGE_MAX_FIELD_LENGTH, multiline=True) for c in columns[2:5]
|
|
190
|
+
]
|
|
191
|
+
if not correspondent:
|
|
192
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
193
|
+
return MessageSummary(
|
|
194
|
+
_reference(columns[2], columns[3], folder, account),
|
|
195
|
+
correspondent,
|
|
196
|
+
subject,
|
|
197
|
+
_timestamp(stamp),
|
|
198
|
+
_unread(columns[3]) if folder is MessageFolder.RECEIVED else None,
|
|
199
|
+
next(columns[1].iter("img"), None) is not None,
|
|
200
|
+
text(columns[5], MESSAGE_MAX_FIELD_LENGTH, multiline=True)
|
|
201
|
+
if folder is MessageFolder.SENT
|
|
202
|
+
else None,
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def parse_messages(
|
|
207
|
+
body: bytes, folder: MessageFolder, page: int, account: str
|
|
208
|
+
) -> tuple[tuple[MessageSummary, ...], int, str]:
|
|
209
|
+
message_page_form(folder, page)
|
|
210
|
+
document = parse_page(body)
|
|
211
|
+
if any(
|
|
212
|
+
notice not in MESSAGE_INFORMATION_NOTICES for notice in page_notices(document)
|
|
213
|
+
):
|
|
214
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
215
|
+
count = _pagination(document, page)
|
|
216
|
+
tables = [
|
|
217
|
+
t
|
|
218
|
+
for t in document.iter("table")
|
|
219
|
+
if {"decorated", "stretch"} <= set(t.get("class", "").split())
|
|
220
|
+
]
|
|
221
|
+
if len(tables) != 1:
|
|
222
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
223
|
+
table = tables[0]
|
|
224
|
+
if list(table.iterdescendants("table")):
|
|
225
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
226
|
+
width = _header(table, folder)
|
|
227
|
+
# The observed legacy table has a blank tfoot. Only tbody contains messages.
|
|
228
|
+
body_rows = [
|
|
229
|
+
r for r in rows(table) if (p := r.getparent()) is not None and p.tag == "tbody"
|
|
230
|
+
]
|
|
231
|
+
if not body_rows:
|
|
232
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
233
|
+
first = cells(body_rows[0])
|
|
234
|
+
empty = len(first) == 1 and text(first[0]) == MESSAGE_EMPTY_TEXT
|
|
235
|
+
items: list[MessageSummary] = []
|
|
236
|
+
identifiers: set[str] = set()
|
|
237
|
+
total = 0
|
|
238
|
+
if empty:
|
|
239
|
+
if (
|
|
240
|
+
len(body_rows) != 1
|
|
241
|
+
or first[0].get("colspan") != str(width)
|
|
242
|
+
or first[0].get("rowspan", "1") != "1"
|
|
243
|
+
or count != 1
|
|
244
|
+
or page
|
|
245
|
+
):
|
|
246
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
247
|
+
else:
|
|
248
|
+
for row in body_rows:
|
|
249
|
+
item = _summary(row, width, folder, account)
|
|
250
|
+
if item.reference.identifier in identifiers:
|
|
251
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
252
|
+
identifiers.add(item.reference.identifier)
|
|
253
|
+
total += (
|
|
254
|
+
len(item.correspondent)
|
|
255
|
+
+ len(item.subject)
|
|
256
|
+
+ len(item.timestamp.raw)
|
|
257
|
+
+ len(item.recipient_read_status or "")
|
|
258
|
+
)
|
|
259
|
+
if (
|
|
260
|
+
len(items) >= MESSAGE_MAX_PAGE_ITEMS
|
|
261
|
+
or total > MESSAGE_MAX_TOTAL_TEXT_LENGTH
|
|
262
|
+
):
|
|
263
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
264
|
+
items.append(item)
|
|
265
|
+
if len(items) == MESSAGE_MAX_PAGE_ITEMS and not any(
|
|
266
|
+
"pagination" in node.get("class", "").split() for node in document.iter("div")
|
|
267
|
+
):
|
|
268
|
+
# A full page without a pager cannot prove this is the last page.
|
|
269
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
270
|
+
result = tuple(items)
|
|
271
|
+
# Account/folder scope is already explicit; ordered row identities detect
|
|
272
|
+
# mid-page drift. Read state is excluded: opening a message, here or on
|
|
273
|
+
# another device, changes it without moving any row.
|
|
274
|
+
fingerprint = hashlib.sha256(
|
|
275
|
+
repr(
|
|
276
|
+
tuple(
|
|
277
|
+
(
|
|
278
|
+
astuple(i.reference),
|
|
279
|
+
i.correspondent,
|
|
280
|
+
i.subject,
|
|
281
|
+
astuple(i.timestamp),
|
|
282
|
+
i.has_attachment,
|
|
283
|
+
)
|
|
284
|
+
for i in result
|
|
285
|
+
)
|
|
286
|
+
).encode()
|
|
287
|
+
).hexdigest()
|
|
288
|
+
return result, count, fingerprint
|