librus-python-api 1.0.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- librus_python_api/__init__.py +243 -0
- librus_python_api/_notification_codec.py +310 -0
- librus_python_api/_storage.py +367 -0
- librus_python_api/announcements.py +159 -0
- librus_python_api/attachment_routes.py +114 -0
- librus_python_api/attachments.py +297 -0
- librus_python_api/attendance.py +182 -0
- librus_python_api/attendance_frequency.py +112 -0
- librus_python_api/budget.py +79 -0
- librus_python_api/checkpoint.py +61 -0
- librus_python_api/completed_lessons.py +216 -0
- librus_python_api/config.py +1410 -0
- librus_python_api/detail_fields.py +50 -0
- librus_python_api/diagnostics.py +25 -0
- librus_python_api/exceptions.py +172 -0
- librus_python_api/files.py +156 -0
- librus_python_api/grade_parsers.py +169 -0
- librus_python_api/grade_records.py +454 -0
- librus_python_api/homework_range.py +41 -0
- librus_python_api/lifecycle.py +24 -0
- librus_python_api/markup.py +147 -0
- librus_python_api/message_content.py +230 -0
- librus_python_api/messages.py +288 -0
- librus_python_api/models.py +1160 -0
- librus_python_api/modern_body.py +75 -0
- librus_python_api/modern_mailbox.py +459 -0
- librus_python_api/modern_messages.py +276 -0
- librus_python_api/notification_models.py +67 -0
- librus_python_api/notification_persistence.py +1044 -0
- librus_python_api/notification_workflow.py +303 -0
- librus_python_api/notifications.py +216 -0
- librus_python_api/parsers.py +232 -0
- librus_python_api/parsing.py +49 -0
- librus_python_api/persistence.py +405 -0
- librus_python_api/py.typed +0 -0
- librus_python_api/recipients.py +271 -0
- librus_python_api/scheduler.py +287 -0
- librus_python_api/school_reads.py +400 -0
- librus_python_api/sending.py +125 -0
- librus_python_api/service.py +2285 -0
- librus_python_api/timetable.py +261 -0
- librus_python_api/transport.py +956 -0
- librus_python_api-1.0.0rc1.dist-info/METADATA +254 -0
- librus_python_api-1.0.0rc1.dist-info/RECORD +46 -0
- librus_python_api-1.0.0rc1.dist-info/WHEEL +4 -0
- librus_python_api-1.0.0rc1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,454 @@
|
|
|
1
|
+
"""Original bounded parser for inline semester grades and school averages."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from typing import Literal, cast
|
|
6
|
+
|
|
7
|
+
from lxml import html
|
|
8
|
+
|
|
9
|
+
from librus_python_api import markup
|
|
10
|
+
from librus_python_api.config import (
|
|
11
|
+
GRADE_AVERAGE_HEADERS,
|
|
12
|
+
GRADE_BODY_PREFIX_COLUMNS,
|
|
13
|
+
GRADE_CURRENT_HEADER,
|
|
14
|
+
GRADE_EMPTY_MARKERS,
|
|
15
|
+
GRADE_MAX_METADATA_LENGTH,
|
|
16
|
+
GRADE_MAX_RECORDS,
|
|
17
|
+
GRADE_MAX_SUBJECTS,
|
|
18
|
+
GRADE_PERIOD_HEADERS,
|
|
19
|
+
GRADE_PREDICTED_ANNUAL_HEADER,
|
|
20
|
+
GRADE_PREDICTED_PERIOD_HEADERS,
|
|
21
|
+
GRADE_PUBLICATION_DATE_LABEL,
|
|
22
|
+
GRADE_PUBLICATION_PERIOD_LABELS,
|
|
23
|
+
GRADE_PUBLICATION_TEACHER_LABEL,
|
|
24
|
+
)
|
|
25
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
26
|
+
from librus_python_api.grade_parsers import _locate, _subject_rows, _summary
|
|
27
|
+
from librus_python_api.models import (
|
|
28
|
+
Availability,
|
|
29
|
+
DescriptiveGrade,
|
|
30
|
+
DescriptiveGradeSummary,
|
|
31
|
+
GradeKind,
|
|
32
|
+
GradeRecords,
|
|
33
|
+
GradeSummaryValue,
|
|
34
|
+
NumericGrade,
|
|
35
|
+
SchoolAverage,
|
|
36
|
+
)
|
|
37
|
+
from librus_python_api.parsers import parse_page
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def _label(title: str) -> str:
|
|
41
|
+
return " ".join(re.split(r"<br\s*/?>", title, flags=re.I)[0].split())
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _layout(table: html.HtmlElement) -> tuple[list[int], dict[int, int]]:
|
|
45
|
+
current: list[int] = []
|
|
46
|
+
averages: dict[int, int] = {}
|
|
47
|
+
for row in markup.rows(table):
|
|
48
|
+
if next(row.iterancestors("thead"), None) is None:
|
|
49
|
+
continue
|
|
50
|
+
cells = markup.cells(row)
|
|
51
|
+
if not any(markup.text(c) == GRADE_CURRENT_HEADER for c in cells):
|
|
52
|
+
continue
|
|
53
|
+
if current:
|
|
54
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
55
|
+
position = GRADE_BODY_PREFIX_COLUMNS
|
|
56
|
+
for cell in cells:
|
|
57
|
+
if markup.text(cell) == GRADE_CURRENT_HEADER:
|
|
58
|
+
if markup.colspan(cell) != 1:
|
|
59
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
60
|
+
current.append(position)
|
|
61
|
+
period = GRADE_AVERAGE_HEADERS.get(_label(cell.get("title", "")))
|
|
62
|
+
if period is not None:
|
|
63
|
+
if period in averages or markup.colspan(cell) != 1:
|
|
64
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
65
|
+
averages[period] = position
|
|
66
|
+
position += markup.colspan(cell)
|
|
67
|
+
if len(current) != 2:
|
|
68
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
69
|
+
return current, averages
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _numeric(
|
|
73
|
+
element: html.HtmlElement,
|
|
74
|
+
subject: str,
|
|
75
|
+
semester: Literal[0, 1, 2],
|
|
76
|
+
kind: GradeKind = GradeKind.CURRENT,
|
|
77
|
+
) -> NumericGrade:
|
|
78
|
+
metadata = markup.tooltip_fields(element)
|
|
79
|
+
counts = metadata.get("Licz do średniej")
|
|
80
|
+
if counts is not None and counts.casefold() not in ("tak", "nie"):
|
|
81
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
82
|
+
weight = metadata.get("Waga")
|
|
83
|
+
if weight is not None and not re.fullmatch(r"[0-9]{1,4}", weight):
|
|
84
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
85
|
+
raw = markup.text(element)
|
|
86
|
+
if not raw:
|
|
87
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
88
|
+
href = element.get("href")
|
|
89
|
+
if href is not None and len(href) > GRADE_MAX_METADATA_LENGTH:
|
|
90
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
91
|
+
return NumericGrade(
|
|
92
|
+
subject,
|
|
93
|
+
raw,
|
|
94
|
+
markup.civil_date(metadata.get("Data")),
|
|
95
|
+
semester,
|
|
96
|
+
None if counts is None else counts.casefold() == "tak",
|
|
97
|
+
None if weight is None else int(weight),
|
|
98
|
+
metadata.get("Kategoria"),
|
|
99
|
+
metadata.get("Nauczyciel"),
|
|
100
|
+
metadata.get("Komentarz"),
|
|
101
|
+
href,
|
|
102
|
+
tuple(metadata.items()),
|
|
103
|
+
kind,
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _descriptive(
|
|
108
|
+
box: html.HtmlElement,
|
|
109
|
+
element: html.HtmlElement,
|
|
110
|
+
subject: str,
|
|
111
|
+
semester: Literal[0, 1, 2],
|
|
112
|
+
kind: GradeKind = GradeKind.CURRENT,
|
|
113
|
+
) -> DescriptiveGrade:
|
|
114
|
+
metadata = markup.tooltip_fields(box)
|
|
115
|
+
raw = markup.text(element)
|
|
116
|
+
if not raw:
|
|
117
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
118
|
+
return DescriptiveGrade(
|
|
119
|
+
subject,
|
|
120
|
+
raw,
|
|
121
|
+
markup.civil_date(metadata.get("Data")),
|
|
122
|
+
semester,
|
|
123
|
+
metadata.get("Nauczyciel"),
|
|
124
|
+
metadata.get("Komentarz"),
|
|
125
|
+
tuple(metadata.items()),
|
|
126
|
+
kind,
|
|
127
|
+
)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@dataclass(slots=True)
|
|
131
|
+
class _Collection:
|
|
132
|
+
numeric: list[NumericGrade] = field(default_factory=list, repr=False)
|
|
133
|
+
descriptive: list[DescriptiveGrade] = field(default_factory=list, repr=False)
|
|
134
|
+
descriptive_summaries: list[DescriptiveGradeSummary] = field(
|
|
135
|
+
default_factory=list, repr=False
|
|
136
|
+
)
|
|
137
|
+
averages: list[SchoolAverage] = field(default_factory=list, repr=False)
|
|
138
|
+
subjects: set[tuple[bool, str]] = field(default_factory=set, repr=False)
|
|
139
|
+
consumed: set[html.HtmlElement] = field(default_factory=set, repr=False)
|
|
140
|
+
publication_rows: set[html.HtmlElement] = field(default_factory=set, repr=False)
|
|
141
|
+
|
|
142
|
+
def check_size(self) -> None:
|
|
143
|
+
if (
|
|
144
|
+
len(self.numeric) + len(self.descriptive) + len(self.descriptive_summaries)
|
|
145
|
+
> GRADE_MAX_RECORDS
|
|
146
|
+
):
|
|
147
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
148
|
+
|
|
149
|
+
def subject(self, value: str, *, descriptive: bool = False) -> None:
|
|
150
|
+
key = (descriptive, value)
|
|
151
|
+
if not value or key in self.subjects:
|
|
152
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
153
|
+
self.subjects.add(key)
|
|
154
|
+
if len(self.subjects) > GRADE_MAX_SUBJECTS:
|
|
155
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _boxes(cell: html.HtmlElement) -> list[html.HtmlElement]:
|
|
159
|
+
return [
|
|
160
|
+
node
|
|
161
|
+
for node in cell.iter("span")
|
|
162
|
+
if "grade-box" in node.get("class", "").split()
|
|
163
|
+
]
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _entries(
|
|
167
|
+
cell: html.HtmlElement,
|
|
168
|
+
subject: str,
|
|
169
|
+
period: Literal[0, 1, 2],
|
|
170
|
+
collection: _Collection,
|
|
171
|
+
*,
|
|
172
|
+
kind: GradeKind = GradeKind.CURRENT,
|
|
173
|
+
descriptive: bool = False,
|
|
174
|
+
require_empty_marker: bool = True,
|
|
175
|
+
) -> None:
|
|
176
|
+
boxes = _boxes(cell)
|
|
177
|
+
if (
|
|
178
|
+
not boxes
|
|
179
|
+
and require_empty_marker
|
|
180
|
+
and markup.text(cell) not in GRADE_EMPTY_MARKERS
|
|
181
|
+
):
|
|
182
|
+
if (
|
|
183
|
+
not descriptive
|
|
184
|
+
or period == 0
|
|
185
|
+
or any(isinstance(node.tag, str) for node in cell)
|
|
186
|
+
):
|
|
187
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
188
|
+
raw = markup.text(cell)
|
|
189
|
+
if len(raw) > GRADE_MAX_METADATA_LENGTH:
|
|
190
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
191
|
+
collection.descriptive_summaries.append(
|
|
192
|
+
DescriptiveGradeSummary(subject, period, raw)
|
|
193
|
+
)
|
|
194
|
+
collection.check_size()
|
|
195
|
+
for box in boxes:
|
|
196
|
+
anchors = [e for e in box if e.tag == "a"]
|
|
197
|
+
descriptions = [
|
|
198
|
+
e for e in box if e.tag == "span" and "ocena" in e.get("class", "").split()
|
|
199
|
+
]
|
|
200
|
+
if len(anchors) + len(descriptions) != 1:
|
|
201
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
202
|
+
if anchors:
|
|
203
|
+
record = _numeric(anchors[0], subject, period, kind)
|
|
204
|
+
if descriptive:
|
|
205
|
+
href = record.href
|
|
206
|
+
if href is not None and re.sub(
|
|
207
|
+
r"[\x00-\x20]", "", href
|
|
208
|
+
).casefold().startswith(("javascript:", "data:", "vbscript:")):
|
|
209
|
+
href = None
|
|
210
|
+
collection.descriptive.append(
|
|
211
|
+
DescriptiveGrade(
|
|
212
|
+
record.subject,
|
|
213
|
+
record.raw,
|
|
214
|
+
record.day,
|
|
215
|
+
record.semester,
|
|
216
|
+
record.teacher,
|
|
217
|
+
record.comment,
|
|
218
|
+
record.metadata,
|
|
219
|
+
kind,
|
|
220
|
+
href,
|
|
221
|
+
)
|
|
222
|
+
)
|
|
223
|
+
else:
|
|
224
|
+
collection.numeric.append(record)
|
|
225
|
+
else:
|
|
226
|
+
collection.descriptive.append(
|
|
227
|
+
_descriptive(box, descriptions[0], subject, period, kind)
|
|
228
|
+
)
|
|
229
|
+
collection.consumed.add(box)
|
|
230
|
+
collection.check_size()
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def _descriptive_row(cells: list[html.HtmlElement]) -> bool:
|
|
234
|
+
return (
|
|
235
|
+
len(cells) in (4, 6)
|
|
236
|
+
and {"micro", "center", "screen-only"} <= set(cells[0].get("class", "").split())
|
|
237
|
+
and all(markup.colspan(cell) == 1 for cell in cells)
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _read_descriptive_row(
|
|
242
|
+
cells: list[html.HtmlElement], collection: _Collection
|
|
243
|
+
) -> None:
|
|
244
|
+
subject = markup.text(cells[1])
|
|
245
|
+
collection.subject(subject, descriptive=True)
|
|
246
|
+
for period in (1, 2):
|
|
247
|
+
_entries(cells[period + 1], subject, period, collection, descriptive=True)
|
|
248
|
+
if len(cells) == 6:
|
|
249
|
+
for cell in cells[4:]:
|
|
250
|
+
if _boxes(cell):
|
|
251
|
+
# These columns have no established semester semantics.
|
|
252
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _period_columns(table: html.HtmlElement) -> dict[tuple[int, GradeKind], int]:
|
|
256
|
+
result: dict[tuple[int, GradeKind], int] = {}
|
|
257
|
+
for row in markup.rows(table):
|
|
258
|
+
cells = markup.cells(row)
|
|
259
|
+
if next(row.iterancestors("thead"), None) is None or not any(
|
|
260
|
+
markup.text(c) == GRADE_CURRENT_HEADER for c in cells
|
|
261
|
+
):
|
|
262
|
+
continue
|
|
263
|
+
position = GRADE_BODY_PREFIX_COLUMNS
|
|
264
|
+
for cell in cells:
|
|
265
|
+
label = _label(cell.get("title", ""))
|
|
266
|
+
period = GRADE_PERIOD_HEADERS.get(label)
|
|
267
|
+
kind = GradeKind.ANNUAL if period == 0 else GradeKind.PERIOD
|
|
268
|
+
if label == GRADE_PREDICTED_ANNUAL_HEADER:
|
|
269
|
+
period, kind = 0, GradeKind.PREDICTED_ANNUAL
|
|
270
|
+
elif label in GRADE_PREDICTED_PERIOD_HEADERS:
|
|
271
|
+
period, kind = (
|
|
272
|
+
GRADE_PREDICTED_PERIOD_HEADERS[label],
|
|
273
|
+
GradeKind.PREDICTED_PERIOD,
|
|
274
|
+
)
|
|
275
|
+
if period is not None:
|
|
276
|
+
if (period, kind) in result or markup.colspan(cell) != 1:
|
|
277
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
278
|
+
result[(period, kind)] = position
|
|
279
|
+
position += markup.colspan(cell)
|
|
280
|
+
return result
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _read_numeric_table(table: html.HtmlElement, collection: _Collection) -> None:
|
|
284
|
+
# Preserve the established full-row and summary alignment validation.
|
|
285
|
+
_, fields, width = _locate(table)
|
|
286
|
+
current, means = _layout(table)
|
|
287
|
+
periods = _period_columns(table)
|
|
288
|
+
for cells in _subject_rows(table, width):
|
|
289
|
+
if cells[0].getparent() in collection.publication_rows:
|
|
290
|
+
continue
|
|
291
|
+
if _descriptive_row(cells):
|
|
292
|
+
_read_descriptive_row(cells, collection)
|
|
293
|
+
continue
|
|
294
|
+
subject = _summary(cells, fields, width).subject
|
|
295
|
+
collection.subject(subject)
|
|
296
|
+
columns = [cell for cell in cells for _ in range(markup.colspan(cell))]
|
|
297
|
+
for period in (1, 2, 0):
|
|
298
|
+
index = means.get(period)
|
|
299
|
+
value = (
|
|
300
|
+
GradeSummaryValue(Availability.UNAVAILABLE, None)
|
|
301
|
+
if index is None or markup.colspan(columns[index]) != 1
|
|
302
|
+
else GradeSummaryValue(
|
|
303
|
+
Availability.AVAILABLE, markup.text(columns[index])
|
|
304
|
+
)
|
|
305
|
+
)
|
|
306
|
+
collection.averages.append(SchoolAverage(subject, period, value))
|
|
307
|
+
if columns[current[0]] is columns[current[1]] and _boxes(columns[current[0]]):
|
|
308
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
309
|
+
seen: set[html.HtmlElement] = set()
|
|
310
|
+
for period, index in enumerate(current, start=1):
|
|
311
|
+
if columns[index] not in seen:
|
|
312
|
+
_entries(
|
|
313
|
+
columns[index], subject, cast(Literal[1, 2], period), collection
|
|
314
|
+
)
|
|
315
|
+
seen.add(columns[index])
|
|
316
|
+
for (period, kind), index in periods.items():
|
|
317
|
+
cell = columns[index]
|
|
318
|
+
if cell in seen and _boxes(cell):
|
|
319
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
320
|
+
_entries(
|
|
321
|
+
cell,
|
|
322
|
+
subject,
|
|
323
|
+
cast(Literal[0, 1, 2], period),
|
|
324
|
+
collection,
|
|
325
|
+
kind=kind,
|
|
326
|
+
require_empty_marker=False,
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _read_publications(table: html.HtmlElement, collection: _Collection) -> None:
|
|
331
|
+
rows = list(markup.rows(table))
|
|
332
|
+
for index, row in enumerate(rows):
|
|
333
|
+
headers = [
|
|
334
|
+
cell
|
|
335
|
+
for cell in markup.cells(row)
|
|
336
|
+
if cell.tag == "th"
|
|
337
|
+
and list(cell.iter("strong"))
|
|
338
|
+
and GRADE_PUBLICATION_DATE_LABEL in cell.text_content()
|
|
339
|
+
]
|
|
340
|
+
if not headers:
|
|
341
|
+
continue
|
|
342
|
+
if len(headers) != 1 or index + 1 == len(rows):
|
|
343
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
344
|
+
header = headers[0]
|
|
345
|
+
titles = list(header.iter("strong"))
|
|
346
|
+
if len(titles) != 1:
|
|
347
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
348
|
+
title = markup.text(titles[0])
|
|
349
|
+
info = markup.text(header)
|
|
350
|
+
date_info = info.partition(GRADE_PUBLICATION_DATE_LABEL)[2].strip()
|
|
351
|
+
day = markup.civil_date(date_info[:10])
|
|
352
|
+
if len(date_info) > 10 and date_info[10] not in (" ", ",", ")"):
|
|
353
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
354
|
+
teacher_info = info.partition(GRADE_PUBLICATION_TEACHER_LABEL)[2].strip()
|
|
355
|
+
teacher = teacher_info.rsplit(")", 1)[0].strip() if teacher_info else None
|
|
356
|
+
matching = {
|
|
357
|
+
period
|
|
358
|
+
for label, period in GRADE_PUBLICATION_PERIOD_LABELS.items()
|
|
359
|
+
if re.search(r"(?<!\w)" + re.escape(label) + r"(?!\w)", title)
|
|
360
|
+
}
|
|
361
|
+
if len(matching) > 1:
|
|
362
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
363
|
+
paragraphs = list(rows[index + 1].iter("p"))
|
|
364
|
+
raw = "\n".join(markup.text(paragraph) for paragraph in paragraphs).strip()
|
|
365
|
+
if not title or not raw:
|
|
366
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
367
|
+
if len(raw) > GRADE_MAX_METADATA_LENGTH:
|
|
368
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
369
|
+
metadata = (("Data", date_info[:10]),)
|
|
370
|
+
collection.descriptive.append(
|
|
371
|
+
DescriptiveGrade(
|
|
372
|
+
title,
|
|
373
|
+
raw,
|
|
374
|
+
day,
|
|
375
|
+
cast(Literal[1, 2] | None, next(iter(matching), None)),
|
|
376
|
+
teacher,
|
|
377
|
+
None,
|
|
378
|
+
metadata,
|
|
379
|
+
GradeKind.PUBLICATION,
|
|
380
|
+
)
|
|
381
|
+
)
|
|
382
|
+
collection.consumed.add(header)
|
|
383
|
+
collection.publication_rows.update((row, rows[index + 1]))
|
|
384
|
+
collection.check_size()
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def parse_grade_records(body: bytes) -> GradeRecords:
|
|
388
|
+
document = parse_page(body)
|
|
389
|
+
collection = _Collection()
|
|
390
|
+
numeric_tables = []
|
|
391
|
+
tables = list(document.iter("table"))
|
|
392
|
+
for table in tables:
|
|
393
|
+
if not {"decorated", "stretch"} <= set(table.get("class", "").split()):
|
|
394
|
+
continue
|
|
395
|
+
if any(
|
|
396
|
+
next(row.iterancestors("thead"), None) is not None
|
|
397
|
+
and any(markup.text(c) == GRADE_CURRENT_HEADER for c in markup.cells(row))
|
|
398
|
+
for row in markup.rows(table)
|
|
399
|
+
):
|
|
400
|
+
numeric_tables.append(table)
|
|
401
|
+
if len(numeric_tables) > 1:
|
|
402
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
403
|
+
for table in tables:
|
|
404
|
+
_read_publications(table, collection)
|
|
405
|
+
if table in numeric_tables:
|
|
406
|
+
_read_numeric_table(table, collection)
|
|
407
|
+
else:
|
|
408
|
+
for row in markup.rows(table):
|
|
409
|
+
cells = markup.cells(row)
|
|
410
|
+
if _descriptive_row(cells):
|
|
411
|
+
_read_descriptive_row(cells, collection)
|
|
412
|
+
elif (
|
|
413
|
+
cells
|
|
414
|
+
and cells[0].tag == "td"
|
|
415
|
+
and {"micro", "center", "screen-only"}
|
|
416
|
+
<= set(cells[0].get("class", "").split())
|
|
417
|
+
):
|
|
418
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
419
|
+
for element in document.iter():
|
|
420
|
+
if not isinstance(element.tag, str) or element in collection.consumed:
|
|
421
|
+
continue
|
|
422
|
+
dated_box = "grade-box" in element.get("class", "").split() and any(
|
|
423
|
+
"Data:" in (node.get("title") or "") for node in element.iter()
|
|
424
|
+
)
|
|
425
|
+
if dated_box:
|
|
426
|
+
if any(
|
|
427
|
+
parent in numeric_tables for parent in element.iterancestors("table")
|
|
428
|
+
):
|
|
429
|
+
# Expanded full-row detail mirrors were intentionally excluded.
|
|
430
|
+
cell = next(element.iterancestors("td", "th"), None)
|
|
431
|
+
if (
|
|
432
|
+
cell is not None
|
|
433
|
+
and next(cell.iterancestors("table"), None) not in numeric_tables
|
|
434
|
+
):
|
|
435
|
+
continue
|
|
436
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
437
|
+
if not collection.subjects and not collection.descriptive:
|
|
438
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
439
|
+
for descriptive, subject in sorted(collection.subjects):
|
|
440
|
+
if descriptive and (False, subject) not in collection.subjects:
|
|
441
|
+
collection.averages.extend(
|
|
442
|
+
SchoolAverage(
|
|
443
|
+
subject,
|
|
444
|
+
period,
|
|
445
|
+
GradeSummaryValue(Availability.UNAVAILABLE, None),
|
|
446
|
+
)
|
|
447
|
+
for period in (1, 2, 0)
|
|
448
|
+
)
|
|
449
|
+
return GradeRecords(
|
|
450
|
+
tuple(collection.numeric),
|
|
451
|
+
tuple(collection.descriptive),
|
|
452
|
+
tuple(collection.averages),
|
|
453
|
+
tuple(collection.descriptive_summaries),
|
|
454
|
+
)
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""Bounded aggregation by upstream identity, without guessed date-filter semantics."""
|
|
2
|
+
|
|
3
|
+
from librus_python_api.config import SCHOOL_MAX_TOTAL_TEXT_LENGTH
|
|
4
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
5
|
+
from librus_python_api.models import HomeworkItem, SchoolReference
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class HomeworkAccumulator:
|
|
9
|
+
def __init__(self, max_items: int) -> None:
|
|
10
|
+
self.items: list[HomeworkItem] = []
|
|
11
|
+
self._references: dict[SchoolReference, HomeworkItem] = {}
|
|
12
|
+
self._max_items = max_items
|
|
13
|
+
self._text_length = 0
|
|
14
|
+
|
|
15
|
+
def extend(self, items: tuple[HomeworkItem, ...]) -> None:
|
|
16
|
+
for item in items:
|
|
17
|
+
if item.reference is not None:
|
|
18
|
+
previous = self._references.get(item.reference)
|
|
19
|
+
if previous is not None:
|
|
20
|
+
if previous != item:
|
|
21
|
+
# The selection is not a server-side snapshot. Never
|
|
22
|
+
# silently prefer a stale or changed duplicate.
|
|
23
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
24
|
+
continue
|
|
25
|
+
self._references[item.reference] = item
|
|
26
|
+
self._text_length += sum(
|
|
27
|
+
len(value)
|
|
28
|
+
for value in (
|
|
29
|
+
item.subject,
|
|
30
|
+
item.teacher,
|
|
31
|
+
item.topic,
|
|
32
|
+
item.category,
|
|
33
|
+
item.submission_status or "",
|
|
34
|
+
)
|
|
35
|
+
)
|
|
36
|
+
if (
|
|
37
|
+
len(self.items) >= self._max_items
|
|
38
|
+
or self._text_length > SCHOOL_MAX_TOTAL_TEXT_LENGTH
|
|
39
|
+
):
|
|
40
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
41
|
+
self.items.append(item)
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Joined resource ownership even when callers cancel repeatedly."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
async def join_owned[T](future: asyncio.Future[T]) -> bool:
|
|
7
|
+
"""Join without forwarding cancellation; return whether joining was canceled.
|
|
8
|
+
|
|
9
|
+
Cleanup must finish before resource/admission slots can be reused. This helper
|
|
10
|
+
is only for already-owned work, never for ordinary uncancelable operations.
|
|
11
|
+
"""
|
|
12
|
+
interrupted = False
|
|
13
|
+
while not future.done():
|
|
14
|
+
try:
|
|
15
|
+
# wait() never forwards cancellation to owned work and has no
|
|
16
|
+
# abandoned shield wrapper that can log an already-handled error.
|
|
17
|
+
await asyncio.wait((future,))
|
|
18
|
+
except asyncio.CancelledError:
|
|
19
|
+
interrupted = True
|
|
20
|
+
except Exception:
|
|
21
|
+
break
|
|
22
|
+
if not future.cancelled():
|
|
23
|
+
future.exception()
|
|
24
|
+
return interrupted
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""Bounded HTML reading helpers shared by every Synergia page parser."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Iterator
|
|
5
|
+
from datetime import date
|
|
6
|
+
from urllib.parse import urlsplit
|
|
7
|
+
|
|
8
|
+
from lxml import etree, html
|
|
9
|
+
|
|
10
|
+
from librus_python_api.config import (
|
|
11
|
+
ATTENDANCE_DETAIL_PATH_PREFIX,
|
|
12
|
+
GRADE_MAX_COLUMNS,
|
|
13
|
+
GRADE_MAX_METADATA_FIELDS,
|
|
14
|
+
GRADE_MAX_METADATA_LENGTH,
|
|
15
|
+
UPSTREAM_ORIGINS,
|
|
16
|
+
WEEKDAY_LABELS,
|
|
17
|
+
)
|
|
18
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
19
|
+
from librus_python_api.parsers import parse_html_document
|
|
20
|
+
|
|
21
|
+
FIELD_LENGTH = 1024
|
|
22
|
+
_ACTIVE_CONTENT = frozenset({"script", "style", "iframe", "object", "embed", "form"})
|
|
23
|
+
_BLOCKS = frozenset({"br", "p", "div", "li"})
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def text(
|
|
27
|
+
element: html.HtmlElement, limit: int = FIELD_LENGTH, *, multiline: bool = False
|
|
28
|
+
) -> str:
|
|
29
|
+
"""Rendered text with block boundaries as line breaks; never truncated.
|
|
30
|
+
|
|
31
|
+
Active content inside a data cell is an unsupported layout, not text.
|
|
32
|
+
"""
|
|
33
|
+
parts: list[str] = []
|
|
34
|
+
for event, node in etree.iterwalk(element, events=("start", "end", "comment")):
|
|
35
|
+
if node.tag in _ACTIVE_CONTENT:
|
|
36
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
37
|
+
if node.tag in _BLOCKS:
|
|
38
|
+
parts.append("\n")
|
|
39
|
+
if event == "start" and isinstance(node.tag, str) and node.text:
|
|
40
|
+
parts.append(re.sub(r"\s+", " ", node.text))
|
|
41
|
+
if event in ("end", "comment") and node is not element and node.tail:
|
|
42
|
+
parts.append(re.sub(r"\s+", " ", node.tail))
|
|
43
|
+
rendered = "".join(parts)
|
|
44
|
+
value = (
|
|
45
|
+
"\n".join(
|
|
46
|
+
line for raw in rendered.splitlines() if (line := " ".join(raw.split()))
|
|
47
|
+
)
|
|
48
|
+
if multiline
|
|
49
|
+
else " ".join(rendered.split())
|
|
50
|
+
)
|
|
51
|
+
if len(value) > limit:
|
|
52
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
53
|
+
return value
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def cells(row: html.HtmlElement) -> list[html.HtmlElement]:
|
|
57
|
+
return [item for item in row if item.tag in ("td", "th")]
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def rows(table: html.HtmlElement) -> Iterator[html.HtmlElement]:
|
|
61
|
+
"""Rows of this table only, never rows of a nested table."""
|
|
62
|
+
for row in table.iter("tr"):
|
|
63
|
+
if next(row.iterancestors("table"), None) is table:
|
|
64
|
+
yield row
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def in_header(row: html.HtmlElement) -> bool:
|
|
68
|
+
return next(row.iterancestors("thead"), None) is not None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def colspan(cell: html.HtmlElement) -> int:
|
|
72
|
+
value = cell.get("colspan", "1")
|
|
73
|
+
if not re.fullmatch(r"[1-9][0-9]{0,2}", value):
|
|
74
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
75
|
+
span = int(value)
|
|
76
|
+
if span > GRADE_MAX_COLUMNS:
|
|
77
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
78
|
+
if cell.get("rowspan", "1") != "1":
|
|
79
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
80
|
+
return span
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def tooltip_fields(element: html.HtmlElement) -> dict[str, str]:
|
|
84
|
+
"""Parse a 'Label: value<br>' title attribute into unique labelled fields."""
|
|
85
|
+
title = element.get("title", "")
|
|
86
|
+
if len(title) > GRADE_MAX_METADATA_LENGTH:
|
|
87
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
88
|
+
result: dict[str, str] = {}
|
|
89
|
+
chunks = re.split(r"<br\s*/?>|\n", title, flags=re.I)
|
|
90
|
+
if len(chunks) > GRADE_MAX_METADATA_FIELDS:
|
|
91
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
92
|
+
for chunk in chunks:
|
|
93
|
+
if not chunk.strip():
|
|
94
|
+
continue
|
|
95
|
+
# Tooltips are HTML inside an attribute; parse markup only where present.
|
|
96
|
+
rendered = (
|
|
97
|
+
text(parse_html_document(chunk.encode()))
|
|
98
|
+
if "<" in chunk
|
|
99
|
+
else " ".join(chunk.split())
|
|
100
|
+
)
|
|
101
|
+
key, separator, value = rendered.partition(":")
|
|
102
|
+
key, value = key.strip(), value.strip()
|
|
103
|
+
if not separator or not key or key in result:
|
|
104
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
105
|
+
result[key] = value
|
|
106
|
+
return result
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def civil_date(value: str | None) -> date:
|
|
110
|
+
"""An ISO date, optionally followed by its weekday label in parentheses."""
|
|
111
|
+
match = re.fullmatch(r"([0-9]{4}-[0-9]{2}-[0-9]{2})(?: \(([^()]+)\))?", value or "")
|
|
112
|
+
if match is None or (match[2] is not None and match[2] not in WEEKDAY_LABELS):
|
|
113
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
114
|
+
try:
|
|
115
|
+
return date.fromisoformat(match[1])
|
|
116
|
+
except ValueError:
|
|
117
|
+
pass
|
|
118
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def attendance_detail_id(element: html.HtmlElement) -> str | None:
|
|
122
|
+
"""The numeric ID of an inert attendance-detail popup link, if any."""
|
|
123
|
+
script = element.get("onclick", "")
|
|
124
|
+
if len(script) > GRADE_MAX_METADATA_LENGTH:
|
|
125
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
126
|
+
match = re.fullmatch(
|
|
127
|
+
r"\s*otworz_w_nowym_oknie\(\s*(['\"])([^'\"\\\s]+)\1"
|
|
128
|
+
r"""(?:\s*,\s*(?:'[^'\\\r\n]*'|"[^"\\\r\n]*"|[0-9]{1,6})){1,4}"""
|
|
129
|
+
r"\s*\)\s*;?\s*",
|
|
130
|
+
script,
|
|
131
|
+
)
|
|
132
|
+
if match is None:
|
|
133
|
+
return None
|
|
134
|
+
try:
|
|
135
|
+
target = urlsplit(match[2])
|
|
136
|
+
except ValueError:
|
|
137
|
+
return None
|
|
138
|
+
if target.query or target.fragment or target.username or target.password:
|
|
139
|
+
return None
|
|
140
|
+
if target.scheme or target.netloc:
|
|
141
|
+
expected = urlsplit(UPSTREAM_ORIGINS["synergia"])
|
|
142
|
+
if (target.scheme, target.netloc) != (expected.scheme, expected.netloc):
|
|
143
|
+
return None
|
|
144
|
+
if not target.path.startswith(ATTENDANCE_DETAIL_PATH_PREFIX):
|
|
145
|
+
return None
|
|
146
|
+
identifier = target.path[len(ATTENDANCE_DETAIL_PATH_PREFIX) :]
|
|
147
|
+
return identifier if re.fullmatch(r"[0-9]{1,64}", identifier) else None
|