librus-python-api 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. librus_python_api/__init__.py +243 -0
  2. librus_python_api/_notification_bootstrap.py +124 -0
  3. librus_python_api/_notification_codec.py +396 -0
  4. librus_python_api/_storage.py +403 -0
  5. librus_python_api/_windows_filesystem.py +390 -0
  6. librus_python_api/announcements.py +159 -0
  7. librus_python_api/attachment_routes.py +114 -0
  8. librus_python_api/attachments.py +297 -0
  9. librus_python_api/attendance.py +182 -0
  10. librus_python_api/attendance_frequency.py +112 -0
  11. librus_python_api/budget.py +79 -0
  12. librus_python_api/checkpoint.py +61 -0
  13. librus_python_api/completed_lessons.py +216 -0
  14. librus_python_api/config.py +1410 -0
  15. librus_python_api/detail_fields.py +50 -0
  16. librus_python_api/diagnostics.py +25 -0
  17. librus_python_api/exceptions.py +172 -0
  18. librus_python_api/files.py +242 -0
  19. librus_python_api/grade_parsers.py +169 -0
  20. librus_python_api/grade_records.py +454 -0
  21. librus_python_api/homework_range.py +41 -0
  22. librus_python_api/lifecycle.py +24 -0
  23. librus_python_api/markup.py +147 -0
  24. librus_python_api/message_content.py +230 -0
  25. librus_python_api/messages.py +288 -0
  26. librus_python_api/models.py +1160 -0
  27. librus_python_api/modern_body.py +75 -0
  28. librus_python_api/modern_mailbox.py +459 -0
  29. librus_python_api/modern_messages.py +276 -0
  30. librus_python_api/notification_models.py +145 -0
  31. librus_python_api/notification_persistence.py +1243 -0
  32. librus_python_api/notification_workflow.py +337 -0
  33. librus_python_api/notifications.py +216 -0
  34. librus_python_api/parsers.py +232 -0
  35. librus_python_api/parsing.py +49 -0
  36. librus_python_api/persistence.py +419 -0
  37. librus_python_api/py.typed +0 -0
  38. librus_python_api/recipients.py +271 -0
  39. librus_python_api/scheduler.py +287 -0
  40. librus_python_api/school_reads.py +400 -0
  41. librus_python_api/sending.py +125 -0
  42. librus_python_api/service.py +2285 -0
  43. librus_python_api/timetable.py +261 -0
  44. librus_python_api/transport.py +956 -0
  45. librus_python_api-1.0.0.dist-info/METADATA +262 -0
  46. librus_python_api-1.0.0.dist-info/RECORD +48 -0
  47. librus_python_api-1.0.0.dist-info/WHEEL +4 -0
  48. librus_python_api-1.0.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,454 @@
1
+ """Original bounded parser for inline semester grades and school averages."""
2
+
3
+ import re
4
+ from dataclasses import dataclass, field
5
+ from typing import Literal, cast
6
+
7
+ from lxml import html
8
+
9
+ from librus_python_api import markup
10
+ from librus_python_api.config import (
11
+ GRADE_AVERAGE_HEADERS,
12
+ GRADE_BODY_PREFIX_COLUMNS,
13
+ GRADE_CURRENT_HEADER,
14
+ GRADE_EMPTY_MARKERS,
15
+ GRADE_MAX_METADATA_LENGTH,
16
+ GRADE_MAX_RECORDS,
17
+ GRADE_MAX_SUBJECTS,
18
+ GRADE_PERIOD_HEADERS,
19
+ GRADE_PREDICTED_ANNUAL_HEADER,
20
+ GRADE_PREDICTED_PERIOD_HEADERS,
21
+ GRADE_PUBLICATION_DATE_LABEL,
22
+ GRADE_PUBLICATION_PERIOD_LABELS,
23
+ GRADE_PUBLICATION_TEACHER_LABEL,
24
+ )
25
+ from librus_python_api.exceptions import ErrorKind, LibrusError
26
+ from librus_python_api.grade_parsers import _locate, _subject_rows, _summary
27
+ from librus_python_api.models import (
28
+ Availability,
29
+ DescriptiveGrade,
30
+ DescriptiveGradeSummary,
31
+ GradeKind,
32
+ GradeRecords,
33
+ GradeSummaryValue,
34
+ NumericGrade,
35
+ SchoolAverage,
36
+ )
37
+ from librus_python_api.parsers import parse_page
38
+
39
+
40
+ def _label(title: str) -> str:
41
+ return " ".join(re.split(r"<br\s*/?>", title, flags=re.I)[0].split())
42
+
43
+
44
+ def _layout(table: html.HtmlElement) -> tuple[list[int], dict[int, int]]:
45
+ current: list[int] = []
46
+ averages: dict[int, int] = {}
47
+ for row in markup.rows(table):
48
+ if next(row.iterancestors("thead"), None) is None:
49
+ continue
50
+ cells = markup.cells(row)
51
+ if not any(markup.text(c) == GRADE_CURRENT_HEADER for c in cells):
52
+ continue
53
+ if current:
54
+ raise LibrusError(ErrorKind.PARSE)
55
+ position = GRADE_BODY_PREFIX_COLUMNS
56
+ for cell in cells:
57
+ if markup.text(cell) == GRADE_CURRENT_HEADER:
58
+ if markup.colspan(cell) != 1:
59
+ raise LibrusError(ErrorKind.PARSE)
60
+ current.append(position)
61
+ period = GRADE_AVERAGE_HEADERS.get(_label(cell.get("title", "")))
62
+ if period is not None:
63
+ if period in averages or markup.colspan(cell) != 1:
64
+ raise LibrusError(ErrorKind.PARSE)
65
+ averages[period] = position
66
+ position += markup.colspan(cell)
67
+ if len(current) != 2:
68
+ raise LibrusError(ErrorKind.PARSE)
69
+ return current, averages
70
+
71
+
72
+ def _numeric(
73
+ element: html.HtmlElement,
74
+ subject: str,
75
+ semester: Literal[0, 1, 2],
76
+ kind: GradeKind = GradeKind.CURRENT,
77
+ ) -> NumericGrade:
78
+ metadata = markup.tooltip_fields(element)
79
+ counts = metadata.get("Licz do średniej")
80
+ if counts is not None and counts.casefold() not in ("tak", "nie"):
81
+ raise LibrusError(ErrorKind.PARSE)
82
+ weight = metadata.get("Waga")
83
+ if weight is not None and not re.fullmatch(r"[0-9]{1,4}", weight):
84
+ raise LibrusError(ErrorKind.PARSE)
85
+ raw = markup.text(element)
86
+ if not raw:
87
+ raise LibrusError(ErrorKind.PARSE)
88
+ href = element.get("href")
89
+ if href is not None and len(href) > GRADE_MAX_METADATA_LENGTH:
90
+ raise LibrusError(ErrorKind.LIMIT)
91
+ return NumericGrade(
92
+ subject,
93
+ raw,
94
+ markup.civil_date(metadata.get("Data")),
95
+ semester,
96
+ None if counts is None else counts.casefold() == "tak",
97
+ None if weight is None else int(weight),
98
+ metadata.get("Kategoria"),
99
+ metadata.get("Nauczyciel"),
100
+ metadata.get("Komentarz"),
101
+ href,
102
+ tuple(metadata.items()),
103
+ kind,
104
+ )
105
+
106
+
107
+ def _descriptive(
108
+ box: html.HtmlElement,
109
+ element: html.HtmlElement,
110
+ subject: str,
111
+ semester: Literal[0, 1, 2],
112
+ kind: GradeKind = GradeKind.CURRENT,
113
+ ) -> DescriptiveGrade:
114
+ metadata = markup.tooltip_fields(box)
115
+ raw = markup.text(element)
116
+ if not raw:
117
+ raise LibrusError(ErrorKind.PARSE)
118
+ return DescriptiveGrade(
119
+ subject,
120
+ raw,
121
+ markup.civil_date(metadata.get("Data")),
122
+ semester,
123
+ metadata.get("Nauczyciel"),
124
+ metadata.get("Komentarz"),
125
+ tuple(metadata.items()),
126
+ kind,
127
+ )
128
+
129
+
130
+ @dataclass(slots=True)
131
+ class _Collection:
132
+ numeric: list[NumericGrade] = field(default_factory=list, repr=False)
133
+ descriptive: list[DescriptiveGrade] = field(default_factory=list, repr=False)
134
+ descriptive_summaries: list[DescriptiveGradeSummary] = field(
135
+ default_factory=list, repr=False
136
+ )
137
+ averages: list[SchoolAverage] = field(default_factory=list, repr=False)
138
+ subjects: set[tuple[bool, str]] = field(default_factory=set, repr=False)
139
+ consumed: set[html.HtmlElement] = field(default_factory=set, repr=False)
140
+ publication_rows: set[html.HtmlElement] = field(default_factory=set, repr=False)
141
+
142
+ def check_size(self) -> None:
143
+ if (
144
+ len(self.numeric) + len(self.descriptive) + len(self.descriptive_summaries)
145
+ > GRADE_MAX_RECORDS
146
+ ):
147
+ raise LibrusError(ErrorKind.LIMIT)
148
+
149
+ def subject(self, value: str, *, descriptive: bool = False) -> None:
150
+ key = (descriptive, value)
151
+ if not value or key in self.subjects:
152
+ raise LibrusError(ErrorKind.PARSE)
153
+ self.subjects.add(key)
154
+ if len(self.subjects) > GRADE_MAX_SUBJECTS:
155
+ raise LibrusError(ErrorKind.LIMIT)
156
+
157
+
158
+ def _boxes(cell: html.HtmlElement) -> list[html.HtmlElement]:
159
+ return [
160
+ node
161
+ for node in cell.iter("span")
162
+ if "grade-box" in node.get("class", "").split()
163
+ ]
164
+
165
+
166
+ def _entries(
167
+ cell: html.HtmlElement,
168
+ subject: str,
169
+ period: Literal[0, 1, 2],
170
+ collection: _Collection,
171
+ *,
172
+ kind: GradeKind = GradeKind.CURRENT,
173
+ descriptive: bool = False,
174
+ require_empty_marker: bool = True,
175
+ ) -> None:
176
+ boxes = _boxes(cell)
177
+ if (
178
+ not boxes
179
+ and require_empty_marker
180
+ and markup.text(cell) not in GRADE_EMPTY_MARKERS
181
+ ):
182
+ if (
183
+ not descriptive
184
+ or period == 0
185
+ or any(isinstance(node.tag, str) for node in cell)
186
+ ):
187
+ raise LibrusError(ErrorKind.PARSE)
188
+ raw = markup.text(cell)
189
+ if len(raw) > GRADE_MAX_METADATA_LENGTH:
190
+ raise LibrusError(ErrorKind.LIMIT)
191
+ collection.descriptive_summaries.append(
192
+ DescriptiveGradeSummary(subject, period, raw)
193
+ )
194
+ collection.check_size()
195
+ for box in boxes:
196
+ anchors = [e for e in box if e.tag == "a"]
197
+ descriptions = [
198
+ e for e in box if e.tag == "span" and "ocena" in e.get("class", "").split()
199
+ ]
200
+ if len(anchors) + len(descriptions) != 1:
201
+ raise LibrusError(ErrorKind.PARSE)
202
+ if anchors:
203
+ record = _numeric(anchors[0], subject, period, kind)
204
+ if descriptive:
205
+ href = record.href
206
+ if href is not None and re.sub(
207
+ r"[\x00-\x20]", "", href
208
+ ).casefold().startswith(("javascript:", "data:", "vbscript:")):
209
+ href = None
210
+ collection.descriptive.append(
211
+ DescriptiveGrade(
212
+ record.subject,
213
+ record.raw,
214
+ record.day,
215
+ record.semester,
216
+ record.teacher,
217
+ record.comment,
218
+ record.metadata,
219
+ kind,
220
+ href,
221
+ )
222
+ )
223
+ else:
224
+ collection.numeric.append(record)
225
+ else:
226
+ collection.descriptive.append(
227
+ _descriptive(box, descriptions[0], subject, period, kind)
228
+ )
229
+ collection.consumed.add(box)
230
+ collection.check_size()
231
+
232
+
233
+ def _descriptive_row(cells: list[html.HtmlElement]) -> bool:
234
+ return (
235
+ len(cells) in (4, 6)
236
+ and {"micro", "center", "screen-only"} <= set(cells[0].get("class", "").split())
237
+ and all(markup.colspan(cell) == 1 for cell in cells)
238
+ )
239
+
240
+
241
+ def _read_descriptive_row(
242
+ cells: list[html.HtmlElement], collection: _Collection
243
+ ) -> None:
244
+ subject = markup.text(cells[1])
245
+ collection.subject(subject, descriptive=True)
246
+ for period in (1, 2):
247
+ _entries(cells[period + 1], subject, period, collection, descriptive=True)
248
+ if len(cells) == 6:
249
+ for cell in cells[4:]:
250
+ if _boxes(cell):
251
+ # These columns have no established semester semantics.
252
+ raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
253
+
254
+
255
+ def _period_columns(table: html.HtmlElement) -> dict[tuple[int, GradeKind], int]:
256
+ result: dict[tuple[int, GradeKind], int] = {}
257
+ for row in markup.rows(table):
258
+ cells = markup.cells(row)
259
+ if next(row.iterancestors("thead"), None) is None or not any(
260
+ markup.text(c) == GRADE_CURRENT_HEADER for c in cells
261
+ ):
262
+ continue
263
+ position = GRADE_BODY_PREFIX_COLUMNS
264
+ for cell in cells:
265
+ label = _label(cell.get("title", ""))
266
+ period = GRADE_PERIOD_HEADERS.get(label)
267
+ kind = GradeKind.ANNUAL if period == 0 else GradeKind.PERIOD
268
+ if label == GRADE_PREDICTED_ANNUAL_HEADER:
269
+ period, kind = 0, GradeKind.PREDICTED_ANNUAL
270
+ elif label in GRADE_PREDICTED_PERIOD_HEADERS:
271
+ period, kind = (
272
+ GRADE_PREDICTED_PERIOD_HEADERS[label],
273
+ GradeKind.PREDICTED_PERIOD,
274
+ )
275
+ if period is not None:
276
+ if (period, kind) in result or markup.colspan(cell) != 1:
277
+ raise LibrusError(ErrorKind.PARSE)
278
+ result[(period, kind)] = position
279
+ position += markup.colspan(cell)
280
+ return result
281
+
282
+
283
+ def _read_numeric_table(table: html.HtmlElement, collection: _Collection) -> None:
284
+ # Preserve the established full-row and summary alignment validation.
285
+ _, fields, width = _locate(table)
286
+ current, means = _layout(table)
287
+ periods = _period_columns(table)
288
+ for cells in _subject_rows(table, width):
289
+ if cells[0].getparent() in collection.publication_rows:
290
+ continue
291
+ if _descriptive_row(cells):
292
+ _read_descriptive_row(cells, collection)
293
+ continue
294
+ subject = _summary(cells, fields, width).subject
295
+ collection.subject(subject)
296
+ columns = [cell for cell in cells for _ in range(markup.colspan(cell))]
297
+ for period in (1, 2, 0):
298
+ index = means.get(period)
299
+ value = (
300
+ GradeSummaryValue(Availability.UNAVAILABLE, None)
301
+ if index is None or markup.colspan(columns[index]) != 1
302
+ else GradeSummaryValue(
303
+ Availability.AVAILABLE, markup.text(columns[index])
304
+ )
305
+ )
306
+ collection.averages.append(SchoolAverage(subject, period, value))
307
+ if columns[current[0]] is columns[current[1]] and _boxes(columns[current[0]]):
308
+ raise LibrusError(ErrorKind.PARSE)
309
+ seen: set[html.HtmlElement] = set()
310
+ for period, index in enumerate(current, start=1):
311
+ if columns[index] not in seen:
312
+ _entries(
313
+ columns[index], subject, cast(Literal[1, 2], period), collection
314
+ )
315
+ seen.add(columns[index])
316
+ for (period, kind), index in periods.items():
317
+ cell = columns[index]
318
+ if cell in seen and _boxes(cell):
319
+ raise LibrusError(ErrorKind.PARSE)
320
+ _entries(
321
+ cell,
322
+ subject,
323
+ cast(Literal[0, 1, 2], period),
324
+ collection,
325
+ kind=kind,
326
+ require_empty_marker=False,
327
+ )
328
+
329
+
330
+ def _read_publications(table: html.HtmlElement, collection: _Collection) -> None:
331
+ rows = list(markup.rows(table))
332
+ for index, row in enumerate(rows):
333
+ headers = [
334
+ cell
335
+ for cell in markup.cells(row)
336
+ if cell.tag == "th"
337
+ and list(cell.iter("strong"))
338
+ and GRADE_PUBLICATION_DATE_LABEL in cell.text_content()
339
+ ]
340
+ if not headers:
341
+ continue
342
+ if len(headers) != 1 or index + 1 == len(rows):
343
+ raise LibrusError(ErrorKind.PARSE)
344
+ header = headers[0]
345
+ titles = list(header.iter("strong"))
346
+ if len(titles) != 1:
347
+ raise LibrusError(ErrorKind.PARSE)
348
+ title = markup.text(titles[0])
349
+ info = markup.text(header)
350
+ date_info = info.partition(GRADE_PUBLICATION_DATE_LABEL)[2].strip()
351
+ day = markup.civil_date(date_info[:10])
352
+ if len(date_info) > 10 and date_info[10] not in (" ", ",", ")"):
353
+ raise LibrusError(ErrorKind.PARSE)
354
+ teacher_info = info.partition(GRADE_PUBLICATION_TEACHER_LABEL)[2].strip()
355
+ teacher = teacher_info.rsplit(")", 1)[0].strip() if teacher_info else None
356
+ matching = {
357
+ period
358
+ for label, period in GRADE_PUBLICATION_PERIOD_LABELS.items()
359
+ if re.search(r"(?<!\w)" + re.escape(label) + r"(?!\w)", title)
360
+ }
361
+ if len(matching) > 1:
362
+ raise LibrusError(ErrorKind.PARSE)
363
+ paragraphs = list(rows[index + 1].iter("p"))
364
+ raw = "\n".join(markup.text(paragraph) for paragraph in paragraphs).strip()
365
+ if not title or not raw:
366
+ raise LibrusError(ErrorKind.PARSE)
367
+ if len(raw) > GRADE_MAX_METADATA_LENGTH:
368
+ raise LibrusError(ErrorKind.LIMIT)
369
+ metadata = (("Data", date_info[:10]),)
370
+ collection.descriptive.append(
371
+ DescriptiveGrade(
372
+ title,
373
+ raw,
374
+ day,
375
+ cast(Literal[1, 2] | None, next(iter(matching), None)),
376
+ teacher,
377
+ None,
378
+ metadata,
379
+ GradeKind.PUBLICATION,
380
+ )
381
+ )
382
+ collection.consumed.add(header)
383
+ collection.publication_rows.update((row, rows[index + 1]))
384
+ collection.check_size()
385
+
386
+
387
+ def parse_grade_records(body: bytes) -> GradeRecords:
388
+ document = parse_page(body)
389
+ collection = _Collection()
390
+ numeric_tables = []
391
+ tables = list(document.iter("table"))
392
+ for table in tables:
393
+ if not {"decorated", "stretch"} <= set(table.get("class", "").split()):
394
+ continue
395
+ if any(
396
+ next(row.iterancestors("thead"), None) is not None
397
+ and any(markup.text(c) == GRADE_CURRENT_HEADER for c in markup.cells(row))
398
+ for row in markup.rows(table)
399
+ ):
400
+ numeric_tables.append(table)
401
+ if len(numeric_tables) > 1:
402
+ raise LibrusError(ErrorKind.PARSE)
403
+ for table in tables:
404
+ _read_publications(table, collection)
405
+ if table in numeric_tables:
406
+ _read_numeric_table(table, collection)
407
+ else:
408
+ for row in markup.rows(table):
409
+ cells = markup.cells(row)
410
+ if _descriptive_row(cells):
411
+ _read_descriptive_row(cells, collection)
412
+ elif (
413
+ cells
414
+ and cells[0].tag == "td"
415
+ and {"micro", "center", "screen-only"}
416
+ <= set(cells[0].get("class", "").split())
417
+ ):
418
+ raise LibrusError(ErrorKind.PARSE)
419
+ for element in document.iter():
420
+ if not isinstance(element.tag, str) or element in collection.consumed:
421
+ continue
422
+ dated_box = "grade-box" in element.get("class", "").split() and any(
423
+ "Data:" in (node.get("title") or "") for node in element.iter()
424
+ )
425
+ if dated_box:
426
+ if any(
427
+ parent in numeric_tables for parent in element.iterancestors("table")
428
+ ):
429
+ # Expanded full-row detail mirrors were intentionally excluded.
430
+ cell = next(element.iterancestors("td", "th"), None)
431
+ if (
432
+ cell is not None
433
+ and next(cell.iterancestors("table"), None) not in numeric_tables
434
+ ):
435
+ continue
436
+ raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
437
+ if not collection.subjects and not collection.descriptive:
438
+ raise LibrusError(ErrorKind.PARSE)
439
+ for descriptive, subject in sorted(collection.subjects):
440
+ if descriptive and (False, subject) not in collection.subjects:
441
+ collection.averages.extend(
442
+ SchoolAverage(
443
+ subject,
444
+ period,
445
+ GradeSummaryValue(Availability.UNAVAILABLE, None),
446
+ )
447
+ for period in (1, 2, 0)
448
+ )
449
+ return GradeRecords(
450
+ tuple(collection.numeric),
451
+ tuple(collection.descriptive),
452
+ tuple(collection.averages),
453
+ tuple(collection.descriptive_summaries),
454
+ )
@@ -0,0 +1,41 @@
1
+ """Bounded aggregation by upstream identity, without guessed date-filter semantics."""
2
+
3
+ from librus_python_api.config import SCHOOL_MAX_TOTAL_TEXT_LENGTH
4
+ from librus_python_api.exceptions import ErrorKind, LibrusError
5
+ from librus_python_api.models import HomeworkItem, SchoolReference
6
+
7
+
8
+ class HomeworkAccumulator:
9
+ def __init__(self, max_items: int) -> None:
10
+ self.items: list[HomeworkItem] = []
11
+ self._references: dict[SchoolReference, HomeworkItem] = {}
12
+ self._max_items = max_items
13
+ self._text_length = 0
14
+
15
+ def extend(self, items: tuple[HomeworkItem, ...]) -> None:
16
+ for item in items:
17
+ if item.reference is not None:
18
+ previous = self._references.get(item.reference)
19
+ if previous is not None:
20
+ if previous != item:
21
+ # The selection is not a server-side snapshot. Never
22
+ # silently prefer a stale or changed duplicate.
23
+ raise LibrusError(ErrorKind.PARSE)
24
+ continue
25
+ self._references[item.reference] = item
26
+ self._text_length += sum(
27
+ len(value)
28
+ for value in (
29
+ item.subject,
30
+ item.teacher,
31
+ item.topic,
32
+ item.category,
33
+ item.submission_status or "",
34
+ )
35
+ )
36
+ if (
37
+ len(self.items) >= self._max_items
38
+ or self._text_length > SCHOOL_MAX_TOTAL_TEXT_LENGTH
39
+ ):
40
+ raise LibrusError(ErrorKind.LIMIT)
41
+ self.items.append(item)
@@ -0,0 +1,24 @@
1
+ """Joined resource ownership even when callers cancel repeatedly."""
2
+
3
+ import asyncio
4
+
5
+
6
+ async def join_owned[T](future: asyncio.Future[T]) -> bool:
7
+ """Join without forwarding cancellation; return whether joining was canceled.
8
+
9
+ Cleanup must finish before resource/admission slots can be reused. This helper
10
+ is only for already-owned work, never for ordinary uncancelable operations.
11
+ """
12
+ interrupted = False
13
+ while not future.done():
14
+ try:
15
+ # wait() never forwards cancellation to owned work and has no
16
+ # abandoned shield wrapper that can log an already-handled error.
17
+ await asyncio.wait((future,))
18
+ except asyncio.CancelledError:
19
+ interrupted = True
20
+ except Exception:
21
+ break
22
+ if not future.cancelled():
23
+ future.exception()
24
+ return interrupted
@@ -0,0 +1,147 @@
1
+ """Bounded HTML reading helpers shared by every Synergia page parser."""
2
+
3
+ import re
4
+ from collections.abc import Iterator
5
+ from datetime import date
6
+ from urllib.parse import urlsplit
7
+
8
+ from lxml import etree, html
9
+
10
+ from librus_python_api.config import (
11
+ ATTENDANCE_DETAIL_PATH_PREFIX,
12
+ GRADE_MAX_COLUMNS,
13
+ GRADE_MAX_METADATA_FIELDS,
14
+ GRADE_MAX_METADATA_LENGTH,
15
+ UPSTREAM_ORIGINS,
16
+ WEEKDAY_LABELS,
17
+ )
18
+ from librus_python_api.exceptions import ErrorKind, LibrusError
19
+ from librus_python_api.parsers import parse_html_document
20
+
21
+ FIELD_LENGTH = 1024
22
+ _ACTIVE_CONTENT = frozenset({"script", "style", "iframe", "object", "embed", "form"})
23
+ _BLOCKS = frozenset({"br", "p", "div", "li"})
24
+
25
+
26
+ def text(
27
+ element: html.HtmlElement, limit: int = FIELD_LENGTH, *, multiline: bool = False
28
+ ) -> str:
29
+ """Rendered text with block boundaries as line breaks; never truncated.
30
+
31
+ Active content inside a data cell is an unsupported layout, not text.
32
+ """
33
+ parts: list[str] = []
34
+ for event, node in etree.iterwalk(element, events=("start", "end", "comment")):
35
+ if node.tag in _ACTIVE_CONTENT:
36
+ raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
37
+ if node.tag in _BLOCKS:
38
+ parts.append("\n")
39
+ if event == "start" and isinstance(node.tag, str) and node.text:
40
+ parts.append(re.sub(r"\s+", " ", node.text))
41
+ if event in ("end", "comment") and node is not element and node.tail:
42
+ parts.append(re.sub(r"\s+", " ", node.tail))
43
+ rendered = "".join(parts)
44
+ value = (
45
+ "\n".join(
46
+ line for raw in rendered.splitlines() if (line := " ".join(raw.split()))
47
+ )
48
+ if multiline
49
+ else " ".join(rendered.split())
50
+ )
51
+ if len(value) > limit:
52
+ raise LibrusError(ErrorKind.LIMIT)
53
+ return value
54
+
55
+
56
+ def cells(row: html.HtmlElement) -> list[html.HtmlElement]:
57
+ return [item for item in row if item.tag in ("td", "th")]
58
+
59
+
60
+ def rows(table: html.HtmlElement) -> Iterator[html.HtmlElement]:
61
+ """Rows of this table only, never rows of a nested table."""
62
+ for row in table.iter("tr"):
63
+ if next(row.iterancestors("table"), None) is table:
64
+ yield row
65
+
66
+
67
+ def in_header(row: html.HtmlElement) -> bool:
68
+ return next(row.iterancestors("thead"), None) is not None
69
+
70
+
71
+ def colspan(cell: html.HtmlElement) -> int:
72
+ value = cell.get("colspan", "1")
73
+ if not re.fullmatch(r"[1-9][0-9]{0,2}", value):
74
+ raise LibrusError(ErrorKind.PARSE)
75
+ span = int(value)
76
+ if span > GRADE_MAX_COLUMNS:
77
+ raise LibrusError(ErrorKind.LIMIT)
78
+ if cell.get("rowspan", "1") != "1":
79
+ raise LibrusError(ErrorKind.PARSE)
80
+ return span
81
+
82
+
83
+ def tooltip_fields(element: html.HtmlElement) -> dict[str, str]:
84
+ """Parse a 'Label: value<br>' title attribute into unique labelled fields."""
85
+ title = element.get("title", "")
86
+ if len(title) > GRADE_MAX_METADATA_LENGTH:
87
+ raise LibrusError(ErrorKind.LIMIT)
88
+ result: dict[str, str] = {}
89
+ chunks = re.split(r"<br\s*/?>|\n", title, flags=re.I)
90
+ if len(chunks) > GRADE_MAX_METADATA_FIELDS:
91
+ raise LibrusError(ErrorKind.LIMIT)
92
+ for chunk in chunks:
93
+ if not chunk.strip():
94
+ continue
95
+ # Tooltips are HTML inside an attribute; parse markup only where present.
96
+ rendered = (
97
+ text(parse_html_document(chunk.encode()))
98
+ if "<" in chunk
99
+ else " ".join(chunk.split())
100
+ )
101
+ key, separator, value = rendered.partition(":")
102
+ key, value = key.strip(), value.strip()
103
+ if not separator or not key or key in result:
104
+ raise LibrusError(ErrorKind.PARSE)
105
+ result[key] = value
106
+ return result
107
+
108
+
109
+ def civil_date(value: str | None) -> date:
110
+ """An ISO date, optionally followed by its weekday label in parentheses."""
111
+ match = re.fullmatch(r"([0-9]{4}-[0-9]{2}-[0-9]{2})(?: \(([^()]+)\))?", value or "")
112
+ if match is None or (match[2] is not None and match[2] not in WEEKDAY_LABELS):
113
+ raise LibrusError(ErrorKind.PARSE)
114
+ try:
115
+ return date.fromisoformat(match[1])
116
+ except ValueError:
117
+ pass
118
+ raise LibrusError(ErrorKind.PARSE)
119
+
120
+
121
+ def attendance_detail_id(element: html.HtmlElement) -> str | None:
122
+ """The numeric ID of an inert attendance-detail popup link, if any."""
123
+ script = element.get("onclick", "")
124
+ if len(script) > GRADE_MAX_METADATA_LENGTH:
125
+ raise LibrusError(ErrorKind.LIMIT)
126
+ match = re.fullmatch(
127
+ r"\s*otworz_w_nowym_oknie\(\s*(['\"])([^'\"\\\s]+)\1"
128
+ r"""(?:\s*,\s*(?:'[^'\\\r\n]*'|"[^"\\\r\n]*"|[0-9]{1,6})){1,4}"""
129
+ r"\s*\)\s*;?\s*",
130
+ script,
131
+ )
132
+ if match is None:
133
+ return None
134
+ try:
135
+ target = urlsplit(match[2])
136
+ except ValueError:
137
+ return None
138
+ if target.query or target.fragment or target.username or target.password:
139
+ return None
140
+ if target.scheme or target.netloc:
141
+ expected = urlsplit(UPSTREAM_ORIGINS["synergia"])
142
+ if (target.scheme, target.netloc) != (expected.scheme, expected.netloc):
143
+ return None
144
+ if not target.path.startswith(ATTENDANCE_DETAIL_PATH_PREFIX):
145
+ return None
146
+ identifier = target.path[len(ATTENDANCE_DETAIL_PATH_PREFIX) :]
147
+ return identifier if re.fullmatch(r"[0-9]{1,64}", identifier) else None