librus-python-api 1.0.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- librus_python_api/__init__.py +243 -0
- librus_python_api/_notification_codec.py +310 -0
- librus_python_api/_storage.py +367 -0
- librus_python_api/announcements.py +159 -0
- librus_python_api/attachment_routes.py +114 -0
- librus_python_api/attachments.py +297 -0
- librus_python_api/attendance.py +182 -0
- librus_python_api/attendance_frequency.py +112 -0
- librus_python_api/budget.py +79 -0
- librus_python_api/checkpoint.py +61 -0
- librus_python_api/completed_lessons.py +216 -0
- librus_python_api/config.py +1410 -0
- librus_python_api/detail_fields.py +50 -0
- librus_python_api/diagnostics.py +25 -0
- librus_python_api/exceptions.py +172 -0
- librus_python_api/files.py +156 -0
- librus_python_api/grade_parsers.py +169 -0
- librus_python_api/grade_records.py +454 -0
- librus_python_api/homework_range.py +41 -0
- librus_python_api/lifecycle.py +24 -0
- librus_python_api/markup.py +147 -0
- librus_python_api/message_content.py +230 -0
- librus_python_api/messages.py +288 -0
- librus_python_api/models.py +1160 -0
- librus_python_api/modern_body.py +75 -0
- librus_python_api/modern_mailbox.py +459 -0
- librus_python_api/modern_messages.py +276 -0
- librus_python_api/notification_models.py +67 -0
- librus_python_api/notification_persistence.py +1044 -0
- librus_python_api/notification_workflow.py +303 -0
- librus_python_api/notifications.py +216 -0
- librus_python_api/parsers.py +232 -0
- librus_python_api/parsing.py +49 -0
- librus_python_api/persistence.py +405 -0
- librus_python_api/py.typed +0 -0
- librus_python_api/recipients.py +271 -0
- librus_python_api/scheduler.py +287 -0
- librus_python_api/school_reads.py +400 -0
- librus_python_api/sending.py +125 -0
- librus_python_api/service.py +2285 -0
- librus_python_api/timetable.py +261 -0
- librus_python_api/transport.py +956 -0
- librus_python_api-1.0.0rc1.dist-info/METADATA +254 -0
- librus_python_api-1.0.0rc1.dist-info/RECORD +46 -0
- librus_python_api-1.0.0rc1.dist-info/WHEEL +4 -0
- librus_python_api-1.0.0rc1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Normalize bounded detail records using only established per-family labels."""
|
|
2
|
+
|
|
3
|
+
from typing import Literal
|
|
4
|
+
|
|
5
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
6
|
+
from librus_python_api.models import DetailField, DetailFieldKey
|
|
7
|
+
|
|
8
|
+
# School labels are recorded in contracts/school-reads.md. Attendance labels
|
|
9
|
+
# are the narrowly established date/topic fields, not inferred tooltip fields.
|
|
10
|
+
_KEYS: dict[str, dict[str, DetailFieldKey]] = {
|
|
11
|
+
"agenda": {
|
|
12
|
+
"data": "date",
|
|
13
|
+
"nr lekcji": "lesson_number",
|
|
14
|
+
"nauczyciel": "teacher",
|
|
15
|
+
"rodzaj": "category",
|
|
16
|
+
"przedmiot": "subject",
|
|
17
|
+
"sala": "room",
|
|
18
|
+
"opis": "description",
|
|
19
|
+
"data dodania": "published_at",
|
|
20
|
+
},
|
|
21
|
+
"homework": {
|
|
22
|
+
"zajęcia edukacyjne": "subject",
|
|
23
|
+
"temat": "topic",
|
|
24
|
+
"kategoria": "category",
|
|
25
|
+
"data udostępnienia": "published_at",
|
|
26
|
+
"termin wykonania": "due_at",
|
|
27
|
+
"treść": "content",
|
|
28
|
+
},
|
|
29
|
+
"attendance": {"data": "date", "temat zajęć": "topic"},
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def normalize_detail_fields(
|
|
34
|
+
fields: tuple[tuple[str, str], ...],
|
|
35
|
+
kind: Literal["agenda", "homework", "attendance"],
|
|
36
|
+
) -> tuple[DetailField, ...]:
|
|
37
|
+
"""Run after bounded HTML parsing, before publishing or caching a result."""
|
|
38
|
+
labels: set[str] = set()
|
|
39
|
+
keys: set[DetailFieldKey] = set()
|
|
40
|
+
result = []
|
|
41
|
+
for label, value in fields:
|
|
42
|
+
canonical = " ".join(label.split()).rstrip(":").strip().casefold()
|
|
43
|
+
key = _KEYS[kind].get(canonical)
|
|
44
|
+
if not canonical or canonical in labels or (key is not None and key in keys):
|
|
45
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
46
|
+
labels.add(canonical)
|
|
47
|
+
if key is not None:
|
|
48
|
+
keys.add(key)
|
|
49
|
+
result.append(DetailField(key, label, value))
|
|
50
|
+
return tuple(result)
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Opt-in allowlisted structured diagnostics; no global logging configuration."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
from typing import Protocol
|
|
5
|
+
|
|
6
|
+
from librus_python_api.models import DiagnosticEvent
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class DiagnosticSink(Protocol):
|
|
10
|
+
def __call__(self, event: DiagnosticEvent, /) -> None: ...
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def logging_sink(event: DiagnosticEvent) -> None:
|
|
14
|
+
"""Emit allowlisted fields via stdlib logging; the application owns handlers."""
|
|
15
|
+
logging.getLogger("librus_python_api").info(
|
|
16
|
+
"Librus operation completed",
|
|
17
|
+
extra={
|
|
18
|
+
"component": "librus_python_api",
|
|
19
|
+
"operation": event.operation,
|
|
20
|
+
"outcome": event.outcome,
|
|
21
|
+
"elapsed_seconds": event.elapsed_seconds,
|
|
22
|
+
"budget_requests_dispatched": event.budget_requests_dispatched,
|
|
23
|
+
"budget_response_bytes": event.budget_response_bytes,
|
|
24
|
+
},
|
|
25
|
+
)
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""Canonical typed exceptions and the closed, redacted failure factory."""
|
|
2
|
+
|
|
3
|
+
from enum import StrEnum
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"AccessDeniedError",
|
|
7
|
+
"AccountActionRequiredError",
|
|
8
|
+
"ClosedError",
|
|
9
|
+
"CheckpointError",
|
|
10
|
+
"ConnectionError",
|
|
11
|
+
"CredentialsRejectedError",
|
|
12
|
+
"ErrorKind",
|
|
13
|
+
"InvalidInputError",
|
|
14
|
+
"LibrusError",
|
|
15
|
+
"LimitError",
|
|
16
|
+
"MaintenanceError",
|
|
17
|
+
"OperationTimeoutError",
|
|
18
|
+
"ParseError",
|
|
19
|
+
"SessionExpiredError",
|
|
20
|
+
"StorageError",
|
|
21
|
+
"StaleCursorError",
|
|
22
|
+
"ThrottledError",
|
|
23
|
+
"UnknownDeliveryError",
|
|
24
|
+
"UnsupportedCapabilityError",
|
|
25
|
+
"ViewDisabledError",
|
|
26
|
+
"error_for",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class ErrorKind(StrEnum):
|
|
31
|
+
CHECKPOINT = "checkpoint"
|
|
32
|
+
INVALID_INPUT = "invalid_input"
|
|
33
|
+
CREDENTIALS_REJECTED = "credentials_rejected"
|
|
34
|
+
ACCOUNT_ACTION_REQUIRED = "account_action_required"
|
|
35
|
+
SESSION_EXPIRED = "session_expired"
|
|
36
|
+
ACCESS_DENIED = "access_denied"
|
|
37
|
+
UNSUPPORTED_CAPABILITY = "unsupported_capability"
|
|
38
|
+
VIEW_DISABLED = "view_disabled"
|
|
39
|
+
THROTTLED = "throttled"
|
|
40
|
+
MAINTENANCE = "maintenance"
|
|
41
|
+
CONNECTION = "connection"
|
|
42
|
+
TIMEOUT = "timeout"
|
|
43
|
+
LIMIT = "limit"
|
|
44
|
+
PARSE = "parse"
|
|
45
|
+
STALE_CURSOR = "stale_cursor"
|
|
46
|
+
UNKNOWN_DELIVERY = "unknown_delivery"
|
|
47
|
+
CLOSED = "closed"
|
|
48
|
+
STORAGE = "storage"
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class LibrusError(Exception):
|
|
52
|
+
"""A failure with no arbitrary upstream message, URL, or response attachment.
|
|
53
|
+
|
|
54
|
+
Future transport adapters must suppress raw causes at the public boundary.
|
|
55
|
+
This type does not sanitize caller-added exception notes or chained errors.
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
def __new__(cls, kind: ErrorKind) -> "LibrusError":
|
|
59
|
+
if not isinstance(kind, ErrorKind):
|
|
60
|
+
raise TypeError("kind must be an ErrorKind")
|
|
61
|
+
# Preserve the foundation constructor while routing all failures through
|
|
62
|
+
# the same closed factory registry. Callers can catch specific subclasses.
|
|
63
|
+
concrete = _ERROR_TYPES[kind] if cls is LibrusError else cls
|
|
64
|
+
if concrete is not _ERROR_TYPES[kind]:
|
|
65
|
+
raise TypeError("Exception class and kind must agree")
|
|
66
|
+
return super().__new__(concrete)
|
|
67
|
+
|
|
68
|
+
def __init__(self, kind: ErrorKind) -> None:
|
|
69
|
+
if not isinstance(kind, ErrorKind):
|
|
70
|
+
raise TypeError("kind must be an ErrorKind")
|
|
71
|
+
self.kind = kind
|
|
72
|
+
super().__init__(kind.value)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class InvalidInputError(LibrusError):
|
|
76
|
+
pass
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class CredentialsRejectedError(LibrusError):
|
|
80
|
+
pass
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
class AccountActionRequiredError(LibrusError):
|
|
84
|
+
pass
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
class SessionExpiredError(LibrusError):
|
|
88
|
+
# Library-private provenance only; never attach a URL, body or credentials.
|
|
89
|
+
_messages_origin = False
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class AccessDeniedError(LibrusError):
|
|
93
|
+
pass
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class UnsupportedCapabilityError(LibrusError):
|
|
97
|
+
pass
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class ViewDisabledError(LibrusError):
|
|
101
|
+
"""The school administrator has switched this Synergia view off."""
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class ThrottledError(LibrusError):
|
|
105
|
+
pass
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class MaintenanceError(LibrusError):
|
|
109
|
+
pass
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class ConnectionError(LibrusError):
|
|
113
|
+
pass
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class OperationTimeoutError(LibrusError):
|
|
117
|
+
pass
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class LimitError(LibrusError):
|
|
121
|
+
pass
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class ParseError(LibrusError):
|
|
125
|
+
pass
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
class StaleCursorError(LibrusError):
|
|
129
|
+
"""Previously valid continuation no longer matches the current page sequence."""
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class UnknownDeliveryError(LibrusError):
|
|
133
|
+
pass
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class ClosedError(LibrusError):
|
|
137
|
+
pass
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
class CheckpointError(LibrusError):
|
|
141
|
+
"""Durable acknowledgement is unknown; never retry the callback or consume."""
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
class StorageError(LibrusError):
|
|
145
|
+
"""Private durable storage failed; a write may already have committed."""
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
_ERROR_TYPES: dict[ErrorKind, type[LibrusError]] = {
|
|
149
|
+
ErrorKind.CHECKPOINT: CheckpointError,
|
|
150
|
+
ErrorKind.INVALID_INPUT: InvalidInputError,
|
|
151
|
+
ErrorKind.CREDENTIALS_REJECTED: CredentialsRejectedError,
|
|
152
|
+
ErrorKind.ACCOUNT_ACTION_REQUIRED: AccountActionRequiredError,
|
|
153
|
+
ErrorKind.SESSION_EXPIRED: SessionExpiredError,
|
|
154
|
+
ErrorKind.ACCESS_DENIED: AccessDeniedError,
|
|
155
|
+
ErrorKind.UNSUPPORTED_CAPABILITY: UnsupportedCapabilityError,
|
|
156
|
+
ErrorKind.VIEW_DISABLED: ViewDisabledError,
|
|
157
|
+
ErrorKind.THROTTLED: ThrottledError,
|
|
158
|
+
ErrorKind.MAINTENANCE: MaintenanceError,
|
|
159
|
+
ErrorKind.CONNECTION: ConnectionError,
|
|
160
|
+
ErrorKind.TIMEOUT: OperationTimeoutError,
|
|
161
|
+
ErrorKind.LIMIT: LimitError,
|
|
162
|
+
ErrorKind.PARSE: ParseError,
|
|
163
|
+
ErrorKind.STALE_CURSOR: StaleCursorError,
|
|
164
|
+
ErrorKind.UNKNOWN_DELIVERY: UnknownDeliveryError,
|
|
165
|
+
ErrorKind.CLOSED: ClosedError,
|
|
166
|
+
ErrorKind.STORAGE: StorageError,
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def error_for(kind: ErrorKind) -> LibrusError:
|
|
171
|
+
"""Single redacted factory for every library-owned failure."""
|
|
172
|
+
return _ERROR_TYPES[kind](kind)
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
"""Optional local attachment publication; no implicit destination or network client."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import hashlib
|
|
5
|
+
import os
|
|
6
|
+
import secrets
|
|
7
|
+
import unicodedata
|
|
8
|
+
from collections.abc import Callable
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from functools import partial
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from librus_python_api.attachment_routes import validate_max_bytes
|
|
14
|
+
from librus_python_api.attachments import AttachmentStream
|
|
15
|
+
from librus_python_api.config import ATTACHMENT_MAX_BYTES
|
|
16
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
17
|
+
from librus_python_api.lifecycle import join_owned
|
|
18
|
+
|
|
19
|
+
MAX_FILENAME_BYTES = 180
|
|
20
|
+
MAX_FILENAME_ATTEMPTS = 100
|
|
21
|
+
_RESERVED = {"CON", "PRN", "AUX", "NUL"} | {
|
|
22
|
+
f"{prefix}{i}" for prefix in ("COM", "LPT") for i in range(1, 10)
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True, slots=True)
|
|
27
|
+
class PublishedAttachment:
|
|
28
|
+
path: Path = field(repr=False)
|
|
29
|
+
size_bytes: int
|
|
30
|
+
sha256: str = field(repr=False)
|
|
31
|
+
content_type: str | None = field(repr=False)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def safe_attachment_filename(filename: str) -> str:
|
|
35
|
+
"""Treat upstream names as untrusted display data, never as a relative path."""
|
|
36
|
+
if type(filename) is not str or len(filename) > 4096:
|
|
37
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
38
|
+
name = filename.replace("\\", "/").rsplit("/", 1)[-1]
|
|
39
|
+
name = "".join(
|
|
40
|
+
"_" if unicodedata.category(c).startswith("C") or c in '<>:"|?*' else c
|
|
41
|
+
for c in name
|
|
42
|
+
).strip(" .")
|
|
43
|
+
if name.split(".", 1)[0].upper() in _RESERVED:
|
|
44
|
+
name = "_" + name
|
|
45
|
+
# Truncate on UTF-8 codepoint boundaries, leaving room for collision suffixes.
|
|
46
|
+
name = name.encode()[:MAX_FILENAME_BYTES].decode("utf-8", errors="ignore")
|
|
47
|
+
return name.rstrip(" .") or "attachment"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
async def _disk[T](function: Callable[[], T]) -> T:
|
|
51
|
+
task = asyncio.create_task(asyncio.to_thread(function))
|
|
52
|
+
interrupted = await join_owned(task)
|
|
53
|
+
if interrupted:
|
|
54
|
+
raise asyncio.CancelledError
|
|
55
|
+
return task.result()
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _write_all(descriptor: int, chunk: bytes) -> None:
|
|
59
|
+
view = memoryview(chunk)
|
|
60
|
+
while view:
|
|
61
|
+
count = os.write(descriptor, view)
|
|
62
|
+
if count <= 0:
|
|
63
|
+
raise OSError("Local write made no progress")
|
|
64
|
+
view = view[count:]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _publish(directory: int, descriptor: int, temporary: str, name: str) -> str:
|
|
68
|
+
os.fsync(descriptor)
|
|
69
|
+
path = Path(name)
|
|
70
|
+
for index in range(MAX_FILENAME_ATTEMPTS):
|
|
71
|
+
candidate = name if index == 0 else f"{path.stem} ({index}){path.suffix}"
|
|
72
|
+
try:
|
|
73
|
+
# A same-directory hard link is atomic and fails if any target exists,
|
|
74
|
+
# including a dangling symlink. There is no check-then-replace race.
|
|
75
|
+
os.link(
|
|
76
|
+
temporary,
|
|
77
|
+
candidate,
|
|
78
|
+
src_dir_fd=directory,
|
|
79
|
+
dst_dir_fd=directory,
|
|
80
|
+
follow_symlinks=False,
|
|
81
|
+
)
|
|
82
|
+
except FileExistsError:
|
|
83
|
+
continue
|
|
84
|
+
os.fsync(directory)
|
|
85
|
+
return candidate
|
|
86
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
async def publish_attachment(
|
|
90
|
+
stream: AttachmentStream,
|
|
91
|
+
directory: Path,
|
|
92
|
+
*,
|
|
93
|
+
filename: str,
|
|
94
|
+
max_bytes: int = ATTACHMENT_MAX_BYTES,
|
|
95
|
+
) -> PublishedAttachment:
|
|
96
|
+
"""Publish only a complete bounded stream to an existing caller-selected directory.
|
|
97
|
+
|
|
98
|
+
Files are owner-only. Failed/cancelled reads remove only this call's temporary
|
|
99
|
+
file. Cancellation during the atomic commit may leave a complete final file;
|
|
100
|
+
it never overwrites or removes an existing final path. All disk workers are
|
|
101
|
+
joined before closing descriptors or removing their temporary file.
|
|
102
|
+
"""
|
|
103
|
+
if not isinstance(stream, AttachmentStream) or not isinstance(directory, Path):
|
|
104
|
+
raise LibrusError(ErrorKind.INVALID_INPUT)
|
|
105
|
+
validate_max_bytes(max_bytes)
|
|
106
|
+
# Directory-relative, no-follow opens are the safety boundary (POSIX only).
|
|
107
|
+
if not hasattr(os, "O_DIRECTORY") or not hasattr(os, "O_NOFOLLOW"):
|
|
108
|
+
raise LibrusError(ErrorKind.UNSUPPORTED_CAPABILITY)
|
|
109
|
+
name = safe_attachment_filename(filename)
|
|
110
|
+
directory_fd = descriptor = None
|
|
111
|
+
temporary = ".librus-attachment-" + secrets.token_hex(16)
|
|
112
|
+
size = 0
|
|
113
|
+
digest = hashlib.sha256()
|
|
114
|
+
failed = False
|
|
115
|
+
try:
|
|
116
|
+
directory_fd = os.open(directory, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW)
|
|
117
|
+
descriptor = os.open(
|
|
118
|
+
temporary,
|
|
119
|
+
os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW,
|
|
120
|
+
0o600,
|
|
121
|
+
dir_fd=directory_fd,
|
|
122
|
+
)
|
|
123
|
+
async with stream:
|
|
124
|
+
async for chunk in stream:
|
|
125
|
+
size += len(chunk)
|
|
126
|
+
if size > max_bytes:
|
|
127
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
128
|
+
await _disk(partial(_write_all, descriptor, chunk))
|
|
129
|
+
digest.update(chunk)
|
|
130
|
+
if not stream.complete:
|
|
131
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
132
|
+
content_type = stream.metadata.headers.content_type
|
|
133
|
+
final = await _disk(lambda: _publish(directory_fd, descriptor, temporary, name))
|
|
134
|
+
return PublishedAttachment(
|
|
135
|
+
directory / final, size, digest.hexdigest(), content_type
|
|
136
|
+
)
|
|
137
|
+
except OSError:
|
|
138
|
+
failed = True
|
|
139
|
+
finally:
|
|
140
|
+
# Local cleanup has no outstanding worker and targets only our random name.
|
|
141
|
+
try:
|
|
142
|
+
if descriptor is not None:
|
|
143
|
+
os.close(descriptor)
|
|
144
|
+
assert directory_fd is not None
|
|
145
|
+
os.unlink(temporary, dir_fd=directory_fd)
|
|
146
|
+
except OSError:
|
|
147
|
+
failed = True
|
|
148
|
+
finally:
|
|
149
|
+
if directory_fd is not None:
|
|
150
|
+
try:
|
|
151
|
+
os.close(directory_fd)
|
|
152
|
+
except OSError:
|
|
153
|
+
failed = True
|
|
154
|
+
if failed:
|
|
155
|
+
raise LibrusError(ErrorKind.STORAGE) from None
|
|
156
|
+
raise LibrusError(ErrorKind.STORAGE)
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Original bounded HTML summary parsing from source-informed requirements."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Iterator
|
|
5
|
+
|
|
6
|
+
from lxml import html
|
|
7
|
+
|
|
8
|
+
from librus_python_api import markup
|
|
9
|
+
from librus_python_api.config import (
|
|
10
|
+
GRADE_BODY_PREFIX_COLUMNS,
|
|
11
|
+
GRADE_INLINE_DETAIL_LABEL,
|
|
12
|
+
GRADE_MAX_COLUMNS,
|
|
13
|
+
GRADE_MAX_SUBJECTS,
|
|
14
|
+
GRADE_MAX_VALUE_LENGTH,
|
|
15
|
+
GRADE_MERGED_SUBJECTS,
|
|
16
|
+
GRADE_SUMMARY_HEADERS,
|
|
17
|
+
)
|
|
18
|
+
from librus_python_api.exceptions import ErrorKind, LibrusError
|
|
19
|
+
from librus_python_api.models import (
|
|
20
|
+
Availability,
|
|
21
|
+
GradeSummaryValue,
|
|
22
|
+
SubjectGradeSummary,
|
|
23
|
+
)
|
|
24
|
+
from librus_python_api.parsers import parse_page
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _summary_field(cell: html.HtmlElement) -> str | None:
|
|
28
|
+
title = cell.get("title", "")
|
|
29
|
+
if len(title) > GRADE_MAX_VALUE_LENGTH:
|
|
30
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
31
|
+
label = " ".join(re.split(r"<br\s*/?>", title, flags=re.I)[0].split())
|
|
32
|
+
return GRADE_SUMMARY_HEADERS.get(label)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _header(row: html.HtmlElement) -> tuple[dict[str, int], int]:
|
|
36
|
+
fields: dict[str, int] = {}
|
|
37
|
+
position = GRADE_BODY_PREFIX_COLUMNS
|
|
38
|
+
for cell in markup.cells(row):
|
|
39
|
+
span = markup.colspan(cell)
|
|
40
|
+
if position + span > GRADE_MAX_COLUMNS:
|
|
41
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
42
|
+
field = _summary_field(cell)
|
|
43
|
+
if field is not None:
|
|
44
|
+
if field in fields:
|
|
45
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
46
|
+
fields[field] = position
|
|
47
|
+
position += span
|
|
48
|
+
return fields, position
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _locate(
|
|
52
|
+
document: html.HtmlElement,
|
|
53
|
+
) -> tuple[html.HtmlElement, dict[str, int], int]:
|
|
54
|
+
candidates: list[tuple[html.HtmlElement, dict[str, int], int]] = []
|
|
55
|
+
for table in document.iter("table"):
|
|
56
|
+
if not {"decorated", "stretch"} <= set(table.get("class", "").split()):
|
|
57
|
+
continue
|
|
58
|
+
for row in markup.rows(table):
|
|
59
|
+
if next(row.iterancestors("thead"), None) is None:
|
|
60
|
+
continue
|
|
61
|
+
# Group-heading rows can span the entire table. Only the row with
|
|
62
|
+
# an annual summary title defines the body alignment contract.
|
|
63
|
+
if not any(_summary_field(cell) == "annual" for cell in markup.cells(row)):
|
|
64
|
+
continue
|
|
65
|
+
fields, width = _header(row)
|
|
66
|
+
candidates.append((table, fields, width))
|
|
67
|
+
if len(candidates) != 1:
|
|
68
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
69
|
+
return candidates[0]
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _outside_nested_tables(row: html.HtmlElement) -> str:
|
|
73
|
+
"""Reject a subject row masquerading as an expanded-detail wrapper."""
|
|
74
|
+
parts: list[str] = []
|
|
75
|
+
pending = [row]
|
|
76
|
+
while pending:
|
|
77
|
+
element = pending.pop()
|
|
78
|
+
if element.tail:
|
|
79
|
+
parts.append(element.tail)
|
|
80
|
+
if element.tag == "table":
|
|
81
|
+
continue
|
|
82
|
+
if element.text:
|
|
83
|
+
parts.append(element.text)
|
|
84
|
+
pending.extend(element)
|
|
85
|
+
return " ".join(parts).strip()
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _summary(
|
|
89
|
+
cells: list[html.HtmlElement],
|
|
90
|
+
fields: dict[str, int],
|
|
91
|
+
width: int,
|
|
92
|
+
) -> SubjectGradeSummary:
|
|
93
|
+
if len(cells) < GRADE_BODY_PREFIX_COLUMNS:
|
|
94
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
95
|
+
subject = markup.text(cells[1])
|
|
96
|
+
if not subject or markup.colspan(cells[0]) != 1 or markup.colspan(cells[1]) != 1:
|
|
97
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
98
|
+
columns: list[html.HtmlElement] = []
|
|
99
|
+
for cell in cells:
|
|
100
|
+
span = markup.colspan(cell)
|
|
101
|
+
if span > 1 and subject not in GRADE_MERGED_SUBJECTS:
|
|
102
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
103
|
+
if len(columns) + span > GRADE_MAX_COLUMNS:
|
|
104
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
105
|
+
columns.extend([cell] * span)
|
|
106
|
+
if len(columns) != width:
|
|
107
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
108
|
+
selected = [columns[index] for index in fields.values()]
|
|
109
|
+
if len({id(cell) for cell in selected}) != len(selected):
|
|
110
|
+
# A single merged value cannot mean two different summary fields.
|
|
111
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
112
|
+
values = {
|
|
113
|
+
name: GradeSummaryValue(Availability.AVAILABLE, markup.text(columns[index]))
|
|
114
|
+
for name, index in fields.items()
|
|
115
|
+
}
|
|
116
|
+
absent = GradeSummaryValue(Availability.UNAVAILABLE, None)
|
|
117
|
+
return SubjectGradeSummary(
|
|
118
|
+
subject,
|
|
119
|
+
values.get("midterm", absent),
|
|
120
|
+
values.get("predicted_annual", absent),
|
|
121
|
+
values["annual"],
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _subject_rows(
|
|
126
|
+
table: html.HtmlElement, width: int
|
|
127
|
+
) -> Iterator[list[html.HtmlElement]]:
|
|
128
|
+
for row in markup.rows(table):
|
|
129
|
+
if next(row.iterancestors("thead"), None) is not None:
|
|
130
|
+
continue
|
|
131
|
+
cells = markup.cells(row)
|
|
132
|
+
if not cells:
|
|
133
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
134
|
+
if list(row.iter("table")):
|
|
135
|
+
if _outside_nested_tables(row):
|
|
136
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
137
|
+
continue
|
|
138
|
+
# An empty full-width layout spacer is not an unassigned subject. Keep
|
|
139
|
+
# this exception narrow: content or a different width remains malformed.
|
|
140
|
+
if (
|
|
141
|
+
len(cells) == 1
|
|
142
|
+
and markup.colspan(cells[0]) == width
|
|
143
|
+
and not markup.text(cells[0])
|
|
144
|
+
):
|
|
145
|
+
continue
|
|
146
|
+
# Expanded inline detail labels are not subjects. Recognize this narrow
|
|
147
|
+
# source-informed variant, never discard arbitrary malformed subject rows.
|
|
148
|
+
if len(cells) > 1 and markup.text(cells[1]) == GRADE_INLINE_DETAIL_LABEL:
|
|
149
|
+
continue
|
|
150
|
+
yield cells
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def parse_final_grades(body: bytes) -> tuple[SubjectGradeSummary, ...]:
|
|
154
|
+
table, fields, width = _locate(parse_page(body))
|
|
155
|
+
items: list[SubjectGradeSummary] = []
|
|
156
|
+
subjects: set[str] = set()
|
|
157
|
+
for cells in _subject_rows(table, width):
|
|
158
|
+
item = _summary(cells, fields, width)
|
|
159
|
+
if item.subject in subjects:
|
|
160
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
161
|
+
subjects.add(item.subject)
|
|
162
|
+
items.append(item)
|
|
163
|
+
if len(items) > GRADE_MAX_SUBJECTS:
|
|
164
|
+
raise LibrusError(ErrorKind.LIMIT)
|
|
165
|
+
# No evidenced no-subject marker exists. Missing rows are not empty success;
|
|
166
|
+
# a valid all-unassigned page still lists its subjects with raw '-' values.
|
|
167
|
+
if not items:
|
|
168
|
+
raise LibrusError(ErrorKind.PARSE)
|
|
169
|
+
return tuple(items)
|