agent2learn 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent2learn/__init__.py +3 -0
- agent2learn/_release.py +19 -0
- agent2learn/aipolicy.py +182 -0
- agent2learn/api.py +590 -0
- agent2learn/audit.py +358 -0
- agent2learn/auth/__init__.py +282 -0
- agent2learn/auth/cdp.py +1067 -0
- agent2learn/auth/paste.py +378 -0
- agent2learn/calendar.py +525 -0
- agent2learn/calibrate.py +347 -0
- agent2learn/check.py +1091 -0
- agent2learn/cli.py +2039 -0
- agent2learn/clock.py +39 -0
- agent2learn/config.py +205 -0
- agent2learn/console.py +229 -0
- agent2learn/convert.py +1223 -0
- agent2learn/doctor.py +1167 -0
- agent2learn/errors.py +32 -0
- agent2learn/ground.py +735 -0
- agent2learn/index.py +614 -0
- agent2learn/ingest.py +3229 -0
- agent2learn/locations.py +247 -0
- agent2learn/outlines.py +754 -0
- agent2learn/paths.py +683 -0
- agent2learn/pipeline.py +392 -0
- agent2learn/privacy.py +1123 -0
- agent2learn/schools/__init__.py +29 -0
- agent2learn/schools/_base.py +194 -0
- agent2learn/schools/generic.py +78 -0
- agent2learn/schools/uwaterloo.py +66 -0
- agent2learn/session.py +373 -0
- agent2learn/skills.py +1081 -0
- agent2learn/snapshot.py +399 -0
- agent2learn/submit.py +1047 -0
- agent2learn/transactions.py +157 -0
- agent2learn/upgrade.py +288 -0
- agent2learn/vault.py +1134 -0
- agent2learn-0.1.2.data/data/a2l-coursework/SKILL.md +52 -0
- agent2learn-0.1.2.data/data/a2l-setup/SKILL.md +27 -0
- agent2learn-0.1.2.data/data/a2l-study/SKILL.md +27 -0
- agent2learn-0.1.2.data/data/a2l-sync/SKILL.md +30 -0
- agent2learn-0.1.2.dist-info/METADATA +186 -0
- agent2learn-0.1.2.dist-info/RECORD +46 -0
- agent2learn-0.1.2.dist-info/WHEEL +4 -0
- agent2learn-0.1.2.dist-info/entry_points.txt +3 -0
- agent2learn-0.1.2.dist-info/licenses/LICENSE +202 -0
agent2learn/ingest.py
ADDED
|
@@ -0,0 +1,3229 @@
|
|
|
1
|
+
"""Metadata-first, revision-safe ingestion of a LEARN course vault.
|
|
2
|
+
|
|
3
|
+
The module deliberately separates the inexpensive JSON phase from the potentially large file
|
|
4
|
+
phase. Metadata is a typed projection of the API, not a raw response archive: external URLs are
|
|
5
|
+
classified before persistence, and only a query-free LEARN view URL plus a destination hostname
|
|
6
|
+
can survive as a link stub.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hmac
|
|
12
|
+
import json
|
|
13
|
+
import os
|
|
14
|
+
import re
|
|
15
|
+
import secrets
|
|
16
|
+
import stat
|
|
17
|
+
import tempfile
|
|
18
|
+
import unicodedata
|
|
19
|
+
from collections import defaultdict
|
|
20
|
+
from collections.abc import Callable, Iterable, Mapping, Sequence
|
|
21
|
+
from contextlib import suppress
|
|
22
|
+
from dataclasses import dataclass, replace
|
|
23
|
+
from hashlib import sha256
|
|
24
|
+
from html import escape
|
|
25
|
+
from html.parser import HTMLParser
|
|
26
|
+
from pathlib import Path, PurePosixPath
|
|
27
|
+
from typing import Any, Literal, Protocol, cast
|
|
28
|
+
from urllib.parse import urljoin, urlsplit, urlunsplit
|
|
29
|
+
|
|
30
|
+
from requests import RequestException
|
|
31
|
+
|
|
32
|
+
from agent2learn import api, clock, locations, paths, snapshot, transactions
|
|
33
|
+
from agent2learn import index as course_index
|
|
34
|
+
from agent2learn.api import DownloadError, DownloadResult, FileTooLarge
|
|
35
|
+
from agent2learn.calibrate import CourseRef, calibrate, load_calibration
|
|
36
|
+
from agent2learn.convert import DEFAULT_OCR_WORDS_PER_PAGE, _validate_threshold, convert_vault
|
|
37
|
+
from agent2learn.errors import A2LError, NotConfigured, SessionExpired
|
|
38
|
+
from agent2learn.schools import (
|
|
39
|
+
School,
|
|
40
|
+
hostname_matches_suffix,
|
|
41
|
+
parse_api_timestamp,
|
|
42
|
+
topic_is_excluded,
|
|
43
|
+
)
|
|
44
|
+
from agent2learn.vault import DerivedArtifact, ManifestEntry, Vault
|
|
45
|
+
|
|
46
|
+
CONTENT_MAP_VERSION = course_index.CONTENT_MAP_VERSION
|
|
47
|
+
DEFAULT_SCOPE: Literal["all", "priority"] = "all"
|
|
48
|
+
PRIORITY_BUDGET_BYTES = 200_000_000
|
|
49
|
+
_MAX_PAGES = 1000
|
|
50
|
+
_COURSE_CONTENT = "content"
|
|
51
|
+
_OFFICE_LOCK = re.compile(r"^~\$", re.IGNORECASE)
|
|
52
|
+
_MEDIA_SUFFIXES = frozenset(
|
|
53
|
+
{
|
|
54
|
+
".3gp",
|
|
55
|
+
".aac",
|
|
56
|
+
".avi",
|
|
57
|
+
".flac",
|
|
58
|
+
".m4a",
|
|
59
|
+
".m4v",
|
|
60
|
+
".mkv",
|
|
61
|
+
".mov",
|
|
62
|
+
".mp3",
|
|
63
|
+
".mp4",
|
|
64
|
+
".mpeg",
|
|
65
|
+
".mpg",
|
|
66
|
+
".ogg",
|
|
67
|
+
".wav",
|
|
68
|
+
".webm",
|
|
69
|
+
".wmv",
|
|
70
|
+
}
|
|
71
|
+
)
|
|
72
|
+
_DOWNLOADABLE_KINDS = frozenset({"file", "html", "htmlfile"})
|
|
73
|
+
_PENDING_MARKER_SUFFIX = ".meta.json"
|
|
74
|
+
_PENDING_INSTALL_SUFFIX = ".part" + _PENDING_MARKER_SUFFIX
|
|
75
|
+
_PENDING_INSTALL_KEYS = frozenset(
|
|
76
|
+
{
|
|
77
|
+
"version",
|
|
78
|
+
"source_key",
|
|
79
|
+
"destination",
|
|
80
|
+
"sha256",
|
|
81
|
+
"size",
|
|
82
|
+
"etag",
|
|
83
|
+
"last_modified",
|
|
84
|
+
"prior_sha256",
|
|
85
|
+
"revision_preserved",
|
|
86
|
+
}
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass(frozen=True)
|
|
91
|
+
class TopicRecord:
|
|
92
|
+
"""The safe, typed projection of one content topic."""
|
|
93
|
+
|
|
94
|
+
source_key: str
|
|
95
|
+
source_id: str
|
|
96
|
+
topic_id: int
|
|
97
|
+
course_org_unit_id: int
|
|
98
|
+
course_code: str
|
|
99
|
+
course_name: str
|
|
100
|
+
term: str | None
|
|
101
|
+
title: str
|
|
102
|
+
kind: str
|
|
103
|
+
module_path: tuple[str, ...]
|
|
104
|
+
module_ids: tuple[int, ...]
|
|
105
|
+
view_url: str
|
|
106
|
+
outline_url: str | None
|
|
107
|
+
url_path: str | None
|
|
108
|
+
external_host: str | None
|
|
109
|
+
etag: str | None
|
|
110
|
+
last_modified: str | None
|
|
111
|
+
is_broken: bool
|
|
112
|
+
availability: str = "metadata_only"
|
|
113
|
+
source_path: str | None = None
|
|
114
|
+
path: str | None = None
|
|
115
|
+
sha256: str | None = None
|
|
116
|
+
size: int | None = None
|
|
117
|
+
stub_path: str | None = None
|
|
118
|
+
remote_size: int | None = None
|
|
119
|
+
next_action: str = "a2l fetch <topic-id>"
|
|
120
|
+
missing_since: str | None = None
|
|
121
|
+
withdrawn_at: str | None = None
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
@dataclass(frozen=True)
|
|
125
|
+
class CourseMetadata:
|
|
126
|
+
"""Metadata and safe content projections for one selected course."""
|
|
127
|
+
|
|
128
|
+
course: CourseRef
|
|
129
|
+
directory: Path
|
|
130
|
+
topics: tuple[TopicRecord, ...]
|
|
131
|
+
module_tree: tuple[dict[str, object], ...]
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
@dataclass(frozen=True)
|
|
135
|
+
class MetadataReport:
|
|
136
|
+
"""Result of the complete cheap metadata phase."""
|
|
137
|
+
|
|
138
|
+
courses: tuple[CourseMetadata, ...]
|
|
139
|
+
topic_count: int
|
|
140
|
+
deadline_count: int
|
|
141
|
+
errors: tuple[str, ...] = ()
|
|
142
|
+
exit_code: int = 0
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
@dataclass(frozen=True)
|
|
146
|
+
class FileReport:
|
|
147
|
+
"""Result of a resumable source-file phase."""
|
|
148
|
+
|
|
149
|
+
downloaded: int = 0
|
|
150
|
+
skipped: int = 0
|
|
151
|
+
failed: int = 0
|
|
152
|
+
metadata_only: int = 0
|
|
153
|
+
download_gaps: int = 0
|
|
154
|
+
interrupted: bool = False
|
|
155
|
+
errors: tuple[str, ...] = ()
|
|
156
|
+
exit_code: int = 0
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
@dataclass(frozen=True)
|
|
160
|
+
class FetchReport:
|
|
161
|
+
"""Result of resolving and fetching one stable topic identity."""
|
|
162
|
+
|
|
163
|
+
source_key: str
|
|
164
|
+
availability: str
|
|
165
|
+
source_path: str | None
|
|
166
|
+
citation_path: str | None
|
|
167
|
+
changed: bool
|
|
168
|
+
next_action: str | None = None
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
@dataclass(frozen=True)
|
|
172
|
+
class OutlineReport:
|
|
173
|
+
"""The outline renderer's bounded result; implemented in :mod:`outlines`."""
|
|
174
|
+
|
|
175
|
+
rendered: int = 0
|
|
176
|
+
unavailable: int = 0
|
|
177
|
+
errors: tuple[str, ...] = ()
|
|
178
|
+
|
|
179
|
+
|
|
180
|
+
@dataclass(frozen=True)
|
|
181
|
+
class _PendingInstall:
|
|
182
|
+
"""Durable proof that a validated download is waiting only on filesystem installation."""
|
|
183
|
+
|
|
184
|
+
marker: Path
|
|
185
|
+
part: Path
|
|
186
|
+
source_key: str
|
|
187
|
+
destination: str
|
|
188
|
+
sha256: str
|
|
189
|
+
size: int
|
|
190
|
+
etag: str | None
|
|
191
|
+
last_modified: str | None
|
|
192
|
+
prior_sha256: str | None
|
|
193
|
+
revision_preserved: bool
|
|
194
|
+
fetched_at: str | None = None
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
class IngestClient(Protocol):
|
|
198
|
+
"""The small calibrated client surface required by the ingest pipeline.
|
|
199
|
+
|
|
200
|
+
Keeping this as a protocol lets the offline fixture client exercise the same code paths as the
|
|
201
|
+
real HTTP client without pretending that a test double is an authenticated ``api.Client``.
|
|
202
|
+
"""
|
|
203
|
+
|
|
204
|
+
@property
|
|
205
|
+
def school(self) -> School: ...
|
|
206
|
+
|
|
207
|
+
lp_version: str | None
|
|
208
|
+
le_version: str | None
|
|
209
|
+
download_template: str | None
|
|
210
|
+
|
|
211
|
+
def get_json(self, path: str) -> object: ...
|
|
212
|
+
|
|
213
|
+
def download(
|
|
214
|
+
self,
|
|
215
|
+
url: str,
|
|
216
|
+
temp: Path,
|
|
217
|
+
*,
|
|
218
|
+
prior: ManifestEntry | None = None,
|
|
219
|
+
max_bytes: int | None = api.DEFAULT_MAX_BYTES,
|
|
220
|
+
is_html_topic: bool = False,
|
|
221
|
+
root: Path | None = None,
|
|
222
|
+
) -> DownloadResult: ...
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def ingest_metadata(
|
|
226
|
+
client: IngestClient,
|
|
227
|
+
vault: Vault,
|
|
228
|
+
school: School,
|
|
229
|
+
*,
|
|
230
|
+
term: str | None = None,
|
|
231
|
+
only: Iterable[int | str] | None = None,
|
|
232
|
+
include_grades: bool = False,
|
|
233
|
+
create_snapshot: bool = True,
|
|
234
|
+
) -> MetadataReport:
|
|
235
|
+
"""Fetch and persist complete typed metadata for the selected courses.
|
|
236
|
+
|
|
237
|
+
No file endpoint is touched here. The function is safe to call before the user chooses a file
|
|
238
|
+
scope, and every category writer merges stable IDs rather than deleting expired records. Direct
|
|
239
|
+
callers retain the historical metadata snapshot by default; the production pipeline disables
|
|
240
|
+
that interim write and creates its single snapshot after conversion and index reconciliation.
|
|
241
|
+
"""
|
|
242
|
+
|
|
243
|
+
courses = _selected_courses(client, term=term, only=only)
|
|
244
|
+
reports: list[CourseMetadata] = []
|
|
245
|
+
errors: list[str] = []
|
|
246
|
+
deadline_count = 0
|
|
247
|
+
|
|
248
|
+
for course in courses:
|
|
249
|
+
course_dir = _course_directory(vault, school, course)
|
|
250
|
+
try:
|
|
251
|
+
paths.ensure_dir(course_dir, root=vault.root)
|
|
252
|
+
except ValueError as exc:
|
|
253
|
+
raise A2LError("vault course path contains a link component") from exc
|
|
254
|
+
try:
|
|
255
|
+
toc_payload, toc_complete, toc_error = _fetch_one(client, _toc_path(client, course))
|
|
256
|
+
if toc_error is not None:
|
|
257
|
+
errors.append(_safe_error("toc", toc_error))
|
|
258
|
+
records, module_tree, toc_valid = _topics_from_toc(
|
|
259
|
+
toc_payload, course=course, school=school
|
|
260
|
+
)
|
|
261
|
+
if not toc_valid and toc_error is None:
|
|
262
|
+
errors.append("toc: invalid response")
|
|
263
|
+
toc_complete = toc_complete and toc_valid
|
|
264
|
+
except SessionExpired:
|
|
265
|
+
raise
|
|
266
|
+
except Exception as exc:
|
|
267
|
+
errors.append(_safe_error("toc", exc))
|
|
268
|
+
records, module_tree, toc_complete = [], [], False
|
|
269
|
+
|
|
270
|
+
assignments, assignments_complete, assignments_error = _fetch_collection(
|
|
271
|
+
client, _endpoint_path(client, course, "dropbox/folders/")
|
|
272
|
+
)
|
|
273
|
+
if assignments_error is not None:
|
|
274
|
+
errors.append(_safe_error("assignments", assignments_error))
|
|
275
|
+
existing_map = _read_content_map(course_dir)
|
|
276
|
+
attachment_topics = _assignment_attachment_topics(assignments, course=course, school=school)
|
|
277
|
+
merged_topics = _merge_topic_records(
|
|
278
|
+
existing_map.get("topics", []),
|
|
279
|
+
[*records, *attachment_topics],
|
|
280
|
+
# Attachment identities come from Dropbox as well as the TOC. A failure in either
|
|
281
|
+
# listing must not mark a previously captured source absent.
|
|
282
|
+
complete=toc_complete and assignments_complete,
|
|
283
|
+
)
|
|
284
|
+
merged_topics = _materialize_external_stubs(
|
|
285
|
+
merged_topics, course_dir=course_dir, vault=vault, school=school, course=course
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
if toc_complete:
|
|
289
|
+
_write_toc(course_dir, module_tree, root=vault.root)
|
|
290
|
+
else:
|
|
291
|
+
try:
|
|
292
|
+
module_tree = _read_toc_modules(course_dir)
|
|
293
|
+
except A2LError as exc:
|
|
294
|
+
# A failed TOC fetch must not turn a corrupt cached tree into an apparently empty
|
|
295
|
+
# course. Keep the metadata phase usable, but make the coverage gap explicit.
|
|
296
|
+
errors.append(_safe_error("toc cache", exc))
|
|
297
|
+
module_tree = []
|
|
298
|
+
|
|
299
|
+
assignments_rows = _project_assignments(assignments)
|
|
300
|
+
active_assignments = {str(row["id"]) for row in assignments_rows}
|
|
301
|
+
previous_assignments = _read_list(course_dir / "_meta" / "assignments.json")
|
|
302
|
+
assignments_rows = _merge_rows(
|
|
303
|
+
previous_assignments,
|
|
304
|
+
assignments_rows,
|
|
305
|
+
id_field="id",
|
|
306
|
+
complete=assignments_complete,
|
|
307
|
+
)
|
|
308
|
+
assignment_directories = locations.assignment_directories(
|
|
309
|
+
vault,
|
|
310
|
+
course_dir,
|
|
311
|
+
school,
|
|
312
|
+
course,
|
|
313
|
+
assignments_rows,
|
|
314
|
+
previous_assignments,
|
|
315
|
+
active_assignments,
|
|
316
|
+
)
|
|
317
|
+
_write_list(course_dir / "_meta" / "assignments.json", assignments_rows, root=vault.root)
|
|
318
|
+
assignment_artifacts = _materialize_assignments(
|
|
319
|
+
assignments,
|
|
320
|
+
course_dir=course_dir,
|
|
321
|
+
vault=vault,
|
|
322
|
+
school=school,
|
|
323
|
+
course=course,
|
|
324
|
+
directories=assignment_directories,
|
|
325
|
+
)
|
|
326
|
+
for row in assignments_rows:
|
|
327
|
+
artifact = assignment_artifacts.get(str(row.get("id")))
|
|
328
|
+
if artifact is not None:
|
|
329
|
+
row.update(artifact)
|
|
330
|
+
_write_list(course_dir / "_meta" / "assignments.json", assignments_rows, root=vault.root)
|
|
331
|
+
|
|
332
|
+
news, news_complete, news_error = _fetch_collection(
|
|
333
|
+
client, _endpoint_path(client, course, "news/")
|
|
334
|
+
)
|
|
335
|
+
if news_error is not None:
|
|
336
|
+
errors.append(_safe_error("news", news_error))
|
|
337
|
+
news_rows = _project_news(news)
|
|
338
|
+
news_rows = _merge_rows(
|
|
339
|
+
_read_list(course_dir / "_meta" / "news.json"),
|
|
340
|
+
news_rows,
|
|
341
|
+
id_field="id",
|
|
342
|
+
complete=news_complete,
|
|
343
|
+
)
|
|
344
|
+
_write_list(course_dir / "_meta" / "news.json", news_rows, root=vault.root)
|
|
345
|
+
_write_announcements(
|
|
346
|
+
course_dir / "announcements" / "announcements.md", news_rows, root=vault.root
|
|
347
|
+
)
|
|
348
|
+
|
|
349
|
+
quizzes, quizzes_complete, quizzes_error = _fetch_collection(
|
|
350
|
+
client, _endpoint_path(client, course, "quizzes/")
|
|
351
|
+
)
|
|
352
|
+
if quizzes_error is not None:
|
|
353
|
+
errors.append(_safe_error("quizzes", quizzes_error))
|
|
354
|
+
quiz_rows = _project_quizzes(quizzes)
|
|
355
|
+
quiz_rows = _merge_rows(
|
|
356
|
+
_read_list(course_dir / "_meta" / "quizzes.json"),
|
|
357
|
+
quiz_rows,
|
|
358
|
+
id_field="id",
|
|
359
|
+
complete=quizzes_complete,
|
|
360
|
+
)
|
|
361
|
+
_write_list(course_dir / "_meta" / "quizzes.json", quiz_rows, root=vault.root)
|
|
362
|
+
|
|
363
|
+
if include_grades:
|
|
364
|
+
grades, grades_complete, grades_error = _fetch_collection(
|
|
365
|
+
client, _endpoint_path(client, course, "grades/values/myGradeValues/")
|
|
366
|
+
)
|
|
367
|
+
if grades_error is not None:
|
|
368
|
+
errors.append(_safe_error("grades", grades_error))
|
|
369
|
+
grades_path = course_dir / "_meta" / "my_grades.json"
|
|
370
|
+
if grades_complete:
|
|
371
|
+
grade_rows = _merge_rows(
|
|
372
|
+
_read_list(grades_path),
|
|
373
|
+
_project_grades(grades),
|
|
374
|
+
id_field="id",
|
|
375
|
+
complete=True,
|
|
376
|
+
)
|
|
377
|
+
_write_list(grades_path, grade_rows, root=vault.root)
|
|
378
|
+
else:
|
|
379
|
+
# Grade values are sensitive, but they still obey merge-not-replace: an
|
|
380
|
+
# incomplete response must never erase the last complete opt-in snapshot.
|
|
381
|
+
partial_grades = _project_grades(grades)
|
|
382
|
+
if partial_grades:
|
|
383
|
+
grade_rows = _merge_rows(
|
|
384
|
+
_read_list(grades_path),
|
|
385
|
+
partial_grades,
|
|
386
|
+
id_field="id",
|
|
387
|
+
complete=False,
|
|
388
|
+
)
|
|
389
|
+
_write_list(grades_path, grade_rows, root=vault.root)
|
|
390
|
+
errors.append("grades: incomplete response")
|
|
391
|
+
|
|
392
|
+
merged_topics = course_index.reconcile_content_map(vault, merged_topics)
|
|
393
|
+
_write_content_map(course_dir, merged_topics, root=vault.root)
|
|
394
|
+
_materialize_submission_only_readmes(
|
|
395
|
+
assignments,
|
|
396
|
+
artifacts=assignment_artifacts,
|
|
397
|
+
course_dir=course_dir,
|
|
398
|
+
directories=assignment_directories,
|
|
399
|
+
topics=merged_topics,
|
|
400
|
+
root=vault.root,
|
|
401
|
+
)
|
|
402
|
+
deadline_count += sum(1 for row in assignments_rows + quiz_rows if row.get("due_date"))
|
|
403
|
+
typed_topics = tuple(_topic_from_row(row, course=course) for row in merged_topics)
|
|
404
|
+
_write_index(course_dir, school=school, course=course, topics=typed_topics, root=vault.root)
|
|
405
|
+
reports.append(
|
|
406
|
+
CourseMetadata(
|
|
407
|
+
course=course,
|
|
408
|
+
directory=course_dir,
|
|
409
|
+
topics=typed_topics,
|
|
410
|
+
module_tree=tuple(module_tree),
|
|
411
|
+
)
|
|
412
|
+
)
|
|
413
|
+
|
|
414
|
+
if create_snapshot:
|
|
415
|
+
snapshot.write_snapshot(
|
|
416
|
+
vault,
|
|
417
|
+
[report.directory for report in reports],
|
|
418
|
+
include_grades=include_grades,
|
|
419
|
+
timestamp=_now(),
|
|
420
|
+
)
|
|
421
|
+
return MetadataReport(
|
|
422
|
+
courses=tuple(reports),
|
|
423
|
+
topic_count=sum(len(report.topics) for report in reports),
|
|
424
|
+
deadline_count=deadline_count,
|
|
425
|
+
errors=tuple(errors),
|
|
426
|
+
)
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def load_metadata_topics(
|
|
430
|
+
vault: Vault, school: School, courses: Iterable[CourseRef]
|
|
431
|
+
) -> tuple[TopicRecord, ...]:
|
|
432
|
+
"""Load validated topic projections for a completed local metadata phase.
|
|
433
|
+
|
|
434
|
+
This is intentionally a local read. It lets a resumed onboarding run show an honest file
|
|
435
|
+
estimate without repeating the network metadata phase, while using the same row decoder and
|
|
436
|
+
filename/media rules as :func:`ingest_files`.
|
|
437
|
+
"""
|
|
438
|
+
|
|
439
|
+
topics: list[TopicRecord] = []
|
|
440
|
+
for course in courses:
|
|
441
|
+
course_dir = _course_directory(vault, school, course)
|
|
442
|
+
content_map_path = course_dir / "_meta" / "content_map.json"
|
|
443
|
+
if paths.is_link(content_map_path) or not paths.long_path(content_map_path).is_file():
|
|
444
|
+
raise A2LError("course metadata is unavailable; run a2l init")
|
|
445
|
+
content_map = _read_content_map(course_dir)
|
|
446
|
+
topics.extend(_topic_from_row(row, course=course) for row in _map_topics(content_map))
|
|
447
|
+
return tuple(topics)
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
def load_metadata_report(
|
|
451
|
+
vault: Vault, school: School, courses: Iterable[CourseRef]
|
|
452
|
+
) -> MetadataReport:
|
|
453
|
+
"""Reconstruct a completed metadata report from validated local course projections."""
|
|
454
|
+
reports: list[CourseMetadata] = []
|
|
455
|
+
deadline_count = 0
|
|
456
|
+
for course in courses:
|
|
457
|
+
course_dir = _course_directory(vault, school, course)
|
|
458
|
+
topics = load_metadata_topics(vault, school, [course])
|
|
459
|
+
module_tree = tuple(_read_toc_modules(course_dir))
|
|
460
|
+
deadline_count += sum(
|
|
461
|
+
1
|
|
462
|
+
for row in [
|
|
463
|
+
*_read_list(course_dir / "_meta" / "assignments.json"),
|
|
464
|
+
*_read_list(course_dir / "_meta" / "quizzes.json"),
|
|
465
|
+
]
|
|
466
|
+
if row.get("due_date")
|
|
467
|
+
)
|
|
468
|
+
reports.append(
|
|
469
|
+
CourseMetadata(
|
|
470
|
+
course=course,
|
|
471
|
+
directory=course_dir,
|
|
472
|
+
topics=topics,
|
|
473
|
+
module_tree=module_tree,
|
|
474
|
+
)
|
|
475
|
+
)
|
|
476
|
+
return MetadataReport(
|
|
477
|
+
courses=tuple(reports),
|
|
478
|
+
topic_count=sum(len(report.topics) for report in reports),
|
|
479
|
+
deadline_count=deadline_count,
|
|
480
|
+
)
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
def is_downloadable_topic(topic: TopicRecord, *, include_media: bool = False) -> bool:
|
|
484
|
+
"""Return whether a topic is eligible for the selected bulk file plan."""
|
|
485
|
+
return (
|
|
486
|
+
topic.availability != "external_link"
|
|
487
|
+
and topic.url_path is not None
|
|
488
|
+
and topic.kind.casefold() in _DOWNLOADABLE_KINDS
|
|
489
|
+
and not topic.is_broken
|
|
490
|
+
and not _is_office_lock(topic)
|
|
491
|
+
and (include_media or not _is_media(topic))
|
|
492
|
+
)
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
def select_priority_topics(
|
|
496
|
+
rows: Sequence[TopicRecord],
|
|
497
|
+
*,
|
|
498
|
+
budget: int = PRIORITY_BUDGET_BYTES,
|
|
499
|
+
include_media: bool = False,
|
|
500
|
+
) -> tuple[TopicRecord, ...]:
|
|
501
|
+
"""Return the deterministic priority subset used by ingest and onboarding estimates."""
|
|
502
|
+
if isinstance(budget, bool) or not isinstance(budget, int) or budget <= 0:
|
|
503
|
+
raise ValueError("priority budget must be a positive integer")
|
|
504
|
+
candidates = [
|
|
505
|
+
topic for topic in rows if is_downloadable_topic(topic, include_media=include_media)
|
|
506
|
+
]
|
|
507
|
+
return tuple(_priority_rows(candidates, scope="priority", budget=budget))
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
def ingest_files(
|
|
511
|
+
client: IngestClient,
|
|
512
|
+
vault: Vault,
|
|
513
|
+
school: School,
|
|
514
|
+
*,
|
|
515
|
+
term: str | None = None,
|
|
516
|
+
only: Iterable[int | str] | None = None,
|
|
517
|
+
scope: Literal["all", "priority"] = DEFAULT_SCOPE,
|
|
518
|
+
include_media: bool = False,
|
|
519
|
+
priority_budget_bytes: int = PRIORITY_BUDGET_BYTES,
|
|
520
|
+
include_discussions: bool = False,
|
|
521
|
+
discussion_authors: bool = False,
|
|
522
|
+
) -> FileReport:
|
|
523
|
+
"""Download an explicit, resumable source scope after metadata is available."""
|
|
524
|
+
|
|
525
|
+
if scope not in {"all", "priority"}:
|
|
526
|
+
raise ValueError("scope must be 'all' or 'priority'")
|
|
527
|
+
if isinstance(priority_budget_bytes, bool) or not isinstance(priority_budget_bytes, int):
|
|
528
|
+
raise ValueError("priority_budget_bytes must be an integer")
|
|
529
|
+
if priority_budget_bytes <= 0:
|
|
530
|
+
raise ValueError("priority_budget_bytes must be positive")
|
|
531
|
+
|
|
532
|
+
selected = _selected_courses(client, term=term, only=only)
|
|
533
|
+
downloaded = skipped = failed = metadata_only = download_gaps = 0
|
|
534
|
+
errors: list[str] = []
|
|
535
|
+
|
|
536
|
+
for course in selected:
|
|
537
|
+
course_dir = _course_directory(vault, school, course)
|
|
538
|
+
content_map = _read_content_map(course_dir)
|
|
539
|
+
rows = [_topic_from_row(row, course=course) for row in _map_topics(content_map)]
|
|
540
|
+
if not paths.long_path(course_dir / "_meta" / "content_map.json").is_file():
|
|
541
|
+
# Keep the public entry point safe when called directly: metadata remains a separate
|
|
542
|
+
# phase, but a missing map is a configuration problem rather than a silent no-op.
|
|
543
|
+
raise A2LError("course metadata is unavailable; run ingest_metadata first")
|
|
544
|
+
|
|
545
|
+
planned = _plan_file_paths(rows, course_dir=course_dir, vault=vault, scope=scope)
|
|
546
|
+
chosen = (
|
|
547
|
+
list(
|
|
548
|
+
select_priority_topics(
|
|
549
|
+
planned,
|
|
550
|
+
budget=priority_budget_bytes,
|
|
551
|
+
include_media=include_media,
|
|
552
|
+
)
|
|
553
|
+
)
|
|
554
|
+
if scope == "priority"
|
|
555
|
+
else list(planned)
|
|
556
|
+
)
|
|
557
|
+
if include_discussions:
|
|
558
|
+
discussion_error = _ingest_discussions(
|
|
559
|
+
client,
|
|
560
|
+
course,
|
|
561
|
+
course_dir,
|
|
562
|
+
vault,
|
|
563
|
+
include_authors=discussion_authors,
|
|
564
|
+
)
|
|
565
|
+
if discussion_error is not None:
|
|
566
|
+
errors.append(discussion_error)
|
|
567
|
+
|
|
568
|
+
for topic in chosen:
|
|
569
|
+
if topic.availability == "external_link":
|
|
570
|
+
skipped += 1
|
|
571
|
+
continue
|
|
572
|
+
if topic.url_path is None or topic.kind.casefold() not in _DOWNLOADABLE_KINDS:
|
|
573
|
+
skipped += 1
|
|
574
|
+
metadata_only += 1
|
|
575
|
+
_update_row_state(
|
|
576
|
+
course_dir,
|
|
577
|
+
topic.source_key,
|
|
578
|
+
root=vault.root,
|
|
579
|
+
availability="metadata_only",
|
|
580
|
+
next_action="topic is metadata-only until explicitly fetched",
|
|
581
|
+
)
|
|
582
|
+
continue
|
|
583
|
+
if _is_office_lock(topic):
|
|
584
|
+
skipped += 1
|
|
585
|
+
metadata_only += 1
|
|
586
|
+
_update_row_state(
|
|
587
|
+
course_dir,
|
|
588
|
+
topic.source_key,
|
|
589
|
+
root=vault.root,
|
|
590
|
+
availability="metadata_only",
|
|
591
|
+
next_action="office lock file skipped",
|
|
592
|
+
)
|
|
593
|
+
continue
|
|
594
|
+
if _is_media(topic) and not include_media:
|
|
595
|
+
skipped += 1
|
|
596
|
+
metadata_only += 1
|
|
597
|
+
_update_row_state(
|
|
598
|
+
course_dir,
|
|
599
|
+
topic.source_key,
|
|
600
|
+
root=vault.root,
|
|
601
|
+
availability="metadata_only",
|
|
602
|
+
next_action="media excluded; rerun with --include-media",
|
|
603
|
+
)
|
|
604
|
+
continue
|
|
605
|
+
if topic.is_broken:
|
|
606
|
+
skipped += 1
|
|
607
|
+
metadata_only += 1
|
|
608
|
+
_update_row_state(
|
|
609
|
+
course_dir,
|
|
610
|
+
topic.source_key,
|
|
611
|
+
root=vault.root,
|
|
612
|
+
availability="metadata_only",
|
|
613
|
+
next_action="topic is marked broken in LEARN",
|
|
614
|
+
)
|
|
615
|
+
continue
|
|
616
|
+
if topic.remote_size is not None and topic.remote_size > api.DEFAULT_MAX_BYTES:
|
|
617
|
+
skipped += 1
|
|
618
|
+
metadata_only += 1
|
|
619
|
+
_update_row_state(
|
|
620
|
+
course_dir,
|
|
621
|
+
topic.source_key,
|
|
622
|
+
root=vault.root,
|
|
623
|
+
availability="metadata_only",
|
|
624
|
+
next_action=f"a2l fetch --allow-large {topic.source_id}",
|
|
625
|
+
)
|
|
626
|
+
continue
|
|
627
|
+
|
|
628
|
+
try:
|
|
629
|
+
result = _ingest_one_topic(client, vault, school, course_dir, topic)
|
|
630
|
+
except KeyboardInterrupt:
|
|
631
|
+
return FileReport(
|
|
632
|
+
downloaded=downloaded,
|
|
633
|
+
skipped=skipped,
|
|
634
|
+
failed=failed,
|
|
635
|
+
metadata_only=metadata_only,
|
|
636
|
+
download_gaps=download_gaps,
|
|
637
|
+
interrupted=True,
|
|
638
|
+
errors=tuple(errors),
|
|
639
|
+
exit_code=130,
|
|
640
|
+
)
|
|
641
|
+
except SessionExpired:
|
|
642
|
+
raise
|
|
643
|
+
except FileTooLarge:
|
|
644
|
+
skipped += 1
|
|
645
|
+
metadata_only += 1
|
|
646
|
+
_update_row_state(
|
|
647
|
+
course_dir,
|
|
648
|
+
topic.source_key,
|
|
649
|
+
root=vault.root,
|
|
650
|
+
availability="metadata_only",
|
|
651
|
+
next_action=f"a2l fetch --allow-large {topic.source_id}",
|
|
652
|
+
)
|
|
653
|
+
continue
|
|
654
|
+
except api.DiskSpaceExhausted as exc:
|
|
655
|
+
failed += 1
|
|
656
|
+
errors.append(_safe_error("download", exc))
|
|
657
|
+
continue
|
|
658
|
+
except DownloadError as exc:
|
|
659
|
+
download_gaps += 1
|
|
660
|
+
_update_row_state(
|
|
661
|
+
course_dir,
|
|
662
|
+
topic.source_key,
|
|
663
|
+
root=vault.root,
|
|
664
|
+
availability="download_gap",
|
|
665
|
+
next_action=(
|
|
666
|
+
f"download failed ({type(exc).__name__}); "
|
|
667
|
+
f"retry: a2l fetch {topic.source_id}"
|
|
668
|
+
),
|
|
669
|
+
)
|
|
670
|
+
continue
|
|
671
|
+
|
|
672
|
+
if result == "downloaded":
|
|
673
|
+
downloaded += 1
|
|
674
|
+
else:
|
|
675
|
+
skipped += 1
|
|
676
|
+
|
|
677
|
+
return FileReport(
|
|
678
|
+
downloaded=downloaded,
|
|
679
|
+
skipped=skipped,
|
|
680
|
+
failed=failed,
|
|
681
|
+
metadata_only=metadata_only,
|
|
682
|
+
download_gaps=download_gaps,
|
|
683
|
+
errors=tuple(errors),
|
|
684
|
+
)
|
|
685
|
+
|
|
686
|
+
|
|
687
|
+
def fetch_topic(
|
|
688
|
+
client: IngestClient,
|
|
689
|
+
vault: Vault,
|
|
690
|
+
school: School,
|
|
691
|
+
topic: str,
|
|
692
|
+
*,
|
|
693
|
+
allow_large: bool = False,
|
|
694
|
+
confirm: Callable[[int | None], bool] | None = None,
|
|
695
|
+
ocr_words_per_page: int = DEFAULT_OCR_WORDS_PER_PAGE,
|
|
696
|
+
) -> FetchReport:
|
|
697
|
+
"""Resolve one stable topic ID/path/title and fetch only that source."""
|
|
698
|
+
|
|
699
|
+
_validate_threshold(ocr_words_per_page)
|
|
700
|
+
match = _resolve_topic(vault, topic)
|
|
701
|
+
if match is None:
|
|
702
|
+
raise A2LError(f"topic not found: {topic}")
|
|
703
|
+
course, record, course_dir = match
|
|
704
|
+
if record.availability == "external_link":
|
|
705
|
+
raise A2LError("external or licensed topics are link stubs and cannot be fetched")
|
|
706
|
+
known_oversized = record.remote_size is not None and record.remote_size > api.DEFAULT_MAX_BYTES
|
|
707
|
+
unbounded = False
|
|
708
|
+
if known_oversized or (record.remote_size is None and allow_large):
|
|
709
|
+
if not allow_large:
|
|
710
|
+
raise A2LError(
|
|
711
|
+
"source size exceeds the default limit; run: "
|
|
712
|
+
f"a2l fetch --allow-large {record.source_id}"
|
|
713
|
+
)
|
|
714
|
+
if confirm is None or not confirm(record.remote_size):
|
|
715
|
+
raise A2LError("large-file fetch cancelled")
|
|
716
|
+
unbounded = True
|
|
717
|
+
|
|
718
|
+
planned = _plan_file_paths([record], course_dir=course_dir, vault=vault, scope="all")
|
|
719
|
+
if not planned or planned[0].source_path is None:
|
|
720
|
+
raise A2LError("topic has no fetchable first-party source")
|
|
721
|
+
record = planned[0]
|
|
722
|
+
|
|
723
|
+
result = _ingest_one_topic(
|
|
724
|
+
client,
|
|
725
|
+
vault,
|
|
726
|
+
school,
|
|
727
|
+
course_dir,
|
|
728
|
+
record,
|
|
729
|
+
max_bytes=None if unbounded else api.DEFAULT_MAX_BYTES,
|
|
730
|
+
)
|
|
731
|
+
conversion = convert_vault(
|
|
732
|
+
vault, source_keys=(record.source_key,), ocr_words_per_page=ocr_words_per_page
|
|
733
|
+
)
|
|
734
|
+
refreshed = _topic_from_row(
|
|
735
|
+
_find_content_row(course_dir, record.source_key) or _topic_to_row(record), course=course
|
|
736
|
+
)
|
|
737
|
+
return FetchReport(
|
|
738
|
+
source_key=record.source_key,
|
|
739
|
+
availability=refreshed.availability,
|
|
740
|
+
source_path=refreshed.source_path,
|
|
741
|
+
citation_path=refreshed.path,
|
|
742
|
+
changed=result == "downloaded" or conversion.converted > 0,
|
|
743
|
+
next_action=refreshed.next_action,
|
|
744
|
+
)
|
|
745
|
+
|
|
746
|
+
|
|
747
|
+
def _selected_courses(
|
|
748
|
+
client: IngestClient,
|
|
749
|
+
*,
|
|
750
|
+
term: str | None,
|
|
751
|
+
only: Iterable[int | str] | None,
|
|
752
|
+
) -> list[CourseRef]:
|
|
753
|
+
raw_courses: object = getattr(client, "courses", None)
|
|
754
|
+
if raw_courses is None:
|
|
755
|
+
calibration = getattr(client, "calibration", None)
|
|
756
|
+
if calibration is None:
|
|
757
|
+
try:
|
|
758
|
+
calibration = load_calibration()
|
|
759
|
+
except NotConfigured:
|
|
760
|
+
calibration = calibrate(client)
|
|
761
|
+
raw_courses = getattr(calibration, "courses", None)
|
|
762
|
+
if getattr(client, "lp_version", None) is None:
|
|
763
|
+
client.lp_version = getattr(calibration, "lp", None)
|
|
764
|
+
if getattr(client, "le_version", None) is None:
|
|
765
|
+
client.le_version = getattr(calibration, "le", None)
|
|
766
|
+
if client.download_template is None:
|
|
767
|
+
client.download_template = getattr(calibration, "download_template", None)
|
|
768
|
+
|
|
769
|
+
if not isinstance(raw_courses, Sequence) or isinstance(raw_courses, (str, bytes)):
|
|
770
|
+
raise A2LError("course metadata is unavailable; run calibration first")
|
|
771
|
+
selectors = {str(value).casefold() for value in only} if only is not None else None
|
|
772
|
+
courses: list[CourseRef] = []
|
|
773
|
+
for raw in raw_courses:
|
|
774
|
+
course = _course_ref(raw)
|
|
775
|
+
if not course.is_active:
|
|
776
|
+
continue
|
|
777
|
+
if term is not None and course.term != term:
|
|
778
|
+
continue
|
|
779
|
+
if selectors is not None and not (
|
|
780
|
+
str(course.org_unit_id).casefold() in selectors or course.code.casefold() in selectors
|
|
781
|
+
):
|
|
782
|
+
continue
|
|
783
|
+
courses.append(course)
|
|
784
|
+
return sorted(
|
|
785
|
+
courses, key=lambda item: (item.term or "", item.code.casefold(), item.org_unit_id)
|
|
786
|
+
)
|
|
787
|
+
|
|
788
|
+
|
|
789
|
+
def _course_ref(raw: object) -> CourseRef:
|
|
790
|
+
if isinstance(raw, CourseRef):
|
|
791
|
+
return raw
|
|
792
|
+
if not isinstance(raw, Mapping):
|
|
793
|
+
raise A2LError("course metadata contains an invalid course")
|
|
794
|
+
try:
|
|
795
|
+
org_unit_id = raw["org_unit_id"]
|
|
796
|
+
code = raw["code"]
|
|
797
|
+
name = raw["name"]
|
|
798
|
+
term = raw.get("term")
|
|
799
|
+
is_active = raw["is_active"]
|
|
800
|
+
except KeyError as exc:
|
|
801
|
+
raise A2LError("course metadata contains an invalid course") from exc
|
|
802
|
+
if (
|
|
803
|
+
isinstance(org_unit_id, bool)
|
|
804
|
+
or not isinstance(org_unit_id, int)
|
|
805
|
+
or not isinstance(code, str)
|
|
806
|
+
or not code
|
|
807
|
+
or not isinstance(name, str)
|
|
808
|
+
or not name
|
|
809
|
+
or not isinstance(is_active, bool)
|
|
810
|
+
or term is not None
|
|
811
|
+
and not isinstance(term, str)
|
|
812
|
+
):
|
|
813
|
+
raise A2LError("course metadata contains an invalid course")
|
|
814
|
+
return CourseRef(org_unit_id, code, name, term, is_active)
|
|
815
|
+
|
|
816
|
+
|
|
817
|
+
def _course_directory(vault: Vault, school: School, course: CourseRef) -> Path:
|
|
818
|
+
term_code = course.term or "unclassified"
|
|
819
|
+
if course.term is None:
|
|
820
|
+
term_label = "Unclassified"
|
|
821
|
+
else:
|
|
822
|
+
try:
|
|
823
|
+
term_label = school.term_label(course.term)
|
|
824
|
+
except ValueError:
|
|
825
|
+
term_label = f"Term {course.term}"
|
|
826
|
+
course_label = f"{course.code}_{term_code}" if course.code else f"Course-{course.org_unit_id}"
|
|
827
|
+
preferred = vault.root / paths.safe_name(term_label) / paths.safe_name(course_label)
|
|
828
|
+
return locations.course_directory(vault, school, course, preferred)
|
|
829
|
+
|
|
830
|
+
|
|
831
|
+
def _toc_path(client: IngestClient, course: CourseRef) -> str:
|
|
832
|
+
le = getattr(client, "le_version", None)
|
|
833
|
+
if not isinstance(le, str) or not le:
|
|
834
|
+
raise A2LError("LE API version is not calibrated")
|
|
835
|
+
return f"/d2l/api/le/{le}/{course.org_unit_id}/content/toc"
|
|
836
|
+
|
|
837
|
+
|
|
838
|
+
def _endpoint_path(client: IngestClient, course: CourseRef, endpoint: str) -> str:
|
|
839
|
+
le = getattr(client, "le_version", None)
|
|
840
|
+
if not isinstance(le, str) or not le:
|
|
841
|
+
raise A2LError("LE API version is not calibrated")
|
|
842
|
+
return f"/d2l/api/le/{le}/{course.org_unit_id}/{endpoint}"
|
|
843
|
+
|
|
844
|
+
|
|
845
|
+
def _fetch_one(client: IngestClient, path: str) -> tuple[object, bool, Exception | None]:
|
|
846
|
+
try:
|
|
847
|
+
return client.get_json(path), True, None
|
|
848
|
+
except SessionExpired:
|
|
849
|
+
raise
|
|
850
|
+
except Exception as exc:
|
|
851
|
+
return {}, False, exc
|
|
852
|
+
|
|
853
|
+
|
|
854
|
+
def _fetch_collection(
|
|
855
|
+
client: IngestClient, path: str
|
|
856
|
+
) -> tuple[list[object], bool, Exception | None]:
|
|
857
|
+
values: list[object] = []
|
|
858
|
+
seen: set[str] = set()
|
|
859
|
+
current = path
|
|
860
|
+
for _ in range(_MAX_PAGES):
|
|
861
|
+
if current in seen:
|
|
862
|
+
return values, False, A2LError("metadata pagination repeated a page")
|
|
863
|
+
seen.add(current)
|
|
864
|
+
try:
|
|
865
|
+
payload = client.get_json(current)
|
|
866
|
+
except SessionExpired:
|
|
867
|
+
raise
|
|
868
|
+
except Exception as exc:
|
|
869
|
+
return values, False, exc
|
|
870
|
+
page, next_page, valid = _page_values(payload, current)
|
|
871
|
+
if not valid:
|
|
872
|
+
return values, False, A2LError("metadata endpoint returned an invalid page")
|
|
873
|
+
if any(not _collection_item_is_valid(item, current) for item in page):
|
|
874
|
+
return values, False, A2LError("metadata endpoint returned an invalid item")
|
|
875
|
+
values.extend(page)
|
|
876
|
+
if next_page is None:
|
|
877
|
+
return values, True, None
|
|
878
|
+
if not isinstance(next_page, str) or not next_page:
|
|
879
|
+
return values, False, A2LError("metadata pagination returned an invalid next route")
|
|
880
|
+
current = next_page
|
|
881
|
+
return values, False, A2LError("metadata pagination exceeded its limit")
|
|
882
|
+
|
|
883
|
+
|
|
884
|
+
def _page_values(payload: object, current: str) -> tuple[list[object], str | None, bool]:
|
|
885
|
+
if isinstance(payload, list):
|
|
886
|
+
return list(payload), None, True
|
|
887
|
+
if not isinstance(payload, dict):
|
|
888
|
+
return [], None, False
|
|
889
|
+
if isinstance(payload.get("Objects"), list):
|
|
890
|
+
next_page = payload.get("Next")
|
|
891
|
+
return list(payload["Objects"]), cast(str | None, next_page), True
|
|
892
|
+
if isinstance(payload.get("Items"), list):
|
|
893
|
+
paging = payload.get("PagingInfo")
|
|
894
|
+
if paging is None:
|
|
895
|
+
return list(payload["Items"]), cast(str | None, payload.get("Next")), True
|
|
896
|
+
if not isinstance(paging, dict) or not isinstance(paging.get("HasMoreItems"), bool):
|
|
897
|
+
return [], None, False
|
|
898
|
+
if not paging["HasMoreItems"]:
|
|
899
|
+
return list(payload["Items"]), None, True
|
|
900
|
+
bookmark = paging.get("Bookmark")
|
|
901
|
+
if not isinstance(bookmark, str) or not bookmark:
|
|
902
|
+
return [], None, False
|
|
903
|
+
# D2L's bookmark route is endpoint-specific; retain the path and add only the opaque
|
|
904
|
+
# bookmark value needed for the next request. It is never persisted in the vault.
|
|
905
|
+
next_route = _as_route_string(payload.get("Next")) or current
|
|
906
|
+
separator = "&" if "?" in next_route else "?"
|
|
907
|
+
return (
|
|
908
|
+
list(payload["Items"]),
|
|
909
|
+
f"{next_route}{separator}bookmark={bookmark}",
|
|
910
|
+
True,
|
|
911
|
+
)
|
|
912
|
+
return [], None, False
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
def _collection_item_is_valid(value: object, path: str) -> bool:
|
|
916
|
+
"""Require stable IDs before a response can be considered complete for merge purposes."""
|
|
917
|
+
if not isinstance(value, dict):
|
|
918
|
+
return False
|
|
919
|
+
route = path.casefold()
|
|
920
|
+
if "dropbox/folders" in route or "/news/" in route:
|
|
921
|
+
identifier = value.get("Id")
|
|
922
|
+
if not isinstance(identifier, int) or isinstance(identifier, bool):
|
|
923
|
+
return False
|
|
924
|
+
if "dropbox/folders" in route:
|
|
925
|
+
return _assignment_attachments_are_valid(value)
|
|
926
|
+
return True
|
|
927
|
+
if "/quizzes/" in route:
|
|
928
|
+
identifier = value.get("QuizId")
|
|
929
|
+
return isinstance(identifier, int) and not isinstance(identifier, bool)
|
|
930
|
+
if "/grades/" in route:
|
|
931
|
+
identifier = value.get("GradeObjectIdentifier")
|
|
932
|
+
return (isinstance(identifier, int) and not isinstance(identifier, bool)) or (
|
|
933
|
+
isinstance(identifier, str) and bool(identifier)
|
|
934
|
+
)
|
|
935
|
+
if "/discussions/forums/" in route:
|
|
936
|
+
identifier = value.get("ForumId")
|
|
937
|
+
return isinstance(identifier, int) and not isinstance(identifier, bool)
|
|
938
|
+
return True
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
def _as_route_string(value: object) -> str:
|
|
942
|
+
return value if isinstance(value, str) and value else ""
|
|
943
|
+
|
|
944
|
+
|
|
945
|
+
def _topics_from_toc(
|
|
946
|
+
payload: object, *, course: CourseRef, school: School
|
|
947
|
+
) -> tuple[list[TopicRecord], list[dict[str, object]], bool]:
|
|
948
|
+
if not isinstance(payload, dict) or not isinstance(payload.get("Modules"), list):
|
|
949
|
+
return [], [], False
|
|
950
|
+
records: list[TopicRecord] = []
|
|
951
|
+
valid = True
|
|
952
|
+
|
|
953
|
+
def walk(
|
|
954
|
+
modules: object, parent_titles: tuple[str, ...], parent_ids: tuple[int, ...]
|
|
955
|
+
) -> list[dict[str, object]]:
|
|
956
|
+
nonlocal valid
|
|
957
|
+
if not isinstance(modules, list):
|
|
958
|
+
valid = False
|
|
959
|
+
return []
|
|
960
|
+
projected: list[dict[str, object]] = []
|
|
961
|
+
for module in modules:
|
|
962
|
+
if not isinstance(module, dict):
|
|
963
|
+
valid = False
|
|
964
|
+
continue
|
|
965
|
+
module_id = module.get("ModuleId")
|
|
966
|
+
title = module.get("Title")
|
|
967
|
+
children = module.get("Modules")
|
|
968
|
+
topics = module.get("Topics")
|
|
969
|
+
if (
|
|
970
|
+
isinstance(module_id, bool)
|
|
971
|
+
or not isinstance(module_id, int)
|
|
972
|
+
or not isinstance(title, str)
|
|
973
|
+
or not isinstance(children, list)
|
|
974
|
+
or not isinstance(topics, list)
|
|
975
|
+
):
|
|
976
|
+
valid = False
|
|
977
|
+
continue
|
|
978
|
+
module_titles = parent_titles + (title,)
|
|
979
|
+
module_ids = parent_ids + (module_id,)
|
|
980
|
+
topic_projection: list[dict[str, object]] = []
|
|
981
|
+
for raw_topic in topics:
|
|
982
|
+
record = _topic_from_api(
|
|
983
|
+
raw_topic,
|
|
984
|
+
course=course,
|
|
985
|
+
school=school,
|
|
986
|
+
module_titles=module_titles,
|
|
987
|
+
module_ids=module_ids,
|
|
988
|
+
)
|
|
989
|
+
if record is None:
|
|
990
|
+
valid = False
|
|
991
|
+
continue
|
|
992
|
+
records.append(record)
|
|
993
|
+
topic_projection.append(_topic_projection(record))
|
|
994
|
+
projected.append(
|
|
995
|
+
{
|
|
996
|
+
"module_id": module_id,
|
|
997
|
+
"title": title,
|
|
998
|
+
"topics": topic_projection,
|
|
999
|
+
"modules": walk(children, module_titles, module_ids),
|
|
1000
|
+
}
|
|
1001
|
+
)
|
|
1002
|
+
return projected
|
|
1003
|
+
|
|
1004
|
+
modules = walk(payload["Modules"], (), ())
|
|
1005
|
+
records.sort(key=lambda record: record.source_key)
|
|
1006
|
+
return records, modules, valid
|
|
1007
|
+
|
|
1008
|
+
|
|
1009
|
+
def _topic_from_api(
|
|
1010
|
+
raw: object,
|
|
1011
|
+
*,
|
|
1012
|
+
course: CourseRef,
|
|
1013
|
+
school: School,
|
|
1014
|
+
module_titles: tuple[str, ...],
|
|
1015
|
+
module_ids: tuple[int, ...],
|
|
1016
|
+
) -> TopicRecord | None:
|
|
1017
|
+
if not isinstance(raw, dict):
|
|
1018
|
+
return None
|
|
1019
|
+
topic_id = raw.get("TopicId")
|
|
1020
|
+
title = raw.get("Title")
|
|
1021
|
+
kind = raw.get("TypeIdentifier")
|
|
1022
|
+
raw_url = raw.get("Url")
|
|
1023
|
+
if (
|
|
1024
|
+
isinstance(topic_id, bool)
|
|
1025
|
+
or not isinstance(topic_id, int)
|
|
1026
|
+
or not isinstance(title, str)
|
|
1027
|
+
or not isinstance(kind, str)
|
|
1028
|
+
or raw_url is not None
|
|
1029
|
+
and not isinstance(raw_url, str)
|
|
1030
|
+
):
|
|
1031
|
+
return None
|
|
1032
|
+
last_modified = raw.get("LastModifiedDate")
|
|
1033
|
+
if last_modified is not None and not isinstance(last_modified, str):
|
|
1034
|
+
last_modified = None
|
|
1035
|
+
is_broken = raw.get("IsBroken", False)
|
|
1036
|
+
if not isinstance(is_broken, bool):
|
|
1037
|
+
is_broken = False
|
|
1038
|
+
remote_size = raw.get("Size")
|
|
1039
|
+
if isinstance(remote_size, bool) or not isinstance(remote_size, int) or remote_size < 0:
|
|
1040
|
+
remote_size = None
|
|
1041
|
+
|
|
1042
|
+
key = f"{school.id}:{course.org_unit_id}:topic:{topic_id}"
|
|
1043
|
+
view_url = _view_url(school, course.org_unit_id, topic_id)
|
|
1044
|
+
external = bool(raw_url) and topic_is_excluded(kind, raw_url, school.topic_exclusion_policy())
|
|
1045
|
+
url_path = _first_party_path(raw_url, school.base_url)
|
|
1046
|
+
outline_url = _allowed_outline_url(raw_url, school)
|
|
1047
|
+
external_host = _safe_hostname(raw_url)
|
|
1048
|
+
if external or (raw_url and url_path is None and outline_url is None):
|
|
1049
|
+
return TopicRecord(
|
|
1050
|
+
source_key=key,
|
|
1051
|
+
source_id=str(topic_id),
|
|
1052
|
+
topic_id=topic_id,
|
|
1053
|
+
course_org_unit_id=course.org_unit_id,
|
|
1054
|
+
course_code=course.code,
|
|
1055
|
+
course_name=course.name,
|
|
1056
|
+
term=course.term,
|
|
1057
|
+
title=title,
|
|
1058
|
+
kind=kind,
|
|
1059
|
+
module_path=module_titles,
|
|
1060
|
+
module_ids=module_ids,
|
|
1061
|
+
view_url=view_url,
|
|
1062
|
+
outline_url=None,
|
|
1063
|
+
url_path=None,
|
|
1064
|
+
external_host=external_host,
|
|
1065
|
+
etag=_optional_text(raw.get("ETag")),
|
|
1066
|
+
last_modified=last_modified,
|
|
1067
|
+
is_broken=is_broken,
|
|
1068
|
+
availability="external_link",
|
|
1069
|
+
remote_size=remote_size,
|
|
1070
|
+
next_action="open the LEARN link manually",
|
|
1071
|
+
)
|
|
1072
|
+
|
|
1073
|
+
if is_broken:
|
|
1074
|
+
reason = "topic is marked broken in LEARN"
|
|
1075
|
+
elif url_path is None or kind.casefold() not in _DOWNLOADABLE_KINDS:
|
|
1076
|
+
reason = "fetch the topic explicitly"
|
|
1077
|
+
else:
|
|
1078
|
+
reason = f"a2l fetch {topic_id}"
|
|
1079
|
+
return TopicRecord(
|
|
1080
|
+
source_key=key,
|
|
1081
|
+
source_id=str(topic_id),
|
|
1082
|
+
topic_id=topic_id,
|
|
1083
|
+
course_org_unit_id=course.org_unit_id,
|
|
1084
|
+
course_code=course.code,
|
|
1085
|
+
course_name=course.name,
|
|
1086
|
+
term=course.term,
|
|
1087
|
+
title=title,
|
|
1088
|
+
kind=kind,
|
|
1089
|
+
module_path=module_titles,
|
|
1090
|
+
module_ids=module_ids,
|
|
1091
|
+
view_url=view_url,
|
|
1092
|
+
outline_url=outline_url,
|
|
1093
|
+
url_path=url_path,
|
|
1094
|
+
external_host=None,
|
|
1095
|
+
etag=_optional_text(raw.get("ETag")),
|
|
1096
|
+
last_modified=last_modified,
|
|
1097
|
+
is_broken=is_broken,
|
|
1098
|
+
availability="metadata_only",
|
|
1099
|
+
remote_size=remote_size,
|
|
1100
|
+
next_action=reason,
|
|
1101
|
+
)
|
|
1102
|
+
|
|
1103
|
+
|
|
1104
|
+
def _topic_projection(record: TopicRecord) -> dict[str, object]:
|
|
1105
|
+
return {
|
|
1106
|
+
"id": record.topic_id,
|
|
1107
|
+
"title": record.title,
|
|
1108
|
+
"kind": record.kind,
|
|
1109
|
+
"url_path": record.url_path,
|
|
1110
|
+
"external_host": record.external_host,
|
|
1111
|
+
"view_url": record.view_url,
|
|
1112
|
+
"last_modified": record.last_modified,
|
|
1113
|
+
"availability": record.availability,
|
|
1114
|
+
}
|
|
1115
|
+
|
|
1116
|
+
|
|
1117
|
+
def _merge_topic_records(
|
|
1118
|
+
existing: object,
|
|
1119
|
+
incoming: Sequence[TopicRecord],
|
|
1120
|
+
*,
|
|
1121
|
+
complete: bool,
|
|
1122
|
+
) -> list[dict[str, object]]:
|
|
1123
|
+
old_rows = existing if isinstance(existing, list) else []
|
|
1124
|
+
by_key: dict[str, dict[str, object]] = {}
|
|
1125
|
+
for value in old_rows:
|
|
1126
|
+
if isinstance(value, dict) and isinstance(value.get("source_key"), str):
|
|
1127
|
+
by_key[value["source_key"]] = dict(value)
|
|
1128
|
+
incoming_keys: set[str] = set()
|
|
1129
|
+
for record in incoming:
|
|
1130
|
+
incoming_keys.add(record.source_key)
|
|
1131
|
+
prior = by_key.get(record.source_key, {})
|
|
1132
|
+
row = _topic_to_row(record)
|
|
1133
|
+
for field in ("source_path", "path", "sha256", "size", "stub_path", "source_sha256"):
|
|
1134
|
+
if row.get(field) is None and prior.get(field) is not None:
|
|
1135
|
+
row[field] = prior[field]
|
|
1136
|
+
if prior.get("availability") in {
|
|
1137
|
+
"source_only",
|
|
1138
|
+
"markdown_ready",
|
|
1139
|
+
"unsupported_format",
|
|
1140
|
+
"conversion_gap",
|
|
1141
|
+
"integrity_gap",
|
|
1142
|
+
"download_gap",
|
|
1143
|
+
}:
|
|
1144
|
+
row["availability"] = prior["availability"]
|
|
1145
|
+
row["next_action"] = prior.get("next_action", row["next_action"])
|
|
1146
|
+
row["missing_since"] = None
|
|
1147
|
+
row["withdrawn_at"] = None
|
|
1148
|
+
by_key[record.source_key] = row
|
|
1149
|
+
|
|
1150
|
+
if complete:
|
|
1151
|
+
now = _now()
|
|
1152
|
+
for key, row in by_key.items():
|
|
1153
|
+
if key in incoming_keys:
|
|
1154
|
+
continue
|
|
1155
|
+
if row.get("missing_since") is None:
|
|
1156
|
+
row["missing_since"] = now
|
|
1157
|
+
elif row.get("withdrawn_at") is None:
|
|
1158
|
+
row["withdrawn_at"] = now
|
|
1159
|
+
return sorted(by_key.values(), key=lambda row: str(row.get("source_key", "")))
|
|
1160
|
+
|
|
1161
|
+
|
|
1162
|
+
def _topic_to_row(record: TopicRecord) -> dict[str, object]:
|
|
1163
|
+
return {
|
|
1164
|
+
"source_key": record.source_key,
|
|
1165
|
+
"source_id": record.source_id,
|
|
1166
|
+
"topic_id": record.topic_id,
|
|
1167
|
+
"course_org_unit_id": record.course_org_unit_id,
|
|
1168
|
+
"course_code": record.course_code,
|
|
1169
|
+
"course_name": record.course_name,
|
|
1170
|
+
"term": record.term,
|
|
1171
|
+
"title": record.title,
|
|
1172
|
+
"kind": record.kind,
|
|
1173
|
+
"module_path": list(record.module_path),
|
|
1174
|
+
"module_ids": list(record.module_ids),
|
|
1175
|
+
"view_url": record.view_url,
|
|
1176
|
+
"outline_url": record.outline_url,
|
|
1177
|
+
"url_path": record.url_path,
|
|
1178
|
+
"external_host": record.external_host,
|
|
1179
|
+
"etag": record.etag,
|
|
1180
|
+
"last_modified": record.last_modified,
|
|
1181
|
+
"is_broken": record.is_broken,
|
|
1182
|
+
"availability": record.availability,
|
|
1183
|
+
"source_path": record.source_path,
|
|
1184
|
+
"path": record.path,
|
|
1185
|
+
"sha256": record.sha256,
|
|
1186
|
+
"source_sha256": record.sha256,
|
|
1187
|
+
"size": record.size,
|
|
1188
|
+
"stub_path": record.stub_path,
|
|
1189
|
+
"remote_size": record.remote_size,
|
|
1190
|
+
"next_action": record.next_action,
|
|
1191
|
+
"missing_since": record.missing_since,
|
|
1192
|
+
"withdrawn_at": record.withdrawn_at,
|
|
1193
|
+
}
|
|
1194
|
+
|
|
1195
|
+
|
|
1196
|
+
def _topic_from_row(row: object, *, course: CourseRef) -> TopicRecord:
|
|
1197
|
+
if not isinstance(row, dict):
|
|
1198
|
+
raise A2LError("content_map contains an invalid topic row")
|
|
1199
|
+
source_key = row.get("source_key")
|
|
1200
|
+
source_id = row.get("source_id")
|
|
1201
|
+
topic_id = row.get("topic_id")
|
|
1202
|
+
if (
|
|
1203
|
+
not isinstance(source_key, str)
|
|
1204
|
+
or not isinstance(source_id, str)
|
|
1205
|
+
or isinstance(topic_id, bool)
|
|
1206
|
+
or not isinstance(topic_id, int)
|
|
1207
|
+
):
|
|
1208
|
+
raise A2LError("content_map contains an invalid topic identity")
|
|
1209
|
+
return TopicRecord(
|
|
1210
|
+
source_key=source_key,
|
|
1211
|
+
source_id=source_id,
|
|
1212
|
+
topic_id=topic_id,
|
|
1213
|
+
course_org_unit_id=course.org_unit_id,
|
|
1214
|
+
course_code=str(row.get("course_code", course.code)),
|
|
1215
|
+
course_name=str(row.get("course_name", course.name)),
|
|
1216
|
+
term=row.get("term") if isinstance(row.get("term"), str) else course.term,
|
|
1217
|
+
title=str(row.get("title", "untitled")),
|
|
1218
|
+
kind=str(row.get("kind", "File")),
|
|
1219
|
+
module_path=tuple(value for value in row.get("module_path", []) if isinstance(value, str)),
|
|
1220
|
+
module_ids=tuple(value for value in row.get("module_ids", []) if isinstance(value, int)),
|
|
1221
|
+
view_url=str(row.get("view_url", "")),
|
|
1222
|
+
outline_url=row.get("outline_url") if isinstance(row.get("outline_url"), str) else None,
|
|
1223
|
+
url_path=row.get("url_path") if isinstance(row.get("url_path"), str) else None,
|
|
1224
|
+
external_host=row.get("external_host")
|
|
1225
|
+
if isinstance(row.get("external_host"), str)
|
|
1226
|
+
else None,
|
|
1227
|
+
etag=row.get("etag") if isinstance(row.get("etag"), str) else None,
|
|
1228
|
+
last_modified=row.get("last_modified")
|
|
1229
|
+
if isinstance(row.get("last_modified"), str)
|
|
1230
|
+
else None,
|
|
1231
|
+
is_broken=bool(row.get("is_broken", False)),
|
|
1232
|
+
availability=str(row.get("availability", "metadata_only")),
|
|
1233
|
+
source_path=row.get("source_path") if isinstance(row.get("source_path"), str) else None,
|
|
1234
|
+
path=row.get("path") if isinstance(row.get("path"), str) else None,
|
|
1235
|
+
sha256=row.get("sha256") if isinstance(row.get("sha256"), str) else None,
|
|
1236
|
+
size=row.get("size") if isinstance(row.get("size"), int) else None,
|
|
1237
|
+
stub_path=row.get("stub_path") if isinstance(row.get("stub_path"), str) else None,
|
|
1238
|
+
remote_size=row.get("remote_size") if isinstance(row.get("remote_size"), int) else None,
|
|
1239
|
+
next_action=str(row.get("next_action", f"a2l fetch {source_id}")),
|
|
1240
|
+
missing_since=row.get("missing_since")
|
|
1241
|
+
if isinstance(row.get("missing_since"), str)
|
|
1242
|
+
else None,
|
|
1243
|
+
withdrawn_at=row.get("withdrawn_at") if isinstance(row.get("withdrawn_at"), str) else None,
|
|
1244
|
+
)
|
|
1245
|
+
|
|
1246
|
+
|
|
1247
|
+
def _assignment_attachment_topics(
|
|
1248
|
+
values: Sequence[object], *, course: CourseRef, school: School
|
|
1249
|
+
) -> list[TopicRecord]:
|
|
1250
|
+
"""Project only allowlisted first-party Dropbox attachments into the normal file pipeline."""
|
|
1251
|
+
|
|
1252
|
+
records: list[TopicRecord] = []
|
|
1253
|
+
for assignment in values:
|
|
1254
|
+
if (
|
|
1255
|
+
not isinstance(assignment, dict)
|
|
1256
|
+
or not isinstance(assignment.get("Id"), int)
|
|
1257
|
+
or isinstance(assignment.get("Id"), bool)
|
|
1258
|
+
):
|
|
1259
|
+
continue
|
|
1260
|
+
assignment_id = assignment["Id"]
|
|
1261
|
+
assignment_title = _safe_text(assignment.get("Name")) or f"Assignment {assignment_id}"
|
|
1262
|
+
for attachment in _attachment_values(assignment):
|
|
1263
|
+
if not isinstance(attachment, dict):
|
|
1264
|
+
continue
|
|
1265
|
+
raw_id = attachment.get("Id", attachment.get("FileId"))
|
|
1266
|
+
if raw_id is None:
|
|
1267
|
+
continue
|
|
1268
|
+
attachment_id = str(raw_id)
|
|
1269
|
+
source_id = f"{assignment_id}-{attachment_id}".replace(":", "_")
|
|
1270
|
+
raw_url = _first_string(attachment, "Url", "URL", "Href", "DownloadUrl")
|
|
1271
|
+
url_path = _first_party_path(raw_url, school.base_url)
|
|
1272
|
+
if url_path is None:
|
|
1273
|
+
continue
|
|
1274
|
+
title = (
|
|
1275
|
+
_first_string(attachment, "FileName", "Name", "Title")
|
|
1276
|
+
or f"attachment-{attachment_id}"
|
|
1277
|
+
)
|
|
1278
|
+
remote_size = attachment.get("Size")
|
|
1279
|
+
if isinstance(remote_size, bool) or not isinstance(remote_size, int) or remote_size < 0:
|
|
1280
|
+
remote_size = None
|
|
1281
|
+
topic_id = (
|
|
1282
|
+
raw_id
|
|
1283
|
+
if isinstance(raw_id, int) and not isinstance(raw_id, bool)
|
|
1284
|
+
else _stable_numeric_id(source_id)
|
|
1285
|
+
)
|
|
1286
|
+
records.append(
|
|
1287
|
+
TopicRecord(
|
|
1288
|
+
source_key=f"{school.id}:{course.org_unit_id}:attachment:{source_id}",
|
|
1289
|
+
source_id=source_id,
|
|
1290
|
+
topic_id=topic_id,
|
|
1291
|
+
course_org_unit_id=course.org_unit_id,
|
|
1292
|
+
course_code=course.code,
|
|
1293
|
+
course_name=course.name,
|
|
1294
|
+
term=course.term,
|
|
1295
|
+
title=title,
|
|
1296
|
+
kind="File",
|
|
1297
|
+
module_path=("Assignments", assignment_title),
|
|
1298
|
+
module_ids=(),
|
|
1299
|
+
view_url=_view_url(school, course.org_unit_id, topic_id),
|
|
1300
|
+
outline_url=None,
|
|
1301
|
+
url_path=url_path,
|
|
1302
|
+
external_host=None,
|
|
1303
|
+
etag=_first_string(attachment, "ETag", "Etag"),
|
|
1304
|
+
last_modified=_first_string(attachment, "LastModifiedDate", "LastModified"),
|
|
1305
|
+
is_broken=False,
|
|
1306
|
+
remote_size=remote_size,
|
|
1307
|
+
next_action=f"a2l fetch {source_id}",
|
|
1308
|
+
)
|
|
1309
|
+
)
|
|
1310
|
+
return sorted(records, key=lambda item: item.source_key)
|
|
1311
|
+
|
|
1312
|
+
|
|
1313
|
+
def _attachment_values(assignment: Mapping[str, object]) -> list[object]:
|
|
1314
|
+
raw = assignment.get("Attachments", assignment.get("attachments", []))
|
|
1315
|
+
if isinstance(raw, dict):
|
|
1316
|
+
raw = raw.get("Items", raw.get("Objects", []))
|
|
1317
|
+
return list(raw) if isinstance(raw, list) else []
|
|
1318
|
+
|
|
1319
|
+
|
|
1320
|
+
def _assignment_attachments_are_valid(assignment: Mapping[str, object]) -> bool:
|
|
1321
|
+
"""Reject malformed attachment containers before they can mark old files missing."""
|
|
1322
|
+
if "Attachments" not in assignment and "attachments" not in assignment:
|
|
1323
|
+
return True
|
|
1324
|
+
raw = assignment.get("Attachments", assignment.get("attachments"))
|
|
1325
|
+
if isinstance(raw, dict):
|
|
1326
|
+
raw = raw.get("Items", raw.get("Objects"))
|
|
1327
|
+
if not isinstance(raw, list):
|
|
1328
|
+
return False
|
|
1329
|
+
for attachment in raw:
|
|
1330
|
+
if not isinstance(attachment, dict):
|
|
1331
|
+
return False
|
|
1332
|
+
identifier = attachment.get("Id", attachment.get("FileId"))
|
|
1333
|
+
if isinstance(identifier, bool) or not isinstance(identifier, (int, str)):
|
|
1334
|
+
return False
|
|
1335
|
+
if isinstance(identifier, str) and not identifier:
|
|
1336
|
+
return False
|
|
1337
|
+
return True
|
|
1338
|
+
|
|
1339
|
+
|
|
1340
|
+
def _first_string(value: Mapping[str, object], *keys: str) -> str | None:
|
|
1341
|
+
for key in keys:
|
|
1342
|
+
candidate = value.get(key)
|
|
1343
|
+
if isinstance(candidate, str) and candidate:
|
|
1344
|
+
return candidate
|
|
1345
|
+
return None
|
|
1346
|
+
|
|
1347
|
+
|
|
1348
|
+
def _stable_numeric_id(value: str) -> int:
|
|
1349
|
+
return int.from_bytes(sha256(value.encode("utf-8")).digest()[:4], "big") & 0x7FFFFFFF
|
|
1350
|
+
|
|
1351
|
+
|
|
1352
|
+
def _materialize_assignments(
|
|
1353
|
+
values: Sequence[object],
|
|
1354
|
+
*,
|
|
1355
|
+
course_dir: Path,
|
|
1356
|
+
vault: Vault,
|
|
1357
|
+
school: School,
|
|
1358
|
+
course: CourseRef,
|
|
1359
|
+
directories: Mapping[str, Path],
|
|
1360
|
+
) -> dict[str, dict[str, object]]:
|
|
1361
|
+
"""Sanitize Dropbox RichText and persist a provenance-backed source/twin pair."""
|
|
1362
|
+
|
|
1363
|
+
artifacts: dict[str, dict[str, object]] = {}
|
|
1364
|
+
assignments = sorted(
|
|
1365
|
+
(value for value in values if isinstance(value, dict) and isinstance(value.get("Id"), int)),
|
|
1366
|
+
key=lambda value: int(value["Id"]),
|
|
1367
|
+
)
|
|
1368
|
+
for assignment in assignments:
|
|
1369
|
+
assignment_id = int(assignment["Id"])
|
|
1370
|
+
key = f"{school.id}:{course.org_unit_id}:dropbox:{assignment_id}"
|
|
1371
|
+
transactions.recover_generated(vault, key)
|
|
1372
|
+
richtext = _assignment_richtext(assignment)
|
|
1373
|
+
if richtext is None:
|
|
1374
|
+
continue
|
|
1375
|
+
raw_html, raw_text = richtext
|
|
1376
|
+
canonical_html = _sanitize_richtext(raw_html or raw_text or "", school.base_url)
|
|
1377
|
+
if not canonical_html.strip():
|
|
1378
|
+
continue
|
|
1379
|
+
html_bytes = canonical_html.encode("utf-8")
|
|
1380
|
+
source_hash = sha256(html_bytes).hexdigest()
|
|
1381
|
+
prior = vault.entry(key)
|
|
1382
|
+
title = _safe_text(assignment.get("Name")) or f"Assignment {assignment_id}"
|
|
1383
|
+
if prior is not None:
|
|
1384
|
+
source_destination = vault.materialized(prior)
|
|
1385
|
+
else:
|
|
1386
|
+
assignment_directory = directories[str(assignment_id)]
|
|
1387
|
+
source_destination = paths.unique_path(
|
|
1388
|
+
assignment_directory / "instructions.html", reserved=vault.claimed_paths()
|
|
1389
|
+
)
|
|
1390
|
+
|
|
1391
|
+
prior_artifact = prior.derived.get("markdown") if prior is not None else None
|
|
1392
|
+
if prior_artifact is not None:
|
|
1393
|
+
markdown_destination = vault.materialized(
|
|
1394
|
+
ManifestEntry(
|
|
1395
|
+
path=prior_artifact.path,
|
|
1396
|
+
sha256=prior_artifact.sha256,
|
|
1397
|
+
source_id="derived",
|
|
1398
|
+
etag=None,
|
|
1399
|
+
last_modified=None,
|
|
1400
|
+
size=0,
|
|
1401
|
+
fetched_at=_now(),
|
|
1402
|
+
)
|
|
1403
|
+
)
|
|
1404
|
+
else:
|
|
1405
|
+
markdown_destination = source_destination.with_suffix(".md")
|
|
1406
|
+
markdown_destination = vault.derived_destination(key, markdown_destination)
|
|
1407
|
+
|
|
1408
|
+
source_changed = prior is not None and prior.sha256 != source_hash
|
|
1409
|
+
twin_modified = bool(
|
|
1410
|
+
prior_artifact is not None
|
|
1411
|
+
and paths.long_path(markdown_destination).is_file()
|
|
1412
|
+
and _hash_file(markdown_destination)[0] != prior_artifact.sha256
|
|
1413
|
+
)
|
|
1414
|
+
markdown_bytes = _richtext_markdown(title, canonical_html).encode("utf-8")
|
|
1415
|
+
derived = DerivedArtifact(
|
|
1416
|
+
path=paths.rel_posix(markdown_destination, vault.root),
|
|
1417
|
+
sha256=sha256(markdown_bytes).hexdigest(),
|
|
1418
|
+
source_sha256=source_hash,
|
|
1419
|
+
tool="richtext-sanitizer",
|
|
1420
|
+
tool_version="1",
|
|
1421
|
+
created_at=_now(),
|
|
1422
|
+
)
|
|
1423
|
+
entry = ManifestEntry(
|
|
1424
|
+
path=paths.rel_posix(source_destination, vault.root),
|
|
1425
|
+
sha256=source_hash,
|
|
1426
|
+
source_id=str(assignment_id),
|
|
1427
|
+
etag=None,
|
|
1428
|
+
last_modified=_optional_text(assignment.get("LastModifiedDate")),
|
|
1429
|
+
size=len(html_bytes),
|
|
1430
|
+
fetched_at=_now(),
|
|
1431
|
+
derived={"markdown": derived},
|
|
1432
|
+
)
|
|
1433
|
+
transactions.install_generated(
|
|
1434
|
+
vault,
|
|
1435
|
+
key,
|
|
1436
|
+
entry,
|
|
1437
|
+
html_bytes,
|
|
1438
|
+
{"markdown": markdown_bytes},
|
|
1439
|
+
preserve=source_changed or twin_modified,
|
|
1440
|
+
)
|
|
1441
|
+
|
|
1442
|
+
_write_assignment_readme(
|
|
1443
|
+
source_destination.parent,
|
|
1444
|
+
title=title,
|
|
1445
|
+
entry=entry,
|
|
1446
|
+
attachments=_assignment_attachment_display(assignment, school),
|
|
1447
|
+
root=vault.root,
|
|
1448
|
+
)
|
|
1449
|
+
artifacts[str(assignment_id)] = {
|
|
1450
|
+
"instructions_html": entry.path,
|
|
1451
|
+
"instructions_md": derived.path,
|
|
1452
|
+
"instructions_sha256": source_hash,
|
|
1453
|
+
}
|
|
1454
|
+
return artifacts
|
|
1455
|
+
|
|
1456
|
+
|
|
1457
|
+
def _assignment_richtext(assignment: Mapping[str, object]) -> tuple[str | None, str | None] | None:
|
|
1458
|
+
for field in ("CustomInstructions", "Description", "Instructions", "RichText", "Body"):
|
|
1459
|
+
value = assignment.get(field)
|
|
1460
|
+
if isinstance(value, str):
|
|
1461
|
+
return value, None
|
|
1462
|
+
if isinstance(value, dict):
|
|
1463
|
+
html_value = _first_string(value, "Html", "HTML", "html")
|
|
1464
|
+
text_value = _first_string(value, "Text", "text")
|
|
1465
|
+
if html_value is not None or text_value is not None:
|
|
1466
|
+
return html_value, text_value
|
|
1467
|
+
return None
|
|
1468
|
+
|
|
1469
|
+
|
|
1470
|
+
def _assignment_attachment_display(
|
|
1471
|
+
assignment: Mapping[str, object], school: School
|
|
1472
|
+
) -> list[dict[str, str]]:
|
|
1473
|
+
rows: list[dict[str, str]] = []
|
|
1474
|
+
assignment_id = assignment.get("Id")
|
|
1475
|
+
for attachment in _attachment_values(assignment):
|
|
1476
|
+
if not isinstance(attachment, dict):
|
|
1477
|
+
continue
|
|
1478
|
+
raw_id = attachment.get("Id", attachment.get("FileId"))
|
|
1479
|
+
if raw_id is None:
|
|
1480
|
+
continue
|
|
1481
|
+
attachment_id = str(raw_id)
|
|
1482
|
+
name = (
|
|
1483
|
+
_first_string(attachment, "FileName", "Name", "Title") or f"attachment-{attachment_id}"
|
|
1484
|
+
)
|
|
1485
|
+
raw_url = _first_string(attachment, "Url", "URL", "Href", "DownloadUrl")
|
|
1486
|
+
path = _first_party_path(raw_url, school.base_url)
|
|
1487
|
+
if path is not None:
|
|
1488
|
+
rows.append(
|
|
1489
|
+
{
|
|
1490
|
+
"name": name,
|
|
1491
|
+
"action": f"a2l fetch {assignment_id}-{attachment_id}",
|
|
1492
|
+
}
|
|
1493
|
+
)
|
|
1494
|
+
else:
|
|
1495
|
+
host = _safe_hostname(raw_url) or "unknown-host"
|
|
1496
|
+
rows.append({"name": name, "action": f"external link · {host}"})
|
|
1497
|
+
return rows
|
|
1498
|
+
|
|
1499
|
+
|
|
1500
|
+
def _write_assignment_readme(
|
|
1501
|
+
directory: Path,
|
|
1502
|
+
*,
|
|
1503
|
+
title: str,
|
|
1504
|
+
entry: ManifestEntry,
|
|
1505
|
+
attachments: Sequence[Mapping[str, str]],
|
|
1506
|
+
root: Path | None = None,
|
|
1507
|
+
) -> None:
|
|
1508
|
+
markdown_path = entry.derived["markdown"].path
|
|
1509
|
+
lines = [
|
|
1510
|
+
f"# {title}",
|
|
1511
|
+
"",
|
|
1512
|
+
f"- Instructions source: `{PurePosixPath(entry.path).name}`",
|
|
1513
|
+
f"- Instructions twin: `{PurePosixPath(markdown_path).name}`",
|
|
1514
|
+
f"- Source SHA-256: `{entry.sha256}`",
|
|
1515
|
+
"",
|
|
1516
|
+
"## Attachments",
|
|
1517
|
+
"",
|
|
1518
|
+
]
|
|
1519
|
+
if attachments:
|
|
1520
|
+
lines.extend(f"- {row['name']} — {row['action']}" for row in attachments)
|
|
1521
|
+
else:
|
|
1522
|
+
lines.append("- None recorded.")
|
|
1523
|
+
lines.append("")
|
|
1524
|
+
paths.atomic_write_text(directory / "README.md", "\n".join(lines), root=root)
|
|
1525
|
+
|
|
1526
|
+
|
|
1527
|
+
def _materialize_submission_only_readmes(
|
|
1528
|
+
values: Sequence[object],
|
|
1529
|
+
*,
|
|
1530
|
+
artifacts: Mapping[str, Mapping[str, object]],
|
|
1531
|
+
course_dir: Path,
|
|
1532
|
+
directories: Mapping[str, Path],
|
|
1533
|
+
topics: Sequence[Mapping[str, object]],
|
|
1534
|
+
root: Path | None = None,
|
|
1535
|
+
) -> None:
|
|
1536
|
+
"""Give a Dropbox folder without RichText a generated navigation hub.
|
|
1537
|
+
|
|
1538
|
+
This is deliberately a metadata-only projection: it creates no fetch route and does not copy
|
|
1539
|
+
a prompt into the README. Exact display-title matches are only navigation cross-links; topic
|
|
1540
|
+
resolution itself remains stable-ID/manifest based in :mod:`agent2learn.index`.
|
|
1541
|
+
"""
|
|
1542
|
+
for assignment in sorted(
|
|
1543
|
+
(row for row in values if isinstance(row, Mapping) and isinstance(row.get("Id"), int)),
|
|
1544
|
+
key=lambda row: int(row["Id"]),
|
|
1545
|
+
):
|
|
1546
|
+
assignment_id = str(assignment["Id"])
|
|
1547
|
+
if assignment_id in artifacts:
|
|
1548
|
+
continue
|
|
1549
|
+
title = _safe_text(assignment.get("Name")) or f"Assignment {assignment_id}"
|
|
1550
|
+
directory = directories[assignment_id]
|
|
1551
|
+
paths.ensure_dir(directory, root=root)
|
|
1552
|
+
links: list[tuple[str, str]] = []
|
|
1553
|
+
for topic in topics:
|
|
1554
|
+
if str(topic.get("title", "")).casefold() != title.casefold():
|
|
1555
|
+
continue
|
|
1556
|
+
target = topic.get("path") or topic.get("source_path") or topic.get("stub_path")
|
|
1557
|
+
source_id = topic.get("source_id")
|
|
1558
|
+
if isinstance(target, str) and isinstance(source_id, str):
|
|
1559
|
+
links.append((source_id, _course_relative_link(target, course_dir)))
|
|
1560
|
+
course_index.write_submission_readme(directory, title=title, content_links=links, root=root)
|
|
1561
|
+
|
|
1562
|
+
|
|
1563
|
+
def _materialize_external_stubs(
|
|
1564
|
+
rows: list[dict[str, object]],
|
|
1565
|
+
*,
|
|
1566
|
+
course_dir: Path,
|
|
1567
|
+
vault: Vault,
|
|
1568
|
+
school: School,
|
|
1569
|
+
course: CourseRef,
|
|
1570
|
+
) -> list[dict[str, object]]:
|
|
1571
|
+
reserved: set[str] = set()
|
|
1572
|
+
for row in sorted(rows, key=lambda value: str(value.get("source_key", ""))):
|
|
1573
|
+
if row.get("availability") != "external_link":
|
|
1574
|
+
continue
|
|
1575
|
+
prior = row.get("stub_path")
|
|
1576
|
+
if isinstance(prior, str):
|
|
1577
|
+
stub = _vault_relative_path(vault, prior)
|
|
1578
|
+
else:
|
|
1579
|
+
record = _topic_from_row(row, course=course)
|
|
1580
|
+
destination = _content_directory(course_dir, record.module_path) / (
|
|
1581
|
+
f"{paths.safe_name(record.title)}.url.txt"
|
|
1582
|
+
)
|
|
1583
|
+
stub = _unique_reserved(destination, reserved)
|
|
1584
|
+
row["stub_path"] = paths.rel_posix(stub, vault.root)
|
|
1585
|
+
paths.ensure_dir(stub.parent, root=vault.root)
|
|
1586
|
+
if paths.is_link(stub):
|
|
1587
|
+
raise A2LError("external-topic stub path contains a symlink or junction")
|
|
1588
|
+
if not paths.long_path(stub).is_file():
|
|
1589
|
+
record = _topic_from_row(row, course=course)
|
|
1590
|
+
text = (
|
|
1591
|
+
"Agent2Learn external-topic stub\n"
|
|
1592
|
+
f"topic: {record.title}\n"
|
|
1593
|
+
f"view in LEARN: {record.view_url}\n"
|
|
1594
|
+
f"destination host: {record.external_host or 'unknown-host'}\n"
|
|
1595
|
+
)
|
|
1596
|
+
# Keep the ordinary path as the value passed between layers; long_path belongs only
|
|
1597
|
+
# at filesystem boundaries, and atomic_write_text applies it internally.
|
|
1598
|
+
paths.atomic_write_text(stub, text, root=vault.root)
|
|
1599
|
+
return rows
|
|
1600
|
+
|
|
1601
|
+
|
|
1602
|
+
def _plan_file_paths(
|
|
1603
|
+
rows: Sequence[TopicRecord], *, course_dir: Path, vault: Vault, scope: str
|
|
1604
|
+
) -> list[TopicRecord]:
|
|
1605
|
+
del scope
|
|
1606
|
+
manifest = Vault(vault.root).manifest()
|
|
1607
|
+
installed = _installed_pending_paths(
|
|
1608
|
+
vault, course_dir, {row.source_key for row in rows if row.source_key not in manifest}
|
|
1609
|
+
)
|
|
1610
|
+
reserved_by_folder: dict[str, set[str]] = defaultdict(set)
|
|
1611
|
+
for reserved_entry in manifest.values():
|
|
1612
|
+
for relative in (
|
|
1613
|
+
reserved_entry.path,
|
|
1614
|
+
*(artifact.path for artifact in reserved_entry.derived.values()),
|
|
1615
|
+
):
|
|
1616
|
+
occupied = vault.root / PurePosixPath(relative)
|
|
1617
|
+
reserved_by_folder[occupied.parent.as_posix()].add(occupied.name.casefold())
|
|
1618
|
+
planned: list[TopicRecord] = []
|
|
1619
|
+
for topic in sorted(rows, key=lambda item: item.source_key):
|
|
1620
|
+
if topic.availability == "external_link":
|
|
1621
|
+
planned.append(topic)
|
|
1622
|
+
continue
|
|
1623
|
+
entry = manifest.get(topic.source_key)
|
|
1624
|
+
if entry is not None:
|
|
1625
|
+
planned.append(
|
|
1626
|
+
replace(topic, source_path=entry.path, path=_derived_path(manifest, entry))
|
|
1627
|
+
)
|
|
1628
|
+
continue
|
|
1629
|
+
if topic.url_path is None or topic.kind.casefold() not in _DOWNLOADABLE_KINDS:
|
|
1630
|
+
planned.append(topic)
|
|
1631
|
+
continue
|
|
1632
|
+
if topic.source_key in installed:
|
|
1633
|
+
planned.append(
|
|
1634
|
+
replace(topic, source_path=paths.rel_posix(installed[topic.source_key], vault.root))
|
|
1635
|
+
)
|
|
1636
|
+
continue
|
|
1637
|
+
destination = _content_directory(course_dir, topic.module_path) / _topic_filename(topic)
|
|
1638
|
+
folder_key = destination.parent.as_posix()
|
|
1639
|
+
candidate = _unique_reserved(destination, reserved_by_folder[folder_key])
|
|
1640
|
+
planned.append(replace(topic, source_path=paths.rel_posix(candidate, vault.root)))
|
|
1641
|
+
return planned
|
|
1642
|
+
|
|
1643
|
+
|
|
1644
|
+
def _priority_rows(
|
|
1645
|
+
rows: Sequence[TopicRecord], *, scope: Literal["all", "priority"], budget: int
|
|
1646
|
+
) -> list[TopicRecord]:
|
|
1647
|
+
if scope == "all":
|
|
1648
|
+
return list(rows)
|
|
1649
|
+
ranked = sorted(
|
|
1650
|
+
rows,
|
|
1651
|
+
key=lambda topic: (
|
|
1652
|
+
0 if _is_priority(topic) else 1,
|
|
1653
|
+
0 if topic.last_modified is not None else 1,
|
|
1654
|
+
-(int(_timestamp_sort(topic.last_modified)) if topic.last_modified else 0),
|
|
1655
|
+
topic.source_key,
|
|
1656
|
+
),
|
|
1657
|
+
)
|
|
1658
|
+
selected: list[TopicRecord] = []
|
|
1659
|
+
total = 0
|
|
1660
|
+
for topic in ranked:
|
|
1661
|
+
# An unknown size cannot be charged against a hard byte budget without inventing a
|
|
1662
|
+
# bound. Leave it for the full plan (which still has a per-file ceiling) or explicit
|
|
1663
|
+
# ``a2l fetch``; priority must remain provably byte-bounded.
|
|
1664
|
+
if topic.remote_size is None:
|
|
1665
|
+
continue
|
|
1666
|
+
if total + topic.remote_size > budget:
|
|
1667
|
+
continue
|
|
1668
|
+
total += topic.remote_size
|
|
1669
|
+
selected.append(topic)
|
|
1670
|
+
return selected
|
|
1671
|
+
|
|
1672
|
+
|
|
1673
|
+
def _is_priority(topic: TopicRecord) -> bool:
|
|
1674
|
+
value = f"{topic.title} {'/'.join(topic.module_path)}".casefold()
|
|
1675
|
+
return "assignment" in value or "outline" in value or "syllabus" in value
|
|
1676
|
+
|
|
1677
|
+
|
|
1678
|
+
def _ingest_one_topic(
|
|
1679
|
+
client: IngestClient,
|
|
1680
|
+
vault: Vault,
|
|
1681
|
+
school: School,
|
|
1682
|
+
course_dir: Path,
|
|
1683
|
+
topic: TopicRecord,
|
|
1684
|
+
*,
|
|
1685
|
+
max_bytes: int | None = api.DEFAULT_MAX_BYTES,
|
|
1686
|
+
) -> Literal["downloaded", "skipped"]:
|
|
1687
|
+
if topic.url_path is None:
|
|
1688
|
+
raise DownloadError("topic has no first-party download route")
|
|
1689
|
+
key = topic.source_key
|
|
1690
|
+
transactions.recover_generated(vault, key)
|
|
1691
|
+
manifest = vault.manifest()
|
|
1692
|
+
prior = manifest.get(key)
|
|
1693
|
+
destination = _destination_for_topic(vault, course_dir, topic, prior)
|
|
1694
|
+
paths.ensure_dir(destination.parent, root=vault.root)
|
|
1695
|
+
pending = _find_pending_install(vault, destination, topic)
|
|
1696
|
+
if pending is not None:
|
|
1697
|
+
persisted = Vault(vault.root).entry(key)
|
|
1698
|
+
retried = _retry_pending_install(
|
|
1699
|
+
vault, course_dir, school, topic, destination, persisted, pending
|
|
1700
|
+
)
|
|
1701
|
+
prior = vault.entry(key)
|
|
1702
|
+
if retried is not None:
|
|
1703
|
+
return retried
|
|
1704
|
+
if prior is not None and _unchanged_local(prior, topic, vault):
|
|
1705
|
+
_mark_topic_source_only(vault, course_dir, topic, school)
|
|
1706
|
+
return "skipped"
|
|
1707
|
+
|
|
1708
|
+
# Revalidate immediately before reserving the download sibling; the network and manifest
|
|
1709
|
+
# checks above must not leave a race window where a replaced parent redirects the part.
|
|
1710
|
+
if paths.has_link_component(destination.parent, root=vault.root):
|
|
1711
|
+
raise A2LError("download parent contains a link component")
|
|
1712
|
+
fd, raw_temp = tempfile.mkstemp(
|
|
1713
|
+
prefix=f".{destination.name}.",
|
|
1714
|
+
suffix=".part",
|
|
1715
|
+
dir=os.fspath(paths.long_path(destination.parent)),
|
|
1716
|
+
)
|
|
1717
|
+
os.close(fd)
|
|
1718
|
+
temporary = paths.plain_path(Path(raw_temp))
|
|
1719
|
+
install_attempted = False
|
|
1720
|
+
installed = False
|
|
1721
|
+
try:
|
|
1722
|
+
# ``mkstemp`` is intentionally used to reserve the sibling before the network call. The
|
|
1723
|
+
# reservation is a separate filesystem boundary, so revalidate it before any download or
|
|
1724
|
+
# later writer can use a parent that was swapped to a link in the meantime.
|
|
1725
|
+
if paths.has_link_component(temporary, root=vault.root):
|
|
1726
|
+
raise A2LError("download temporary path contains a link component")
|
|
1727
|
+
conditional = prior if prior is not None and _source_bytes_match(prior, vault) else None
|
|
1728
|
+
result = _download_with_candidates(
|
|
1729
|
+
client,
|
|
1730
|
+
school,
|
|
1731
|
+
topic,
|
|
1732
|
+
temporary,
|
|
1733
|
+
prior=conditional,
|
|
1734
|
+
max_bytes=max_bytes,
|
|
1735
|
+
root=vault.root,
|
|
1736
|
+
)
|
|
1737
|
+
if result.not_modified:
|
|
1738
|
+
if prior is not None and _source_bytes_match(prior, vault):
|
|
1739
|
+
_mark_topic_source_only(vault, course_dir, topic, school)
|
|
1740
|
+
return "skipped"
|
|
1741
|
+
result = _download_with_candidates(
|
|
1742
|
+
client, school, topic, temporary, prior=None, max_bytes=max_bytes, root=vault.root
|
|
1743
|
+
)
|
|
1744
|
+
if result.not_modified:
|
|
1745
|
+
raise DownloadError("server returned 304 without a verified local source")
|
|
1746
|
+
if result.temp is None or not paths.long_path(result.temp).is_file():
|
|
1747
|
+
raise DownloadError("download did not produce a source file")
|
|
1748
|
+
actual_hash, actual_size = _hash_file(result.temp)
|
|
1749
|
+
if actual_size <= 0 or result.sha256 != actual_hash or result.size != actual_size:
|
|
1750
|
+
raise DownloadError("download integrity validation failed")
|
|
1751
|
+
pending = _PendingInstall(
|
|
1752
|
+
marker=_pending_marker_path(temporary),
|
|
1753
|
+
part=temporary,
|
|
1754
|
+
source_key=key,
|
|
1755
|
+
destination=paths.rel_posix(destination, vault.root),
|
|
1756
|
+
sha256=actual_hash,
|
|
1757
|
+
size=actual_size,
|
|
1758
|
+
etag=result.etag or topic.etag or (prior.etag if prior else None),
|
|
1759
|
+
last_modified=result.last_modified
|
|
1760
|
+
or topic.last_modified
|
|
1761
|
+
or (prior.last_modified if prior else None),
|
|
1762
|
+
prior_sha256=(
|
|
1763
|
+
prior.sha256 if prior is not None and actual_hash != prior.sha256 else None
|
|
1764
|
+
),
|
|
1765
|
+
revision_preserved=prior is None or actual_hash == prior.sha256,
|
|
1766
|
+
fetched_at=_now(),
|
|
1767
|
+
)
|
|
1768
|
+
_write_pending_install(pending, root=vault.root)
|
|
1769
|
+
install_attempted = True
|
|
1770
|
+
if pending.prior_sha256 is not None:
|
|
1771
|
+
if prior is None:
|
|
1772
|
+
raise A2LError("pending source revision has no current manifest entry")
|
|
1773
|
+
preserved = vault.preserve_revision(key, changed_at=clock.now())
|
|
1774
|
+
if preserved is None and paths.long_path(vault.materialized(prior)).exists():
|
|
1775
|
+
raise A2LError("current source could not be preserved; refusing replacement")
|
|
1776
|
+
pending = replace(pending, revision_preserved=True)
|
|
1777
|
+
_write_pending_install(pending, root=vault.root)
|
|
1778
|
+
|
|
1779
|
+
paths.atomic_install_temp(destination, temporary, root=vault.root)
|
|
1780
|
+
installed = True
|
|
1781
|
+
entry = _manifest_entry_for_install(
|
|
1782
|
+
vault,
|
|
1783
|
+
topic,
|
|
1784
|
+
destination=destination,
|
|
1785
|
+
prior=prior,
|
|
1786
|
+
sha256=actual_hash,
|
|
1787
|
+
size=actual_size,
|
|
1788
|
+
etag=pending.etag,
|
|
1789
|
+
last_modified=pending.last_modified,
|
|
1790
|
+
fetched_at=pending.fetched_at,
|
|
1791
|
+
)
|
|
1792
|
+
vault.mark(key, entry)
|
|
1793
|
+
vault.save_manifest()
|
|
1794
|
+
_mark_topic_source_only(vault, course_dir, topic, school)
|
|
1795
|
+
_remove_pending_install(pending)
|
|
1796
|
+
return "downloaded"
|
|
1797
|
+
except BaseException:
|
|
1798
|
+
# A failed transfer has an incomplete part and may be restarted from byte zero. A
|
|
1799
|
+
# completed part whose *install* failed is different: paths.atomic_install_temp owns the
|
|
1800
|
+
# deliberate retention guarantee, so do not clean it here once installation was attempted.
|
|
1801
|
+
if not install_attempted:
|
|
1802
|
+
_remove_quietly(temporary)
|
|
1803
|
+
raise
|
|
1804
|
+
finally:
|
|
1805
|
+
if installed:
|
|
1806
|
+
_remove_quietly(temporary)
|
|
1807
|
+
|
|
1808
|
+
|
|
1809
|
+
def _pending_marker_path(part: Path) -> Path:
|
|
1810
|
+
return part.with_name(part.name + _PENDING_MARKER_SUFFIX)
|
|
1811
|
+
|
|
1812
|
+
|
|
1813
|
+
def _write_pending_install(pending: _PendingInstall, *, root: Path | None = None) -> None:
|
|
1814
|
+
payload = {
|
|
1815
|
+
"version": 2,
|
|
1816
|
+
"fetched_at": pending.fetched_at or _now(),
|
|
1817
|
+
"source_key": pending.source_key,
|
|
1818
|
+
"destination": pending.destination,
|
|
1819
|
+
"sha256": pending.sha256,
|
|
1820
|
+
"size": pending.size,
|
|
1821
|
+
"etag": pending.etag,
|
|
1822
|
+
"last_modified": pending.last_modified,
|
|
1823
|
+
"prior_sha256": pending.prior_sha256,
|
|
1824
|
+
"revision_preserved": pending.revision_preserved,
|
|
1825
|
+
}
|
|
1826
|
+
paths.atomic_write_text(
|
|
1827
|
+
pending.marker,
|
|
1828
|
+
json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")) + "\n",
|
|
1829
|
+
root=root,
|
|
1830
|
+
)
|
|
1831
|
+
|
|
1832
|
+
|
|
1833
|
+
def _read_pending_install(marker: Path, part: Path) -> _PendingInstall | None:
|
|
1834
|
+
if not _is_safe_local_file(marker):
|
|
1835
|
+
return None
|
|
1836
|
+
try:
|
|
1837
|
+
with open(os.fspath(paths.long_path(marker)), encoding="utf-8", newline="") as handle:
|
|
1838
|
+
raw: Any = json.load(handle)
|
|
1839
|
+
except (OSError, json.JSONDecodeError, UnicodeError):
|
|
1840
|
+
return None
|
|
1841
|
+
if not isinstance(raw, dict):
|
|
1842
|
+
return None
|
|
1843
|
+
version = raw.get("version")
|
|
1844
|
+
if isinstance(version, bool) or not isinstance(version, int) or version not in {1, 2}:
|
|
1845
|
+
return None
|
|
1846
|
+
expected_keys = _PENDING_INSTALL_KEYS | ({"fetched_at"} if version == 2 else set())
|
|
1847
|
+
if set(raw) != expected_keys:
|
|
1848
|
+
return None
|
|
1849
|
+
fetched_at = raw.get("fetched_at")
|
|
1850
|
+
if version == 2:
|
|
1851
|
+
if not isinstance(fetched_at, str):
|
|
1852
|
+
return None
|
|
1853
|
+
try:
|
|
1854
|
+
parse_api_timestamp(fetched_at)
|
|
1855
|
+
except (TypeError, ValueError):
|
|
1856
|
+
return None
|
|
1857
|
+
source_key = raw.get("source_key")
|
|
1858
|
+
destination = raw.get("destination")
|
|
1859
|
+
sha256_value = raw.get("sha256")
|
|
1860
|
+
size = raw.get("size")
|
|
1861
|
+
etag = raw.get("etag")
|
|
1862
|
+
last_modified = raw.get("last_modified")
|
|
1863
|
+
prior_sha256 = raw.get("prior_sha256")
|
|
1864
|
+
revision_preserved = raw.get("revision_preserved")
|
|
1865
|
+
if (
|
|
1866
|
+
not isinstance(source_key, str)
|
|
1867
|
+
or not source_key
|
|
1868
|
+
or not isinstance(destination, str)
|
|
1869
|
+
or not destination
|
|
1870
|
+
or not isinstance(sha256_value, str)
|
|
1871
|
+
or re.fullmatch(r"[0-9a-f]{64}", sha256_value) is None
|
|
1872
|
+
or isinstance(size, bool)
|
|
1873
|
+
or not isinstance(size, int)
|
|
1874
|
+
or size <= 0
|
|
1875
|
+
or (etag is not None and not isinstance(etag, str))
|
|
1876
|
+
or (last_modified is not None and not isinstance(last_modified, str))
|
|
1877
|
+
or (
|
|
1878
|
+
prior_sha256 is not None
|
|
1879
|
+
and (
|
|
1880
|
+
not isinstance(prior_sha256, str)
|
|
1881
|
+
or re.fullmatch(r"[0-9a-f]{64}", prior_sha256) is None
|
|
1882
|
+
)
|
|
1883
|
+
)
|
|
1884
|
+
or not isinstance(revision_preserved, bool)
|
|
1885
|
+
):
|
|
1886
|
+
return None
|
|
1887
|
+
return _PendingInstall(
|
|
1888
|
+
marker=marker,
|
|
1889
|
+
part=part,
|
|
1890
|
+
source_key=source_key,
|
|
1891
|
+
destination=destination,
|
|
1892
|
+
sha256=sha256_value,
|
|
1893
|
+
size=size,
|
|
1894
|
+
etag=etag,
|
|
1895
|
+
last_modified=last_modified,
|
|
1896
|
+
prior_sha256=prior_sha256,
|
|
1897
|
+
revision_preserved=revision_preserved,
|
|
1898
|
+
fetched_at=fetched_at,
|
|
1899
|
+
)
|
|
1900
|
+
|
|
1901
|
+
|
|
1902
|
+
def _is_safe_local_file(path: Path) -> bool:
|
|
1903
|
+
try:
|
|
1904
|
+
file_stat = os.lstat(os.fspath(paths.long_path(path)))
|
|
1905
|
+
except OSError:
|
|
1906
|
+
return False
|
|
1907
|
+
return (
|
|
1908
|
+
not paths.is_link(path)
|
|
1909
|
+
and stat.S_ISREG(file_stat.st_mode)
|
|
1910
|
+
and getattr(file_stat, "st_nlink", 1) == 1
|
|
1911
|
+
)
|
|
1912
|
+
|
|
1913
|
+
|
|
1914
|
+
def _remove_pending_install(pending: _PendingInstall) -> None:
|
|
1915
|
+
# The marker is removed first so an orphaned part is never treated as a validated download.
|
|
1916
|
+
_remove_pending_paths(pending.marker, pending.part)
|
|
1917
|
+
|
|
1918
|
+
|
|
1919
|
+
def _remove_pending_paths(marker: Path, part: Path) -> None:
|
|
1920
|
+
paths.remove_tree(marker, ignore_errors=True)
|
|
1921
|
+
paths.remove_tree(part, ignore_errors=True)
|
|
1922
|
+
|
|
1923
|
+
|
|
1924
|
+
def _pending_matches_topic(pending: _PendingInstall, topic: TopicRecord) -> bool:
|
|
1925
|
+
# A stable key and matching byte count do not prove that a remote file with no validator is
|
|
1926
|
+
# unchanged. Revalidate it over the network rather than replaying a potentially stale part.
|
|
1927
|
+
if topic.etag is None and topic.last_modified is None:
|
|
1928
|
+
return False
|
|
1929
|
+
if topic.etag is not None and pending.etag != topic.etag:
|
|
1930
|
+
return False
|
|
1931
|
+
if topic.last_modified is not None and pending.last_modified != topic.last_modified:
|
|
1932
|
+
return False
|
|
1933
|
+
return topic.remote_size is None or pending.size == topic.remote_size
|
|
1934
|
+
|
|
1935
|
+
|
|
1936
|
+
def _installed_pending_matches(pending: _PendingInstall, destination: Path) -> bool:
|
|
1937
|
+
return (
|
|
1938
|
+
pending.revision_preserved
|
|
1939
|
+
and not paths.collides(pending.part)
|
|
1940
|
+
and _is_safe_local_file(destination)
|
|
1941
|
+
and _hash_file(destination) == (pending.sha256, pending.size)
|
|
1942
|
+
)
|
|
1943
|
+
|
|
1944
|
+
|
|
1945
|
+
def _installed_pending_paths(vault: Vault, course_dir: Path, wanted: set[str]) -> dict[str, Path]:
|
|
1946
|
+
if not wanted:
|
|
1947
|
+
return {}
|
|
1948
|
+
content = course_dir / _COURSE_CONTENT
|
|
1949
|
+
if paths.has_link_component(content, root=vault.root):
|
|
1950
|
+
raise A2LError("pending download directory contains a link component")
|
|
1951
|
+
if not paths.long_path(content).is_dir():
|
|
1952
|
+
return {}
|
|
1953
|
+
result: dict[str, Path] = {}
|
|
1954
|
+
for marker in paths.walk(content):
|
|
1955
|
+
if not marker.name.startswith(".") or not marker.name.endswith(_PENDING_INSTALL_SUFFIX):
|
|
1956
|
+
continue
|
|
1957
|
+
part = marker.with_name(marker.name[: -len(_PENDING_MARKER_SUFFIX)])
|
|
1958
|
+
pending = _read_pending_install(marker, part)
|
|
1959
|
+
if pending is None or pending.source_key not in wanted or pending.prior_sha256 is not None:
|
|
1960
|
+
continue
|
|
1961
|
+
try:
|
|
1962
|
+
destination = _vault_relative_path(vault, pending.destination)
|
|
1963
|
+
except (A2LError, ValueError):
|
|
1964
|
+
continue
|
|
1965
|
+
if destination.parent != marker.parent or not destination.is_relative_to(content):
|
|
1966
|
+
continue
|
|
1967
|
+
if not marker.name.startswith(f".{destination.name}."):
|
|
1968
|
+
continue
|
|
1969
|
+
if not _installed_pending_matches(pending, destination):
|
|
1970
|
+
continue
|
|
1971
|
+
if pending.source_key in result and result[pending.source_key] != destination:
|
|
1972
|
+
raise A2LError("multiple installed revisions require recovery for one source")
|
|
1973
|
+
result[pending.source_key] = destination
|
|
1974
|
+
return result
|
|
1975
|
+
|
|
1976
|
+
|
|
1977
|
+
def _find_pending_install(
|
|
1978
|
+
vault: Vault, destination: Path, topic: TopicRecord
|
|
1979
|
+
) -> _PendingInstall | None:
|
|
1980
|
+
"""Find a validated install left by a prior sync, rejecting stale or untrusted markers."""
|
|
1981
|
+
expected_destination = paths.rel_posix(destination, vault.root)
|
|
1982
|
+
prefix = f".{destination.name}."
|
|
1983
|
+
try:
|
|
1984
|
+
with os.scandir(os.fspath(paths.long_path(destination.parent))) as iterator:
|
|
1985
|
+
candidates = sorted(entry.name for entry in iterator)
|
|
1986
|
+
except FileNotFoundError:
|
|
1987
|
+
return None
|
|
1988
|
+
except OSError as exc:
|
|
1989
|
+
raise A2LError("pending download state is unreadable") from exc
|
|
1990
|
+
|
|
1991
|
+
for name in candidates:
|
|
1992
|
+
if not name.startswith(prefix) or not name.endswith(_PENDING_INSTALL_SUFFIX):
|
|
1993
|
+
continue
|
|
1994
|
+
marker = destination.parent / name
|
|
1995
|
+
part = destination.parent / name[: -len(_PENDING_MARKER_SUFFIX)]
|
|
1996
|
+
pending = _read_pending_install(marker, part)
|
|
1997
|
+
if pending is None:
|
|
1998
|
+
_remove_pending_paths(marker, part)
|
|
1999
|
+
continue
|
|
2000
|
+
if pending.destination == expected_destination and pending.source_key != topic.source_key:
|
|
2001
|
+
_remove_pending_install(pending)
|
|
2002
|
+
continue
|
|
2003
|
+
if pending.source_key != topic.source_key or pending.destination != expected_destination:
|
|
2004
|
+
continue
|
|
2005
|
+
if _installed_pending_matches(pending, destination):
|
|
2006
|
+
return pending
|
|
2007
|
+
if not _pending_matches_topic(pending, topic) or not _is_safe_local_file(pending.part):
|
|
2008
|
+
_remove_pending_install(pending)
|
|
2009
|
+
continue
|
|
2010
|
+
actual_hash, actual_size = _hash_file(pending.part)
|
|
2011
|
+
if actual_hash != pending.sha256 or actual_size != pending.size:
|
|
2012
|
+
_remove_pending_install(pending)
|
|
2013
|
+
continue
|
|
2014
|
+
return pending
|
|
2015
|
+
return None
|
|
2016
|
+
|
|
2017
|
+
|
|
2018
|
+
def _retry_pending_install(
|
|
2019
|
+
vault: Vault,
|
|
2020
|
+
course_dir: Path,
|
|
2021
|
+
school: School,
|
|
2022
|
+
topic: TopicRecord,
|
|
2023
|
+
destination: Path,
|
|
2024
|
+
prior: ManifestEntry | None,
|
|
2025
|
+
pending: _PendingInstall,
|
|
2026
|
+
) -> Literal["downloaded"] | None:
|
|
2027
|
+
"""Install a previously validated part; return ``None`` when it is stale and was removed."""
|
|
2028
|
+
already_installed = _installed_pending_matches(pending, destination)
|
|
2029
|
+
if already_installed and prior is not None and prior.sha256 == pending.sha256:
|
|
2030
|
+
vault.mark(topic.source_key, prior)
|
|
2031
|
+
_remove_pending_install(pending)
|
|
2032
|
+
return None
|
|
2033
|
+
if pending.prior_sha256 is None:
|
|
2034
|
+
if prior is not None and pending.sha256 != prior.sha256:
|
|
2035
|
+
_remove_pending_install(pending)
|
|
2036
|
+
return None
|
|
2037
|
+
elif prior is None or pending.prior_sha256 != prior.sha256:
|
|
2038
|
+
_remove_pending_install(pending)
|
|
2039
|
+
return None
|
|
2040
|
+
|
|
2041
|
+
if pending.prior_sha256 is not None and not pending.revision_preserved:
|
|
2042
|
+
if prior is None:
|
|
2043
|
+
_remove_pending_install(pending)
|
|
2044
|
+
return None
|
|
2045
|
+
preserved = vault.preserve_revision(topic.source_key, changed_at=clock.now())
|
|
2046
|
+
if preserved is None and paths.long_path(vault.materialized(prior)).exists():
|
|
2047
|
+
raise A2LError("current source could not be preserved; refusing replacement")
|
|
2048
|
+
pending = replace(pending, revision_preserved=True)
|
|
2049
|
+
_write_pending_install(pending, root=vault.root)
|
|
2050
|
+
|
|
2051
|
+
if not already_installed:
|
|
2052
|
+
paths.atomic_install_temp(destination, pending.part, root=vault.root)
|
|
2053
|
+
entry = _manifest_entry_for_install(
|
|
2054
|
+
vault,
|
|
2055
|
+
topic,
|
|
2056
|
+
destination=destination,
|
|
2057
|
+
prior=prior,
|
|
2058
|
+
sha256=pending.sha256,
|
|
2059
|
+
size=pending.size,
|
|
2060
|
+
etag=pending.etag,
|
|
2061
|
+
last_modified=pending.last_modified,
|
|
2062
|
+
fetched_at=pending.fetched_at,
|
|
2063
|
+
)
|
|
2064
|
+
vault.mark(topic.source_key, entry)
|
|
2065
|
+
vault.save_manifest()
|
|
2066
|
+
_mark_topic_source_only(vault, course_dir, topic, school)
|
|
2067
|
+
_remove_pending_install(pending)
|
|
2068
|
+
return (
|
|
2069
|
+
None if already_installed and not _pending_matches_topic(pending, topic) else "downloaded"
|
|
2070
|
+
)
|
|
2071
|
+
|
|
2072
|
+
|
|
2073
|
+
def _manifest_entry_for_install(
|
|
2074
|
+
vault: Vault,
|
|
2075
|
+
topic: TopicRecord,
|
|
2076
|
+
*,
|
|
2077
|
+
destination: Path,
|
|
2078
|
+
prior: ManifestEntry | None,
|
|
2079
|
+
sha256: str,
|
|
2080
|
+
size: int,
|
|
2081
|
+
etag: str | None,
|
|
2082
|
+
last_modified: str | None,
|
|
2083
|
+
fetched_at: str | None = None,
|
|
2084
|
+
) -> ManifestEntry:
|
|
2085
|
+
return ManifestEntry(
|
|
2086
|
+
path=paths.rel_posix(destination, vault.root),
|
|
2087
|
+
sha256=sha256,
|
|
2088
|
+
source_id=topic.source_id,
|
|
2089
|
+
etag=etag,
|
|
2090
|
+
last_modified=last_modified,
|
|
2091
|
+
size=size,
|
|
2092
|
+
fetched_at=fetched_at or _now(),
|
|
2093
|
+
derived=prior.derived if prior is not None and sha256 == prior.sha256 else {},
|
|
2094
|
+
)
|
|
2095
|
+
|
|
2096
|
+
|
|
2097
|
+
def _download_with_candidates(
|
|
2098
|
+
client: IngestClient,
|
|
2099
|
+
school: School,
|
|
2100
|
+
topic: TopicRecord,
|
|
2101
|
+
temporary: Path,
|
|
2102
|
+
*,
|
|
2103
|
+
prior: ManifestEntry | None,
|
|
2104
|
+
max_bytes: int | None,
|
|
2105
|
+
root: Path | None = None,
|
|
2106
|
+
) -> DownloadResult:
|
|
2107
|
+
candidates = _download_candidates(client, school, topic)
|
|
2108
|
+
last_error: DownloadError | None = None
|
|
2109
|
+
for candidate in candidates:
|
|
2110
|
+
try:
|
|
2111
|
+
kwargs: dict[str, object] = {
|
|
2112
|
+
"prior": prior,
|
|
2113
|
+
"is_html_topic": topic.kind.casefold() == "html"
|
|
2114
|
+
or (topic.url_path or "").casefold().endswith((".html", ".htm")),
|
|
2115
|
+
}
|
|
2116
|
+
kwargs["max_bytes"] = max_bytes
|
|
2117
|
+
kwargs["root"] = root
|
|
2118
|
+
return client.download(candidate, temporary, **kwargs) # type: ignore[arg-type]
|
|
2119
|
+
except SessionExpired:
|
|
2120
|
+
raise
|
|
2121
|
+
except FileTooLarge:
|
|
2122
|
+
# A size refusal is a safety decision, not a route-health failure. Trying another
|
|
2123
|
+
# provider route could bypass the exact response-size check that protected the vault.
|
|
2124
|
+
raise
|
|
2125
|
+
except api.DiskSpaceExhausted:
|
|
2126
|
+
# Exhausted local disk space is fatal for every route; a later route's ordinary
|
|
2127
|
+
# failure must not downgrade it to a retryable download error.
|
|
2128
|
+
raise
|
|
2129
|
+
except DownloadError as exc:
|
|
2130
|
+
last_error = exc
|
|
2131
|
+
_remove_quietly(temporary)
|
|
2132
|
+
except RequestException:
|
|
2133
|
+
last_error = DownloadError("download route returned an unusable response")
|
|
2134
|
+
_remove_quietly(temporary)
|
|
2135
|
+
if last_error is not None:
|
|
2136
|
+
raise last_error
|
|
2137
|
+
raise DownloadError("no first-party download route was available")
|
|
2138
|
+
|
|
2139
|
+
|
|
2140
|
+
def _download_candidates(client: IngestClient, school: School, topic: TopicRecord) -> list[str]:
|
|
2141
|
+
base = school.base_url.rstrip("/")
|
|
2142
|
+
le = getattr(client, "le_version", None)
|
|
2143
|
+
ou = topic.course_org_unit_id
|
|
2144
|
+
tid = topic.topic_id
|
|
2145
|
+
candidates: list[str] = []
|
|
2146
|
+
if ":attachment:" in topic.source_key and topic.url_path is not None:
|
|
2147
|
+
return [urljoin(base + "/", topic.url_path.lstrip("/"))]
|
|
2148
|
+
template = client.download_template
|
|
2149
|
+
if isinstance(template, str) and template:
|
|
2150
|
+
with suppress(KeyError, ValueError):
|
|
2151
|
+
candidates.append(template.format(base=base, le=le or "", ou=ou, tid=tid))
|
|
2152
|
+
candidates.extend(
|
|
2153
|
+
[
|
|
2154
|
+
f"{base}/d2l/le/content/{ou}/topics/files/download/{tid}/DirectFileTopicDownload",
|
|
2155
|
+
f"{base}/d2l/api/le/{le}/{ou}/content/topics/{tid}/file"
|
|
2156
|
+
if isinstance(le, str) and le
|
|
2157
|
+
else "",
|
|
2158
|
+
]
|
|
2159
|
+
)
|
|
2160
|
+
if topic.url_path is not None:
|
|
2161
|
+
candidates.append(urljoin(base + "/", topic.url_path.lstrip("/")))
|
|
2162
|
+
seen: set[str] = set()
|
|
2163
|
+
unique: list[str] = []
|
|
2164
|
+
for candidate in candidates:
|
|
2165
|
+
if candidate and candidate not in seen:
|
|
2166
|
+
seen.add(candidate)
|
|
2167
|
+
unique.append(candidate)
|
|
2168
|
+
return unique
|
|
2169
|
+
|
|
2170
|
+
|
|
2171
|
+
def _destination_for_topic(
|
|
2172
|
+
vault: Vault, course_dir: Path, topic: TopicRecord, prior: ManifestEntry | None
|
|
2173
|
+
) -> Path:
|
|
2174
|
+
if prior is not None:
|
|
2175
|
+
return vault.materialized(prior)
|
|
2176
|
+
if topic.source_path is None:
|
|
2177
|
+
raise A2LError("topic has no allocated source path")
|
|
2178
|
+
return _vault_relative_path(vault, topic.source_path)
|
|
2179
|
+
|
|
2180
|
+
|
|
2181
|
+
def _unchanged_local(entry: ManifestEntry, topic: TopicRecord, vault: Vault) -> bool:
|
|
2182
|
+
if topic.etag is None and topic.last_modified is None:
|
|
2183
|
+
return False
|
|
2184
|
+
if topic.etag is not None and entry.etag != topic.etag:
|
|
2185
|
+
return False
|
|
2186
|
+
if topic.last_modified is not None and entry.last_modified != topic.last_modified:
|
|
2187
|
+
return False
|
|
2188
|
+
return _source_bytes_match(entry, vault)
|
|
2189
|
+
|
|
2190
|
+
|
|
2191
|
+
def _source_bytes_match(entry: ManifestEntry, vault: Vault) -> bool:
|
|
2192
|
+
source = vault.materialized(entry)
|
|
2193
|
+
try:
|
|
2194
|
+
return _hash_file(source) == (entry.sha256, entry.size)
|
|
2195
|
+
except (FileNotFoundError, IsADirectoryError):
|
|
2196
|
+
return False
|
|
2197
|
+
|
|
2198
|
+
|
|
2199
|
+
def _mark_topic_source_only(
|
|
2200
|
+
vault: Vault,
|
|
2201
|
+
course_dir: Path,
|
|
2202
|
+
topic: TopicRecord,
|
|
2203
|
+
school: School,
|
|
2204
|
+
) -> None:
|
|
2205
|
+
rows = _map_topics(_read_content_map(course_dir))
|
|
2206
|
+
# A manifest artifact record is not proof that the current twin bytes are still trusted.
|
|
2207
|
+
# Reconcile through the same source-and-derived hash checks used by metadata sync.
|
|
2208
|
+
reconciled = course_index.reconcile_content_map(vault, rows)
|
|
2209
|
+
_write_content_map(course_dir, reconciled, root=vault.root)
|
|
2210
|
+
course = CourseRef(
|
|
2211
|
+
topic.course_org_unit_id,
|
|
2212
|
+
topic.course_code,
|
|
2213
|
+
topic.course_name,
|
|
2214
|
+
topic.term,
|
|
2215
|
+
True,
|
|
2216
|
+
)
|
|
2217
|
+
topics = tuple(
|
|
2218
|
+
_topic_from_row(row, course=course) for row in reconciled if isinstance(row, dict)
|
|
2219
|
+
)
|
|
2220
|
+
_write_index(course_dir, school=school, course=course, topics=topics, root=vault.root)
|
|
2221
|
+
|
|
2222
|
+
|
|
2223
|
+
def _update_row_state(
|
|
2224
|
+
course_dir: Path, source_key: str, *, root: Path | None = None, **updates: object
|
|
2225
|
+
) -> None:
|
|
2226
|
+
content_map = _read_content_map(course_dir)
|
|
2227
|
+
rows = _map_topics(content_map)
|
|
2228
|
+
for row in rows:
|
|
2229
|
+
if isinstance(row, dict) and row.get("source_key") == source_key:
|
|
2230
|
+
row.update(updates)
|
|
2231
|
+
_write_content_map(course_dir, rows, root=root)
|
|
2232
|
+
|
|
2233
|
+
|
|
2234
|
+
def _find_content_row(course_dir: Path, source_key: str) -> dict[str, object] | None:
|
|
2235
|
+
for row in _map_topics(_read_content_map(course_dir)):
|
|
2236
|
+
if isinstance(row, dict) and row.get("source_key") == source_key:
|
|
2237
|
+
return row
|
|
2238
|
+
return None
|
|
2239
|
+
|
|
2240
|
+
|
|
2241
|
+
def _resolve_topic(vault: Vault, query: str) -> tuple[CourseRef, TopicRecord, Path] | None:
|
|
2242
|
+
if not isinstance(query, str) or not query.strip():
|
|
2243
|
+
raise A2LError("topic selector must not be empty")
|
|
2244
|
+
exact: list[tuple[CourseRef, TopicRecord, Path]] = []
|
|
2245
|
+
fuzzy: list[tuple[CourseRef, TopicRecord, Path]] = []
|
|
2246
|
+
folded = query.casefold()
|
|
2247
|
+
for map_path in sorted(
|
|
2248
|
+
path for path in paths.walk(vault.root) if path.name == "content_map.json"
|
|
2249
|
+
):
|
|
2250
|
+
course_dir = map_path.parent.parent
|
|
2251
|
+
raw = _read_content_map(course_dir)
|
|
2252
|
+
for row in _map_topics(raw):
|
|
2253
|
+
if not isinstance(row, dict):
|
|
2254
|
+
continue
|
|
2255
|
+
try:
|
|
2256
|
+
course = CourseRef(
|
|
2257
|
+
int(row["course_org_unit_id"]),
|
|
2258
|
+
str(row["course_code"]),
|
|
2259
|
+
str(row["course_name"]),
|
|
2260
|
+
row.get("term") if isinstance(row.get("term"), str) else None,
|
|
2261
|
+
True,
|
|
2262
|
+
)
|
|
2263
|
+
record = _topic_from_row(row, course=course)
|
|
2264
|
+
except (KeyError, TypeError, ValueError, A2LError):
|
|
2265
|
+
continue
|
|
2266
|
+
candidate = (course, record, course_dir)
|
|
2267
|
+
if query in {record.source_key, record.source_id} or query == record.source_path:
|
|
2268
|
+
exact.append(candidate)
|
|
2269
|
+
elif folded in record.title.casefold() or (
|
|
2270
|
+
record.path is not None and folded in record.path.casefold()
|
|
2271
|
+
):
|
|
2272
|
+
fuzzy.append(candidate)
|
|
2273
|
+
if exact:
|
|
2274
|
+
if len(exact) > 1:
|
|
2275
|
+
raise A2LError(f"ambiguous topic selector: {query}")
|
|
2276
|
+
return exact[0]
|
|
2277
|
+
if len(fuzzy) > 1:
|
|
2278
|
+
raise A2LError(f"ambiguous topic selector: {query}")
|
|
2279
|
+
return fuzzy[0] if fuzzy else None
|
|
2280
|
+
|
|
2281
|
+
|
|
2282
|
+
def _read_content_map(course_dir: Path) -> dict[str, object]:
|
|
2283
|
+
return course_index.read_content_map(course_dir)
|
|
2284
|
+
|
|
2285
|
+
|
|
2286
|
+
def _map_topics(content_map: Mapping[str, object]) -> list[object]:
|
|
2287
|
+
topics = content_map.get("topics")
|
|
2288
|
+
if not isinstance(topics, list):
|
|
2289
|
+
raise A2LError("content_map.json topics must be an array")
|
|
2290
|
+
return topics
|
|
2291
|
+
|
|
2292
|
+
|
|
2293
|
+
def _write_content_map(
|
|
2294
|
+
course_dir: Path, rows: Sequence[object], *, root: Path | None = None
|
|
2295
|
+
) -> None:
|
|
2296
|
+
course_index.write_content_map(course_dir, rows, root=root)
|
|
2297
|
+
|
|
2298
|
+
|
|
2299
|
+
def _write_toc(
|
|
2300
|
+
course_dir: Path, modules: Sequence[dict[str, object]], *, root: Path | None = None
|
|
2301
|
+
) -> None:
|
|
2302
|
+
_write_json(
|
|
2303
|
+
course_dir / "_meta" / "toc.json",
|
|
2304
|
+
{"schema_version": 1, "modules": list(modules)},
|
|
2305
|
+
root=root,
|
|
2306
|
+
)
|
|
2307
|
+
|
|
2308
|
+
|
|
2309
|
+
def _read_toc_modules(course_dir: Path) -> list[dict[str, object]]:
|
|
2310
|
+
destination = course_dir / "_meta" / "toc.json"
|
|
2311
|
+
try:
|
|
2312
|
+
with open(os.fspath(paths.long_path(destination)), encoding="utf-8", newline="") as handle:
|
|
2313
|
+
raw: Any = json.load(handle)
|
|
2314
|
+
except FileNotFoundError:
|
|
2315
|
+
return []
|
|
2316
|
+
except (OSError, UnicodeError, json.JSONDecodeError) as exc:
|
|
2317
|
+
raise A2LError(f"{destination.name} is unreadable") from exc
|
|
2318
|
+
if not isinstance(raw, dict):
|
|
2319
|
+
raise A2LError(f"{destination.name} must contain an object")
|
|
2320
|
+
modules = raw.get("modules")
|
|
2321
|
+
if not isinstance(modules, list):
|
|
2322
|
+
raise A2LError(f"{destination.name} must contain a modules list")
|
|
2323
|
+
if any(not isinstance(value, dict) for value in modules):
|
|
2324
|
+
raise A2LError(f"{destination.name} contains an invalid module")
|
|
2325
|
+
return [cast(dict[str, object], value) for value in modules]
|
|
2326
|
+
|
|
2327
|
+
|
|
2328
|
+
def _write_json(destination: Path, payload: object, *, root: Path | None = None) -> None:
|
|
2329
|
+
paths.ensure_dir(destination.parent, root=root)
|
|
2330
|
+
text = (
|
|
2331
|
+
json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2, separators=(",", ": "))
|
|
2332
|
+
+ "\n"
|
|
2333
|
+
)
|
|
2334
|
+
paths.atomic_write_text(destination, text, root=root)
|
|
2335
|
+
|
|
2336
|
+
|
|
2337
|
+
def _read_list(destination: Path) -> list[dict[str, object]]:
|
|
2338
|
+
try:
|
|
2339
|
+
with open(os.fspath(paths.long_path(destination)), encoding="utf-8", newline="") as handle:
|
|
2340
|
+
raw: Any = json.load(handle)
|
|
2341
|
+
except FileNotFoundError:
|
|
2342
|
+
return []
|
|
2343
|
+
except (OSError, UnicodeError, json.JSONDecodeError) as exc:
|
|
2344
|
+
raise A2LError(f"{destination.name} is unreadable") from exc
|
|
2345
|
+
if not isinstance(raw, list):
|
|
2346
|
+
raise A2LError(f"{destination.name} must contain a list")
|
|
2347
|
+
if any(not isinstance(row, dict) for row in raw):
|
|
2348
|
+
raise A2LError(f"{destination.name} contains an invalid item")
|
|
2349
|
+
return [cast(dict[str, object], row) for row in raw]
|
|
2350
|
+
|
|
2351
|
+
|
|
2352
|
+
def _write_list(
|
|
2353
|
+
destination: Path, rows: Sequence[Mapping[str, object]], *, root: Path | None = None
|
|
2354
|
+
) -> None:
|
|
2355
|
+
_write_json(destination, list(rows), root=root)
|
|
2356
|
+
|
|
2357
|
+
|
|
2358
|
+
def _merge_rows(
|
|
2359
|
+
existing: Sequence[Mapping[str, object]],
|
|
2360
|
+
incoming: Sequence[Mapping[str, object]],
|
|
2361
|
+
*,
|
|
2362
|
+
id_field: str,
|
|
2363
|
+
complete: bool,
|
|
2364
|
+
) -> list[dict[str, object]]:
|
|
2365
|
+
by_id: dict[str, dict[str, object]] = {}
|
|
2366
|
+
for row in existing:
|
|
2367
|
+
value = row.get(id_field)
|
|
2368
|
+
if value is not None:
|
|
2369
|
+
by_id[str(value)] = dict(row)
|
|
2370
|
+
incoming_ids: set[str] = set()
|
|
2371
|
+
for row in incoming:
|
|
2372
|
+
value = row.get(id_field)
|
|
2373
|
+
if value is None:
|
|
2374
|
+
continue
|
|
2375
|
+
key = str(value)
|
|
2376
|
+
incoming_ids.add(key)
|
|
2377
|
+
merged = dict(by_id.get(key, {}))
|
|
2378
|
+
merged.update(row)
|
|
2379
|
+
merged["missing_since"] = None
|
|
2380
|
+
merged["withdrawn_at"] = None
|
|
2381
|
+
by_id[key] = merged
|
|
2382
|
+
if complete:
|
|
2383
|
+
now = _now()
|
|
2384
|
+
for key, row in by_id.items():
|
|
2385
|
+
if key in incoming_ids:
|
|
2386
|
+
continue
|
|
2387
|
+
if row.get("missing_since") is None:
|
|
2388
|
+
row["missing_since"] = now
|
|
2389
|
+
elif row.get("withdrawn_at") is None:
|
|
2390
|
+
row["withdrawn_at"] = now
|
|
2391
|
+
return sorted(
|
|
2392
|
+
by_id.values(), key=lambda row: (_date_key(row.get("date")), str(row.get(id_field)))
|
|
2393
|
+
)
|
|
2394
|
+
|
|
2395
|
+
|
|
2396
|
+
def _project_assignments(values: Sequence[object]) -> list[dict[str, object]]:
|
|
2397
|
+
rows: list[dict[str, object]] = []
|
|
2398
|
+
for value in values:
|
|
2399
|
+
if (
|
|
2400
|
+
not isinstance(value, dict)
|
|
2401
|
+
or not isinstance(value.get("Id"), int)
|
|
2402
|
+
or isinstance(value.get("Id"), bool)
|
|
2403
|
+
):
|
|
2404
|
+
continue
|
|
2405
|
+
availability = value.get("Availability")
|
|
2406
|
+
rows.append(
|
|
2407
|
+
{
|
|
2408
|
+
"id": value["Id"],
|
|
2409
|
+
"title": _safe_text(value.get("Name")),
|
|
2410
|
+
"due_date": _optional_text(value.get("DueDate")),
|
|
2411
|
+
"start_date": _nested_optional_text(availability, "StartDate"),
|
|
2412
|
+
"end_date": _nested_optional_text(availability, "EndDate"),
|
|
2413
|
+
"grade_item": isinstance(value.get("GradeItemId"), int),
|
|
2414
|
+
"group": isinstance(value.get("GroupTypeId"), int),
|
|
2415
|
+
"date": _optional_text(value.get("DueDate")) or "",
|
|
2416
|
+
}
|
|
2417
|
+
)
|
|
2418
|
+
return rows
|
|
2419
|
+
|
|
2420
|
+
|
|
2421
|
+
def _project_news(values: Sequence[object]) -> list[dict[str, object]]:
|
|
2422
|
+
rows: list[dict[str, object]] = []
|
|
2423
|
+
for value in values:
|
|
2424
|
+
if (
|
|
2425
|
+
not isinstance(value, dict)
|
|
2426
|
+
or not isinstance(value.get("Id"), int)
|
|
2427
|
+
or isinstance(value.get("Id"), bool)
|
|
2428
|
+
):
|
|
2429
|
+
continue
|
|
2430
|
+
body = value.get("Body")
|
|
2431
|
+
body_text = body.get("Text") if isinstance(body, dict) else None
|
|
2432
|
+
body_html = body.get("Html") if isinstance(body, dict) else None
|
|
2433
|
+
start = _optional_text(value.get("StartDate")) or ""
|
|
2434
|
+
rows.append(
|
|
2435
|
+
{
|
|
2436
|
+
"id": value["Id"],
|
|
2437
|
+
"title": _safe_text(value.get("Title")),
|
|
2438
|
+
"text": _safe_text(body_text),
|
|
2439
|
+
"html": _sanitize_html_text(body_html) if isinstance(body_html, str) else None,
|
|
2440
|
+
"start_date": _optional_text(value.get("StartDate")),
|
|
2441
|
+
"end_date": _optional_text(value.get("EndDate")),
|
|
2442
|
+
"published": bool(value.get("IsPublished", False)),
|
|
2443
|
+
"date": start,
|
|
2444
|
+
}
|
|
2445
|
+
)
|
|
2446
|
+
return rows
|
|
2447
|
+
|
|
2448
|
+
|
|
2449
|
+
def _project_quizzes(values: Sequence[object]) -> list[dict[str, object]]:
|
|
2450
|
+
rows: list[dict[str, object]] = []
|
|
2451
|
+
for value in values:
|
|
2452
|
+
if (
|
|
2453
|
+
not isinstance(value, dict)
|
|
2454
|
+
or not isinstance(value.get("QuizId"), int)
|
|
2455
|
+
or isinstance(value.get("QuizId"), bool)
|
|
2456
|
+
):
|
|
2457
|
+
continue
|
|
2458
|
+
due = _optional_text(value.get("DueDate"))
|
|
2459
|
+
rows.append(
|
|
2460
|
+
{
|
|
2461
|
+
"id": value["QuizId"],
|
|
2462
|
+
"title": _safe_text(value.get("Name")),
|
|
2463
|
+
"due_date": due,
|
|
2464
|
+
"start_date": _optional_text(value.get("StartDate")),
|
|
2465
|
+
"end_date": _optional_text(value.get("EndDate")),
|
|
2466
|
+
"active": bool(value.get("IsActive", False)),
|
|
2467
|
+
"date": due or "",
|
|
2468
|
+
}
|
|
2469
|
+
)
|
|
2470
|
+
return rows
|
|
2471
|
+
|
|
2472
|
+
|
|
2473
|
+
def _project_grades(values: Sequence[object]) -> list[dict[str, object]]:
|
|
2474
|
+
# Grades are opt-in, but the projection still excludes all unrelated response fields.
|
|
2475
|
+
rows: list[dict[str, object]] = []
|
|
2476
|
+
for value in values:
|
|
2477
|
+
if not isinstance(value, dict):
|
|
2478
|
+
continue
|
|
2479
|
+
identifier = value.get("GradeObjectIdentifier")
|
|
2480
|
+
if identifier is None:
|
|
2481
|
+
continue
|
|
2482
|
+
rows.append(
|
|
2483
|
+
{
|
|
2484
|
+
"id": str(identifier),
|
|
2485
|
+
"name": _safe_text(value.get("GradeObjectName")),
|
|
2486
|
+
"type": _safe_text(value.get("GradeObjectType")),
|
|
2487
|
+
"numerator": value.get("PointsNumerator"),
|
|
2488
|
+
"denominator": value.get("PointsDenominator"),
|
|
2489
|
+
"displayed": _safe_text(value.get("DisplayedGrade")),
|
|
2490
|
+
}
|
|
2491
|
+
)
|
|
2492
|
+
return rows
|
|
2493
|
+
|
|
2494
|
+
|
|
2495
|
+
def _write_announcements(
|
|
2496
|
+
destination: Path,
|
|
2497
|
+
rows: Sequence[Mapping[str, object]],
|
|
2498
|
+
*,
|
|
2499
|
+
root: Path | None = None,
|
|
2500
|
+
) -> None:
|
|
2501
|
+
paths.ensure_dir(destination.parent, root=root)
|
|
2502
|
+
lines = ["# Announcements", ""]
|
|
2503
|
+
for row in rows:
|
|
2504
|
+
title = str(row.get("title") or "Untitled")
|
|
2505
|
+
date = str(row.get("start_date") or "")
|
|
2506
|
+
lines.extend([f"## {title}", "", f"Date: {date}" if date else "", ""])
|
|
2507
|
+
if row.get("withdrawn_at"):
|
|
2508
|
+
lines.extend(["> No longer posted in LEARN.", ""])
|
|
2509
|
+
text = str(row.get("text") or "").strip()
|
|
2510
|
+
if text:
|
|
2511
|
+
lines.extend([text, ""])
|
|
2512
|
+
paths.atomic_write_text(destination, "\n".join(lines).rstrip() + "\n", root=root)
|
|
2513
|
+
|
|
2514
|
+
|
|
2515
|
+
def _write_index(
|
|
2516
|
+
course_dir: Path,
|
|
2517
|
+
*,
|
|
2518
|
+
school: School | None,
|
|
2519
|
+
course: CourseRef | None,
|
|
2520
|
+
topics: Sequence[TopicRecord] | None,
|
|
2521
|
+
root: Path | None = None,
|
|
2522
|
+
) -> None:
|
|
2523
|
+
if school is None or course is None or topics is None:
|
|
2524
|
+
# A file-only checkpoint updates the coverage tree without needing to retain a second
|
|
2525
|
+
# network response in memory. The existing index remains valid until metadata runs again.
|
|
2526
|
+
return
|
|
2527
|
+
term_label = school.term_label(course.term) if course.term else "Unclassified"
|
|
2528
|
+
assignments = _read_list(course_dir / "_meta" / "assignments.json")
|
|
2529
|
+
quizzes = _read_list(course_dir / "_meta" / "quizzes.json")
|
|
2530
|
+
deadlines = [
|
|
2531
|
+
(str(row.get("due_date")), str(row.get("title") or "Untitled"), "assignment")
|
|
2532
|
+
for row in assignments
|
|
2533
|
+
if row.get("due_date")
|
|
2534
|
+
] + [
|
|
2535
|
+
(str(row.get("due_date")), str(row.get("title") or "Untitled"), "quiz")
|
|
2536
|
+
for row in quizzes
|
|
2537
|
+
if row.get("due_date")
|
|
2538
|
+
]
|
|
2539
|
+
course_index.write_course_index(
|
|
2540
|
+
course_dir,
|
|
2541
|
+
course_code=course.code,
|
|
2542
|
+
course_name=course.name,
|
|
2543
|
+
term_label=term_label,
|
|
2544
|
+
term_code=course.term,
|
|
2545
|
+
topics=[_topic_to_row(topic) for topic in topics],
|
|
2546
|
+
deadlines=deadlines,
|
|
2547
|
+
root=root,
|
|
2548
|
+
)
|
|
2549
|
+
|
|
2550
|
+
|
|
2551
|
+
def _course_relative_link(value: str, course_dir: Path) -> str:
|
|
2552
|
+
parts = list(PurePosixPath(value).parts)
|
|
2553
|
+
try:
|
|
2554
|
+
index = parts.index(course_dir.name)
|
|
2555
|
+
except ValueError:
|
|
2556
|
+
return PurePosixPath(value).as_posix()
|
|
2557
|
+
relative = parts[index + 1 :]
|
|
2558
|
+
return PurePosixPath(*relative).as_posix() if relative else "."
|
|
2559
|
+
|
|
2560
|
+
|
|
2561
|
+
def _content_directory(course_dir: Path, module_path: Sequence[str]) -> Path:
|
|
2562
|
+
directory = course_dir / _COURSE_CONTENT
|
|
2563
|
+
for component in module_path:
|
|
2564
|
+
directory /= paths.safe_name(component)
|
|
2565
|
+
return directory
|
|
2566
|
+
|
|
2567
|
+
|
|
2568
|
+
def _extension(name: str) -> str:
|
|
2569
|
+
"""Return a usable extension, treating a bare trailing dot as no extension.
|
|
2570
|
+
|
|
2571
|
+
Python 3.14 changed ``PurePath.suffix``: ``'Reading list.'`` now reports ``'.'`` where
|
|
2572
|
+
earlier versions report ``''``. Taken at face value that lone dot looks like an
|
|
2573
|
+
extension, so a topic titled with a trailing dot keeps no real extension and the file
|
|
2574
|
+
lands with none at all — a different vault on 3.14 than on 3.11. A single dot is not an
|
|
2575
|
+
extension on any version, so it is normalized away here rather than at each call site.
|
|
2576
|
+
"""
|
|
2577
|
+
suffix = PurePosixPath(name).suffix
|
|
2578
|
+
return suffix if len(suffix) > 1 else ""
|
|
2579
|
+
|
|
2580
|
+
|
|
2581
|
+
def _topic_filename(topic: TopicRecord) -> str:
|
|
2582
|
+
title = topic.title.strip() or "untitled"
|
|
2583
|
+
title_path = PurePosixPath(title).name
|
|
2584
|
+
url_name = PurePosixPath(urlsplit(topic.url_path or "").path).name
|
|
2585
|
+
title_suffix = _extension(title_path)
|
|
2586
|
+
suffix = title_suffix or _extension(url_name)
|
|
2587
|
+
base = title_path if title_suffix else f"{title_path}{suffix}"
|
|
2588
|
+
return paths.safe_name(base)
|
|
2589
|
+
|
|
2590
|
+
|
|
2591
|
+
def _unique_reserved(destination: Path, reserved: set[str]) -> Path:
|
|
2592
|
+
candidate = paths.unique_path(destination)
|
|
2593
|
+
for number in range(1, 100_000):
|
|
2594
|
+
if _canonical_name(candidate) not in reserved and not paths.collides(candidate):
|
|
2595
|
+
reserved.add(_canonical_name(candidate))
|
|
2596
|
+
return candidate
|
|
2597
|
+
stem, extension = _split_name(candidate.name)
|
|
2598
|
+
suffix = "" if number == 1 else f"_{number}"
|
|
2599
|
+
candidate = candidate.with_name(paths.safe_name(f"{stem}{suffix}{extension}"))
|
|
2600
|
+
raise A2LError("could not allocate a unique content path")
|
|
2601
|
+
|
|
2602
|
+
|
|
2603
|
+
def _canonical_name(path: Path) -> str:
|
|
2604
|
+
return path.name.casefold()
|
|
2605
|
+
|
|
2606
|
+
|
|
2607
|
+
def _split_name(name: str) -> tuple[str, str]:
|
|
2608
|
+
# ``_extension`` rather than ``.suffix``: a collision suffix must be inserted at the
|
|
2609
|
+
# same place on every Python version (see the 3.14 note there).
|
|
2610
|
+
suffix = _extension(name)
|
|
2611
|
+
return (name[: -len(suffix)], suffix) if suffix else (name, "")
|
|
2612
|
+
|
|
2613
|
+
|
|
2614
|
+
def _vault_relative_path(vault: Vault, value: str) -> Path:
|
|
2615
|
+
if (
|
|
2616
|
+
not isinstance(value, str)
|
|
2617
|
+
or not value
|
|
2618
|
+
or "\\" in value
|
|
2619
|
+
or re.match(r"^[A-Za-z]:[\\/]", value) is not None
|
|
2620
|
+
or PurePosixPath(value).is_absolute()
|
|
2621
|
+
or any(part in {"", ".", ".."} for part in PurePosixPath(value).parts)
|
|
2622
|
+
):
|
|
2623
|
+
raise A2LError("content path must be a normalized relative POSIX path")
|
|
2624
|
+
candidate = (vault.root / Path(*PurePosixPath(value).parts)).resolve()
|
|
2625
|
+
try:
|
|
2626
|
+
candidate.relative_to(vault.root)
|
|
2627
|
+
except ValueError as exc:
|
|
2628
|
+
raise A2LError("content path escapes the vault root") from exc
|
|
2629
|
+
return candidate
|
|
2630
|
+
|
|
2631
|
+
|
|
2632
|
+
def _derived_path(manifest: Mapping[str, ManifestEntry], entry: ManifestEntry) -> str | None:
|
|
2633
|
+
artifact = entry.derived.get("markdown")
|
|
2634
|
+
if artifact is None:
|
|
2635
|
+
return None
|
|
2636
|
+
return artifact.path if artifact.path else None
|
|
2637
|
+
|
|
2638
|
+
|
|
2639
|
+
def _is_media(topic: TopicRecord) -> bool:
|
|
2640
|
+
return PurePosixPath(_topic_filename(topic)).suffix.casefold() in _MEDIA_SUFFIXES
|
|
2641
|
+
|
|
2642
|
+
|
|
2643
|
+
def is_media_topic(topic: TopicRecord) -> bool:
|
|
2644
|
+
"""Return the canonical media classification used by file ingestion and previews."""
|
|
2645
|
+
|
|
2646
|
+
return _is_media(topic)
|
|
2647
|
+
|
|
2648
|
+
|
|
2649
|
+
def _is_office_lock(topic: TopicRecord) -> bool:
|
|
2650
|
+
filename = _topic_filename(topic)
|
|
2651
|
+
url_name = PurePosixPath(urlsplit(topic.url_path or "").path).name
|
|
2652
|
+
return bool(_OFFICE_LOCK.match(filename) or _OFFICE_LOCK.match(url_name))
|
|
2653
|
+
|
|
2654
|
+
|
|
2655
|
+
def _safe_hostname(value: str | None) -> str | None:
|
|
2656
|
+
if not value:
|
|
2657
|
+
return None
|
|
2658
|
+
try:
|
|
2659
|
+
hostname = urlsplit(value).hostname
|
|
2660
|
+
if hostname is None:
|
|
2661
|
+
return None
|
|
2662
|
+
return hostname.encode("idna").decode("ascii").casefold().rstrip(".")
|
|
2663
|
+
except (UnicodeError, ValueError):
|
|
2664
|
+
return None
|
|
2665
|
+
|
|
2666
|
+
|
|
2667
|
+
def _first_party_path(value: str | None, base_url: str) -> str | None:
|
|
2668
|
+
if not value:
|
|
2669
|
+
return None
|
|
2670
|
+
try:
|
|
2671
|
+
candidate = urljoin(base_url.rstrip("/") + "/", value)
|
|
2672
|
+
parsed = urlsplit(candidate)
|
|
2673
|
+
base = urlsplit(base_url)
|
|
2674
|
+
if parsed.scheme.casefold() not in {"http", "https"}:
|
|
2675
|
+
return None
|
|
2676
|
+
if parsed.username is not None or parsed.password is not None:
|
|
2677
|
+
return None
|
|
2678
|
+
host = _safe_hostname(candidate)
|
|
2679
|
+
base_host = _safe_hostname(base_url)
|
|
2680
|
+
if host != base_host or parsed.port != base.port:
|
|
2681
|
+
return None
|
|
2682
|
+
path = parsed.path or "/"
|
|
2683
|
+
return urlunsplit(("", "", path, "", ""))
|
|
2684
|
+
except (TypeError, ValueError):
|
|
2685
|
+
return None
|
|
2686
|
+
|
|
2687
|
+
|
|
2688
|
+
def _allowed_outline_url(value: str | None, school: School) -> str | None:
|
|
2689
|
+
if not value:
|
|
2690
|
+
return None
|
|
2691
|
+
try:
|
|
2692
|
+
parsed = urlsplit(value)
|
|
2693
|
+
if (
|
|
2694
|
+
parsed.scheme.casefold() != "https"
|
|
2695
|
+
or parsed.username is not None
|
|
2696
|
+
or parsed.password is not None
|
|
2697
|
+
or not hostname_matches_suffix(value, school.outline_hosts())
|
|
2698
|
+
):
|
|
2699
|
+
return None
|
|
2700
|
+
return urlunsplit(("https", parsed.netloc, parsed.path or "/", "", ""))
|
|
2701
|
+
except (TypeError, ValueError):
|
|
2702
|
+
return None
|
|
2703
|
+
|
|
2704
|
+
|
|
2705
|
+
def _view_url(school: School, org_unit_id: int, topic_id: int) -> str:
|
|
2706
|
+
return f"{school.base_url.rstrip('/')}/d2l/le/content/{org_unit_id}/viewContent/{topic_id}/View"
|
|
2707
|
+
|
|
2708
|
+
|
|
2709
|
+
def _safe_text(value: object) -> str:
|
|
2710
|
+
return value if isinstance(value, str) else ""
|
|
2711
|
+
|
|
2712
|
+
|
|
2713
|
+
def _optional_text(value: object) -> str | None:
|
|
2714
|
+
return value if isinstance(value, str) else None
|
|
2715
|
+
|
|
2716
|
+
|
|
2717
|
+
def _nested_optional_text(value: object, key: str) -> str | None:
|
|
2718
|
+
return _optional_text(value.get(key)) if isinstance(value, dict) else None
|
|
2719
|
+
|
|
2720
|
+
|
|
2721
|
+
class _RichTextSanitizer(HTMLParser):
|
|
2722
|
+
"""Render inert, deterministic HTML without executing or retaining active attributes."""
|
|
2723
|
+
|
|
2724
|
+
_ALLOWED_TAGS = frozenset(
|
|
2725
|
+
{
|
|
2726
|
+
"a",
|
|
2727
|
+
"b",
|
|
2728
|
+
"blockquote",
|
|
2729
|
+
"br",
|
|
2730
|
+
"code",
|
|
2731
|
+
"dd",
|
|
2732
|
+
"div",
|
|
2733
|
+
"em",
|
|
2734
|
+
"h1",
|
|
2735
|
+
"h2",
|
|
2736
|
+
"h3",
|
|
2737
|
+
"h4",
|
|
2738
|
+
"h5",
|
|
2739
|
+
"h6",
|
|
2740
|
+
"hr",
|
|
2741
|
+
"img",
|
|
2742
|
+
"li",
|
|
2743
|
+
"ol",
|
|
2744
|
+
"p",
|
|
2745
|
+
"pre",
|
|
2746
|
+
"section",
|
|
2747
|
+
"span",
|
|
2748
|
+
"strong",
|
|
2749
|
+
"table",
|
|
2750
|
+
"tbody",
|
|
2751
|
+
"td",
|
|
2752
|
+
"th",
|
|
2753
|
+
"thead",
|
|
2754
|
+
"tr",
|
|
2755
|
+
"u",
|
|
2756
|
+
"ul",
|
|
2757
|
+
}
|
|
2758
|
+
)
|
|
2759
|
+
_VOID_TAGS = frozenset({"br", "hr", "img"})
|
|
2760
|
+
|
|
2761
|
+
def __init__(self, base_url: str) -> None:
|
|
2762
|
+
super().__init__(convert_charrefs=False)
|
|
2763
|
+
self.base_url = base_url
|
|
2764
|
+
self.parts: list[str] = []
|
|
2765
|
+
self._skip_depth = 0
|
|
2766
|
+
|
|
2767
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
2768
|
+
tag = tag.casefold()
|
|
2769
|
+
if self._skip_depth:
|
|
2770
|
+
if tag not in self._VOID_TAGS:
|
|
2771
|
+
self._skip_depth += 1
|
|
2772
|
+
return
|
|
2773
|
+
if tag in {"script", "style", "form", "iframe", "object", "embed", "template"}:
|
|
2774
|
+
self._skip_depth = 1
|
|
2775
|
+
return
|
|
2776
|
+
if tag not in self._ALLOWED_TAGS:
|
|
2777
|
+
return
|
|
2778
|
+
safe_attrs: list[str] = []
|
|
2779
|
+
for name, value in attrs:
|
|
2780
|
+
name = name.casefold()
|
|
2781
|
+
if name == "href" and tag == "a" and value is not None:
|
|
2782
|
+
safe_url = _safe_richtext_url(value, self.base_url)
|
|
2783
|
+
if safe_url is not None:
|
|
2784
|
+
safe_attrs.append(f'href="{escape(safe_url, quote=True)}"')
|
|
2785
|
+
elif name in {"alt", "title"} and value is not None and tag in {"a", "img"}:
|
|
2786
|
+
safe_attrs.append(f'{name}="{escape(value, quote=True)}"')
|
|
2787
|
+
suffix = "" if not safe_attrs else " " + " ".join(safe_attrs)
|
|
2788
|
+
self.parts.append(f"<{tag}{suffix}>")
|
|
2789
|
+
|
|
2790
|
+
def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
2791
|
+
self.handle_starttag(tag, attrs)
|
|
2792
|
+
|
|
2793
|
+
def handle_endtag(self, tag: str) -> None:
|
|
2794
|
+
tag = tag.casefold()
|
|
2795
|
+
if self._skip_depth:
|
|
2796
|
+
self._skip_depth -= 1
|
|
2797
|
+
return
|
|
2798
|
+
if tag in self._ALLOWED_TAGS and tag not in self._VOID_TAGS:
|
|
2799
|
+
self.parts.append(f"</{tag}>")
|
|
2800
|
+
|
|
2801
|
+
def handle_data(self, data: str) -> None:
|
|
2802
|
+
if not self._skip_depth:
|
|
2803
|
+
self.parts.append(escape(data))
|
|
2804
|
+
|
|
2805
|
+
def handle_entityref(self, name: str) -> None:
|
|
2806
|
+
if not self._skip_depth:
|
|
2807
|
+
self.parts.append(f"&{name};")
|
|
2808
|
+
|
|
2809
|
+
def handle_charref(self, name: str) -> None:
|
|
2810
|
+
if not self._skip_depth:
|
|
2811
|
+
self.parts.append(f"&#{name};")
|
|
2812
|
+
|
|
2813
|
+
def handle_comment(self, _data: str) -> None:
|
|
2814
|
+
return
|
|
2815
|
+
|
|
2816
|
+
|
|
2817
|
+
def _sanitize_richtext(value: str, base_url: str) -> str:
|
|
2818
|
+
parser = _RichTextSanitizer(base_url)
|
|
2819
|
+
try:
|
|
2820
|
+
parser.feed(value)
|
|
2821
|
+
parser.close()
|
|
2822
|
+
except (TypeError, ValueError):
|
|
2823
|
+
return ""
|
|
2824
|
+
return "".join(parser.parts).strip()
|
|
2825
|
+
|
|
2826
|
+
|
|
2827
|
+
def _safe_richtext_url(value: str, base_url: str) -> str | None:
|
|
2828
|
+
try:
|
|
2829
|
+
candidate = urljoin(base_url.rstrip("/") + "/", value)
|
|
2830
|
+
parsed = urlsplit(candidate)
|
|
2831
|
+
if parsed.scheme.casefold() not in {"http", "https"}:
|
|
2832
|
+
return None
|
|
2833
|
+
if parsed.username is not None or parsed.password is not None or parsed.hostname is None:
|
|
2834
|
+
return None
|
|
2835
|
+
# RichText links are inert and query-free. They are never followed by conversion or sync.
|
|
2836
|
+
return urlunsplit((parsed.scheme.casefold(), parsed.netloc, parsed.path or "/", "", ""))
|
|
2837
|
+
except (TypeError, ValueError):
|
|
2838
|
+
return None
|
|
2839
|
+
|
|
2840
|
+
|
|
2841
|
+
def _richtext_markdown(title: str, sanitized_html: str) -> str:
|
|
2842
|
+
body = re.sub(r"(?s)<br\s*/?>", "\n", sanitized_html, flags=re.IGNORECASE)
|
|
2843
|
+
body = re.sub(r"(?s)</(?:p|div|h[1-6]|li|tr|blockquote)>", "\n", body, flags=re.IGNORECASE)
|
|
2844
|
+
body = re.sub(r"(?s)<[^>]+>", "", body)
|
|
2845
|
+
body = re.sub(r"\n{3,}", "\n\n", body)
|
|
2846
|
+
body = re.sub(r"[ \t]+", " ", body)
|
|
2847
|
+
body = body.strip()
|
|
2848
|
+
return f"# {title}\n\n{body}\n"
|
|
2849
|
+
|
|
2850
|
+
|
|
2851
|
+
def _sanitize_html_text(value: str) -> str:
|
|
2852
|
+
# News remains a typed projection; unlike assignment instructions it does not become an
|
|
2853
|
+
# evidence-bearing source/twin, so retain a compact plain-text rendering only.
|
|
2854
|
+
text = re.sub(r"<[^>]*>", " ", value)
|
|
2855
|
+
return re.sub(r"\s+", " ", text).strip()
|
|
2856
|
+
|
|
2857
|
+
|
|
2858
|
+
def _now() -> str:
|
|
2859
|
+
return clock.stamp()
|
|
2860
|
+
|
|
2861
|
+
|
|
2862
|
+
def _date_key(value: object) -> str:
|
|
2863
|
+
return value if isinstance(value, str) else ""
|
|
2864
|
+
|
|
2865
|
+
|
|
2866
|
+
def _timestamp_sort(value: str | None) -> float:
|
|
2867
|
+
if value is None:
|
|
2868
|
+
return 0.0
|
|
2869
|
+
try:
|
|
2870
|
+
return parse_api_timestamp(value).timestamp()
|
|
2871
|
+
except ValueError:
|
|
2872
|
+
return 0.0
|
|
2873
|
+
|
|
2874
|
+
|
|
2875
|
+
def _hash_file(path: Path) -> tuple[str, int]:
|
|
2876
|
+
digest = sha256()
|
|
2877
|
+
size = 0
|
|
2878
|
+
with open(os.fspath(paths.long_path(path)), "rb") as handle:
|
|
2879
|
+
while chunk := handle.read(1024 * 1024):
|
|
2880
|
+
digest.update(chunk)
|
|
2881
|
+
size += len(chunk)
|
|
2882
|
+
return digest.hexdigest(), size
|
|
2883
|
+
|
|
2884
|
+
|
|
2885
|
+
def _remove_quietly(path: Path) -> None:
|
|
2886
|
+
try:
|
|
2887
|
+
os.unlink(os.fspath(paths.long_path(path)))
|
|
2888
|
+
except OSError:
|
|
2889
|
+
return
|
|
2890
|
+
|
|
2891
|
+
|
|
2892
|
+
def _safe_error(stage: str, exc: BaseException) -> str:
|
|
2893
|
+
# Reports are intentionally category-level. Never include URL, title, path, or response text.
|
|
2894
|
+
return f"{stage}: {type(exc).__name__}"
|
|
2895
|
+
|
|
2896
|
+
|
|
2897
|
+
def _ingest_discussions(
|
|
2898
|
+
client: IngestClient,
|
|
2899
|
+
course: CourseRef,
|
|
2900
|
+
course_dir: Path,
|
|
2901
|
+
vault: Vault,
|
|
2902
|
+
*,
|
|
2903
|
+
include_authors: bool,
|
|
2904
|
+
) -> str | None:
|
|
2905
|
+
"""Fetch opt-in discussions while keeping raw author identity out of the vault."""
|
|
2906
|
+
|
|
2907
|
+
values, complete, error = _fetch_collection(
|
|
2908
|
+
client, _endpoint_path(client, course, "discussions/forums/")
|
|
2909
|
+
)
|
|
2910
|
+
if error is not None:
|
|
2911
|
+
return _safe_error("discussions", error)
|
|
2912
|
+
if not complete:
|
|
2913
|
+
return _safe_error("discussions", A2LError("incomplete response"))
|
|
2914
|
+
if any(
|
|
2915
|
+
not isinstance(value, dict) or not _discussion_forum_is_valid(value) for value in values
|
|
2916
|
+
):
|
|
2917
|
+
return _safe_error("discussions", A2LError("metadata endpoint returned an invalid forum"))
|
|
2918
|
+
forum_ids = [value["ForumId"] for value in values if isinstance(value, dict)]
|
|
2919
|
+
if len({str(value) for value in forum_ids}) != len(forum_ids):
|
|
2920
|
+
return _safe_error("discussions", A2LError("metadata endpoint returned duplicate forums"))
|
|
2921
|
+
try:
|
|
2922
|
+
existing_rows = _read_discussion_rows(course_dir / "_meta" / "discussions.json")
|
|
2923
|
+
except A2LError as exc:
|
|
2924
|
+
return _safe_error("discussions", exc)
|
|
2925
|
+
key = _discussion_key(vault)
|
|
2926
|
+
posts = [
|
|
2927
|
+
post for value in values if isinstance(value, dict) for post in _discussion_posts(value)
|
|
2928
|
+
]
|
|
2929
|
+
identities = {
|
|
2930
|
+
identity for post in posts if (identity := _discussion_identity(post)) is not None
|
|
2931
|
+
}
|
|
2932
|
+
pseudonyms = _discussion_pseudonyms(key, identities)
|
|
2933
|
+
incoming_rows: list[dict[str, object]] = []
|
|
2934
|
+
for value in values:
|
|
2935
|
+
if not isinstance(value, dict) or not isinstance(value.get("ForumId"), int):
|
|
2936
|
+
continue
|
|
2937
|
+
description = value.get("Description")
|
|
2938
|
+
description_text = _discussion_body_text(description, school_base=client.school.base_url)
|
|
2939
|
+
forum_row: dict[str, object] = {
|
|
2940
|
+
"id": value["ForumId"],
|
|
2941
|
+
"name": _safe_text(value.get("Name")),
|
|
2942
|
+
"description": description_text,
|
|
2943
|
+
}
|
|
2944
|
+
rendered_posts: list[dict[str, object]] = []
|
|
2945
|
+
for post in _discussion_posts(value):
|
|
2946
|
+
identity = _discussion_identity(post)
|
|
2947
|
+
body = _discussion_body_text(post.get("Body", post), school_base=client.school.base_url)
|
|
2948
|
+
author = _discussion_author(post, identity, pseudonyms, include_authors)
|
|
2949
|
+
rendered_posts.append(
|
|
2950
|
+
{
|
|
2951
|
+
"id": _discussion_post_id(post),
|
|
2952
|
+
"author": author,
|
|
2953
|
+
"text": body,
|
|
2954
|
+
"date": _first_string(post, "Date", "PostingDate", "LastModifiedDate"),
|
|
2955
|
+
}
|
|
2956
|
+
)
|
|
2957
|
+
if rendered_posts:
|
|
2958
|
+
forum_row["posts"] = rendered_posts
|
|
2959
|
+
incoming_rows.append(forum_row)
|
|
2960
|
+
|
|
2961
|
+
rows = _merge_discussion_rows(existing_rows, incoming_rows)
|
|
2962
|
+
_write_list(course_dir / "_meta" / "discussions.json", rows, root=vault.root)
|
|
2963
|
+
markdown_lines = ["# Discussions", ""]
|
|
2964
|
+
for forum_row in rows:
|
|
2965
|
+
markdown_lines.extend([f"## {forum_row['name'] or 'Untitled forum'}", ""])
|
|
2966
|
+
description_text = str(forum_row.get("description") or "")
|
|
2967
|
+
if description_text:
|
|
2968
|
+
markdown_lines.extend([description_text, ""])
|
|
2969
|
+
if forum_row.get("withdrawn_at"):
|
|
2970
|
+
markdown_lines.extend(["> No longer posted in LEARN.", ""])
|
|
2971
|
+
raw_posts = forum_row.get("posts", [])
|
|
2972
|
+
for post in raw_posts if isinstance(raw_posts, list) else []:
|
|
2973
|
+
if not isinstance(post, dict):
|
|
2974
|
+
continue
|
|
2975
|
+
markdown_lines.extend(
|
|
2976
|
+
[
|
|
2977
|
+
f"### {post.get('author') or 'author-unknown'}",
|
|
2978
|
+
"",
|
|
2979
|
+
str(post.get("text") or ""),
|
|
2980
|
+
"",
|
|
2981
|
+
]
|
|
2982
|
+
)
|
|
2983
|
+
|
|
2984
|
+
discussion_dir = course_dir / "discussions"
|
|
2985
|
+
paths.ensure_dir(discussion_dir, root=vault.root)
|
|
2986
|
+
paths.atomic_write_text(
|
|
2987
|
+
discussion_dir / "discussions.md", "\n".join(markdown_lines), root=vault.root
|
|
2988
|
+
)
|
|
2989
|
+
return None
|
|
2990
|
+
|
|
2991
|
+
|
|
2992
|
+
def _discussion_key(vault: Vault) -> bytes:
|
|
2993
|
+
private = vault.state() / "private"
|
|
2994
|
+
paths.ensure_dir(private, root=vault.root, mode=0o700)
|
|
2995
|
+
destination = private / "discussion-hmac.key"
|
|
2996
|
+
if paths.has_link_component(destination, root=vault.root):
|
|
2997
|
+
raise A2LError("discussion pseudonym key path contains a link component")
|
|
2998
|
+
try:
|
|
2999
|
+
with open(os.fspath(paths.long_path(destination)), "rb") as handle:
|
|
3000
|
+
key = handle.read()
|
|
3001
|
+
except FileNotFoundError:
|
|
3002
|
+
key = secrets.token_bytes(32)
|
|
3003
|
+
paths.atomic_write_bytes(destination, key, root=vault.root)
|
|
3004
|
+
if len(key) != 32:
|
|
3005
|
+
raise A2LError("discussion pseudonym key is invalid")
|
|
3006
|
+
return key
|
|
3007
|
+
|
|
3008
|
+
|
|
3009
|
+
def _discussion_posts(forum: Mapping[str, object]) -> list[dict[str, object]]:
|
|
3010
|
+
raw_topics = forum.get("Topics", forum.get("topics", []))
|
|
3011
|
+
if isinstance(raw_topics, dict):
|
|
3012
|
+
raw_topics = raw_topics.get("Items", raw_topics.get("Objects", []))
|
|
3013
|
+
raw_posts: list[object] = []
|
|
3014
|
+
if isinstance(raw_topics, list):
|
|
3015
|
+
for topic in raw_topics:
|
|
3016
|
+
if isinstance(topic, dict):
|
|
3017
|
+
posts = topic.get("Posts", topic.get("posts", []))
|
|
3018
|
+
if isinstance(posts, list):
|
|
3019
|
+
raw_posts.extend(posts)
|
|
3020
|
+
direct_posts = forum.get("Posts", forum.get("posts", []))
|
|
3021
|
+
if isinstance(direct_posts, list):
|
|
3022
|
+
raw_posts.extend(direct_posts)
|
|
3023
|
+
return [post for post in raw_posts if isinstance(post, dict)]
|
|
3024
|
+
|
|
3025
|
+
|
|
3026
|
+
def _read_discussion_rows(destination: Path) -> list[dict[str, object]]:
|
|
3027
|
+
rows = _read_list(destination)
|
|
3028
|
+
validated: list[dict[str, object]] = []
|
|
3029
|
+
seen_forums: set[str] = set()
|
|
3030
|
+
for row in rows:
|
|
3031
|
+
identifier = row.get("id")
|
|
3032
|
+
if isinstance(identifier, bool) or not isinstance(identifier, int):
|
|
3033
|
+
raise A2LError("discussions.json contains an invalid forum ID")
|
|
3034
|
+
forum_key = str(identifier)
|
|
3035
|
+
if forum_key in seen_forums:
|
|
3036
|
+
raise A2LError("discussions.json contains duplicate forum IDs")
|
|
3037
|
+
seen_forums.add(forum_key)
|
|
3038
|
+
raw_posts = row.get("posts", [])
|
|
3039
|
+
if not isinstance(raw_posts, list) or any(
|
|
3040
|
+
not isinstance(post, dict) or not _discussion_post_is_valid(post) for post in raw_posts
|
|
3041
|
+
):
|
|
3042
|
+
raise A2LError("discussions.json contains invalid posts")
|
|
3043
|
+
post_ids = [str(post["id"]) for post in raw_posts if isinstance(post, dict)]
|
|
3044
|
+
if len(set(post_ids)) != len(post_ids):
|
|
3045
|
+
raise A2LError("discussions.json contains duplicate post IDs")
|
|
3046
|
+
validated.append(dict(row))
|
|
3047
|
+
return validated
|
|
3048
|
+
|
|
3049
|
+
|
|
3050
|
+
def _merge_discussion_rows(
|
|
3051
|
+
existing: Sequence[Mapping[str, object]], incoming: Sequence[Mapping[str, object]]
|
|
3052
|
+
) -> list[dict[str, object]]:
|
|
3053
|
+
"""Union complete discussion captures by forum and post ID without deleting history."""
|
|
3054
|
+
by_forum: dict[str, dict[str, object]] = {
|
|
3055
|
+
str(row["id"]): dict(row) for row in existing if isinstance(row.get("id"), int)
|
|
3056
|
+
}
|
|
3057
|
+
incoming_forums: set[str] = set()
|
|
3058
|
+
for row in incoming:
|
|
3059
|
+
forum_id = row.get("id")
|
|
3060
|
+
if isinstance(forum_id, bool) or not isinstance(forum_id, int):
|
|
3061
|
+
raise A2LError("discussion capture contains an invalid forum ID")
|
|
3062
|
+
forum_key = str(forum_id)
|
|
3063
|
+
incoming_forums.add(forum_key)
|
|
3064
|
+
prior = by_forum.get(forum_key, {})
|
|
3065
|
+
merged = dict(prior)
|
|
3066
|
+
for field, value in row.items():
|
|
3067
|
+
if field == "posts":
|
|
3068
|
+
continue
|
|
3069
|
+
if field in {"name", "description"} and not value and prior.get(field):
|
|
3070
|
+
continue
|
|
3071
|
+
merged[field] = value
|
|
3072
|
+
old_posts = prior.get("posts", [])
|
|
3073
|
+
new_posts = row.get("posts", [])
|
|
3074
|
+
merged["posts"] = _merge_rows(
|
|
3075
|
+
old_posts if isinstance(old_posts, list) else [],
|
|
3076
|
+
new_posts if isinstance(new_posts, list) else [],
|
|
3077
|
+
id_field="id",
|
|
3078
|
+
complete=True,
|
|
3079
|
+
)
|
|
3080
|
+
merged["missing_since"] = None
|
|
3081
|
+
merged["withdrawn_at"] = None
|
|
3082
|
+
by_forum[forum_key] = merged
|
|
3083
|
+
|
|
3084
|
+
now = _now()
|
|
3085
|
+
for forum_key, row in by_forum.items():
|
|
3086
|
+
if forum_key in incoming_forums:
|
|
3087
|
+
continue
|
|
3088
|
+
if row.get("missing_since") is None:
|
|
3089
|
+
row["missing_since"] = now
|
|
3090
|
+
elif row.get("withdrawn_at") is None:
|
|
3091
|
+
row["withdrawn_at"] = now
|
|
3092
|
+
return sorted(by_forum.values(), key=lambda row: str(row.get("id")))
|
|
3093
|
+
|
|
3094
|
+
|
|
3095
|
+
def _discussion_forum_is_valid(forum: Mapping[str, object]) -> bool:
|
|
3096
|
+
"""Validate nested discussion containers before replacing a prior capture."""
|
|
3097
|
+
raw_topics = forum.get("Topics", forum.get("topics", []))
|
|
3098
|
+
if "Topics" in forum or "topics" in forum:
|
|
3099
|
+
if isinstance(raw_topics, dict):
|
|
3100
|
+
raw_topics = raw_topics.get("Items", raw_topics.get("Objects"))
|
|
3101
|
+
if not isinstance(raw_topics, list) or any(
|
|
3102
|
+
not isinstance(topic, dict) for topic in raw_topics
|
|
3103
|
+
):
|
|
3104
|
+
return False
|
|
3105
|
+
for topic in raw_topics:
|
|
3106
|
+
if "Posts" not in topic and "posts" not in topic:
|
|
3107
|
+
return False
|
|
3108
|
+
raw_posts = topic.get("Posts", topic.get("posts", []))
|
|
3109
|
+
if not isinstance(raw_posts, list) or any(
|
|
3110
|
+
not isinstance(post, dict) or not _discussion_post_is_valid(post)
|
|
3111
|
+
for post in raw_posts
|
|
3112
|
+
):
|
|
3113
|
+
return False
|
|
3114
|
+
|
|
3115
|
+
direct_posts = forum.get("Posts", forum.get("posts", []))
|
|
3116
|
+
if ("Posts" in forum or "posts" in forum) and (
|
|
3117
|
+
not isinstance(direct_posts, list)
|
|
3118
|
+
or any(
|
|
3119
|
+
not isinstance(post, dict) or not _discussion_post_is_valid(post)
|
|
3120
|
+
for post in direct_posts
|
|
3121
|
+
)
|
|
3122
|
+
):
|
|
3123
|
+
return False
|
|
3124
|
+
posts = _discussion_posts(forum)
|
|
3125
|
+
post_ids = [str(_discussion_post_id(post)) for post in posts]
|
|
3126
|
+
return len(set(post_ids)) == len(post_ids)
|
|
3127
|
+
|
|
3128
|
+
|
|
3129
|
+
def _discussion_post_is_valid(post: Mapping[str, object]) -> bool:
|
|
3130
|
+
identifier = _discussion_post_id(post)
|
|
3131
|
+
return identifier is not None and (not isinstance(identifier, str) or bool(identifier))
|
|
3132
|
+
|
|
3133
|
+
|
|
3134
|
+
def _discussion_identity(post: Mapping[str, object]) -> str | None:
|
|
3135
|
+
author = post.get("Author", post.get("author", post.get("User", post.get("user"))))
|
|
3136
|
+
if isinstance(author, dict):
|
|
3137
|
+
for key in ("Identifier", "UserId", "AuthorId", "Id", "id"):
|
|
3138
|
+
value = author.get(key)
|
|
3139
|
+
if isinstance(value, (str, int)) and not isinstance(value, bool) and str(value):
|
|
3140
|
+
return f"id:{value}"
|
|
3141
|
+
for key in ("Identifier", "UserId", "AuthorId"):
|
|
3142
|
+
value = post.get(key)
|
|
3143
|
+
if isinstance(value, (str, int)) and not isinstance(value, bool) and str(value):
|
|
3144
|
+
return f"id:{value}"
|
|
3145
|
+
candidates: list[object] = [author, post]
|
|
3146
|
+
for candidate in candidates:
|
|
3147
|
+
if not isinstance(candidate, dict):
|
|
3148
|
+
continue
|
|
3149
|
+
for key in ("DisplayName", "Name", "UserName", "name"):
|
|
3150
|
+
value = candidate.get(key)
|
|
3151
|
+
if isinstance(value, str) and value.strip():
|
|
3152
|
+
normalized = unicodedata.normalize("NFKC", value)
|
|
3153
|
+
normalized = re.sub(r"\s+", " ", normalized).strip().casefold()
|
|
3154
|
+
return f"name:{normalized}"
|
|
3155
|
+
return None
|
|
3156
|
+
|
|
3157
|
+
|
|
3158
|
+
def _discussion_pseudonyms(key: bytes, identities: Iterable[str]) -> dict[str, str]:
|
|
3159
|
+
grouped: dict[str, list[str]] = defaultdict(list)
|
|
3160
|
+
for identity in identities:
|
|
3161
|
+
digest = hmac.new(key, identity.encode("utf-8"), "sha256").hexdigest()
|
|
3162
|
+
grouped[digest[:20]].append(identity)
|
|
3163
|
+
result: dict[str, str] = {}
|
|
3164
|
+
for digest, values in grouped.items():
|
|
3165
|
+
for identity in sorted(values):
|
|
3166
|
+
suffix = ""
|
|
3167
|
+
if len(values) > 1:
|
|
3168
|
+
suffix = "-" + sha256(identity.encode("utf-8")).hexdigest()[:16]
|
|
3169
|
+
result[identity] = f"author-{digest}{suffix}"
|
|
3170
|
+
return result
|
|
3171
|
+
|
|
3172
|
+
|
|
3173
|
+
def _discussion_author(
|
|
3174
|
+
post: Mapping[str, object],
|
|
3175
|
+
identity: str | None,
|
|
3176
|
+
pseudonyms: Mapping[str, str],
|
|
3177
|
+
include_authors: bool,
|
|
3178
|
+
) -> str:
|
|
3179
|
+
if include_authors:
|
|
3180
|
+
author = post.get("Author", post.get("author", post.get("User", post.get("user"))))
|
|
3181
|
+
if isinstance(author, dict):
|
|
3182
|
+
return (
|
|
3183
|
+
_first_string(author, "DisplayName", "Name", "UserName")
|
|
3184
|
+
or _first_string(post, "DisplayName", "Name")
|
|
3185
|
+
or "unknown-author"
|
|
3186
|
+
)
|
|
3187
|
+
return pseudonyms.get(identity or "", "author-unknown")
|
|
3188
|
+
|
|
3189
|
+
|
|
3190
|
+
def _discussion_post_id(post: Mapping[str, object]) -> str | int | None:
|
|
3191
|
+
for key in ("PostId", "Id", "id"):
|
|
3192
|
+
value = post.get(key)
|
|
3193
|
+
if isinstance(value, (str, int)) and not isinstance(value, bool):
|
|
3194
|
+
return value
|
|
3195
|
+
return None
|
|
3196
|
+
|
|
3197
|
+
|
|
3198
|
+
def _discussion_body_text(value: object, *, school_base: str) -> str:
|
|
3199
|
+
if isinstance(value, dict):
|
|
3200
|
+
html_value = _first_string(value, "Html", "HTML", "html")
|
|
3201
|
+
text_value = _first_string(value, "Text", "text")
|
|
3202
|
+
value = html_value if html_value is not None else text_value
|
|
3203
|
+
if not isinstance(value, str):
|
|
3204
|
+
return ""
|
|
3205
|
+
sanitized = _sanitize_richtext(value, school_base)
|
|
3206
|
+
return _richtext_markdown("", sanitized).split("\n\n", 1)[-1].strip()
|
|
3207
|
+
|
|
3208
|
+
|
|
3209
|
+
def _update_course_index_from_map(course_dir: Path) -> None:
|
|
3210
|
+
del course_dir
|
|
3211
|
+
|
|
3212
|
+
|
|
3213
|
+
__all__ = [
|
|
3214
|
+
"CourseMetadata",
|
|
3215
|
+
"FetchReport",
|
|
3216
|
+
"FileReport",
|
|
3217
|
+
"MetadataReport",
|
|
3218
|
+
"OutlineReport",
|
|
3219
|
+
"PRIORITY_BUDGET_BYTES",
|
|
3220
|
+
"TopicRecord",
|
|
3221
|
+
"fetch_topic",
|
|
3222
|
+
"is_downloadable_topic",
|
|
3223
|
+
"is_media_topic",
|
|
3224
|
+
"ingest_files",
|
|
3225
|
+
"ingest_metadata",
|
|
3226
|
+
"load_metadata_report",
|
|
3227
|
+
"load_metadata_topics",
|
|
3228
|
+
"select_priority_topics",
|
|
3229
|
+
]
|