agent2learn 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. agent2learn/__init__.py +3 -0
  2. agent2learn/_release.py +19 -0
  3. agent2learn/aipolicy.py +182 -0
  4. agent2learn/api.py +590 -0
  5. agent2learn/audit.py +358 -0
  6. agent2learn/auth/__init__.py +282 -0
  7. agent2learn/auth/cdp.py +1067 -0
  8. agent2learn/auth/paste.py +378 -0
  9. agent2learn/calendar.py +525 -0
  10. agent2learn/calibrate.py +347 -0
  11. agent2learn/check.py +1091 -0
  12. agent2learn/cli.py +2039 -0
  13. agent2learn/clock.py +39 -0
  14. agent2learn/config.py +205 -0
  15. agent2learn/console.py +229 -0
  16. agent2learn/convert.py +1223 -0
  17. agent2learn/doctor.py +1167 -0
  18. agent2learn/errors.py +32 -0
  19. agent2learn/ground.py +735 -0
  20. agent2learn/index.py +614 -0
  21. agent2learn/ingest.py +3229 -0
  22. agent2learn/locations.py +247 -0
  23. agent2learn/outlines.py +754 -0
  24. agent2learn/paths.py +683 -0
  25. agent2learn/pipeline.py +392 -0
  26. agent2learn/privacy.py +1123 -0
  27. agent2learn/schools/__init__.py +29 -0
  28. agent2learn/schools/_base.py +194 -0
  29. agent2learn/schools/generic.py +78 -0
  30. agent2learn/schools/uwaterloo.py +66 -0
  31. agent2learn/session.py +373 -0
  32. agent2learn/skills.py +1081 -0
  33. agent2learn/snapshot.py +399 -0
  34. agent2learn/submit.py +1047 -0
  35. agent2learn/transactions.py +157 -0
  36. agent2learn/upgrade.py +288 -0
  37. agent2learn/vault.py +1134 -0
  38. agent2learn-0.1.2.data/data/a2l-coursework/SKILL.md +52 -0
  39. agent2learn-0.1.2.data/data/a2l-setup/SKILL.md +27 -0
  40. agent2learn-0.1.2.data/data/a2l-study/SKILL.md +27 -0
  41. agent2learn-0.1.2.data/data/a2l-sync/SKILL.md +30 -0
  42. agent2learn-0.1.2.dist-info/METADATA +186 -0
  43. agent2learn-0.1.2.dist-info/RECORD +46 -0
  44. agent2learn-0.1.2.dist-info/WHEEL +4 -0
  45. agent2learn-0.1.2.dist-info/entry_points.txt +3 -0
  46. agent2learn-0.1.2.dist-info/licenses/LICENSE +202 -0
agent2learn/ingest.py ADDED
@@ -0,0 +1,3229 @@
1
+ """Metadata-first, revision-safe ingestion of a LEARN course vault.
2
+
3
+ The module deliberately separates the inexpensive JSON phase from the potentially large file
4
+ phase. Metadata is a typed projection of the API, not a raw response archive: external URLs are
5
+ classified before persistence, and only a query-free LEARN view URL plus a destination hostname
6
+ can survive as a link stub.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import hmac
12
+ import json
13
+ import os
14
+ import re
15
+ import secrets
16
+ import stat
17
+ import tempfile
18
+ import unicodedata
19
+ from collections import defaultdict
20
+ from collections.abc import Callable, Iterable, Mapping, Sequence
21
+ from contextlib import suppress
22
+ from dataclasses import dataclass, replace
23
+ from hashlib import sha256
24
+ from html import escape
25
+ from html.parser import HTMLParser
26
+ from pathlib import Path, PurePosixPath
27
+ from typing import Any, Literal, Protocol, cast
28
+ from urllib.parse import urljoin, urlsplit, urlunsplit
29
+
30
+ from requests import RequestException
31
+
32
+ from agent2learn import api, clock, locations, paths, snapshot, transactions
33
+ from agent2learn import index as course_index
34
+ from agent2learn.api import DownloadError, DownloadResult, FileTooLarge
35
+ from agent2learn.calibrate import CourseRef, calibrate, load_calibration
36
+ from agent2learn.convert import DEFAULT_OCR_WORDS_PER_PAGE, _validate_threshold, convert_vault
37
+ from agent2learn.errors import A2LError, NotConfigured, SessionExpired
38
+ from agent2learn.schools import (
39
+ School,
40
+ hostname_matches_suffix,
41
+ parse_api_timestamp,
42
+ topic_is_excluded,
43
+ )
44
+ from agent2learn.vault import DerivedArtifact, ManifestEntry, Vault
45
+
46
+ CONTENT_MAP_VERSION = course_index.CONTENT_MAP_VERSION
47
+ DEFAULT_SCOPE: Literal["all", "priority"] = "all"
48
+ PRIORITY_BUDGET_BYTES = 200_000_000
49
+ _MAX_PAGES = 1000
50
+ _COURSE_CONTENT = "content"
51
+ _OFFICE_LOCK = re.compile(r"^~\$", re.IGNORECASE)
52
+ _MEDIA_SUFFIXES = frozenset(
53
+ {
54
+ ".3gp",
55
+ ".aac",
56
+ ".avi",
57
+ ".flac",
58
+ ".m4a",
59
+ ".m4v",
60
+ ".mkv",
61
+ ".mov",
62
+ ".mp3",
63
+ ".mp4",
64
+ ".mpeg",
65
+ ".mpg",
66
+ ".ogg",
67
+ ".wav",
68
+ ".webm",
69
+ ".wmv",
70
+ }
71
+ )
72
+ _DOWNLOADABLE_KINDS = frozenset({"file", "html", "htmlfile"})
73
+ _PENDING_MARKER_SUFFIX = ".meta.json"
74
+ _PENDING_INSTALL_SUFFIX = ".part" + _PENDING_MARKER_SUFFIX
75
+ _PENDING_INSTALL_KEYS = frozenset(
76
+ {
77
+ "version",
78
+ "source_key",
79
+ "destination",
80
+ "sha256",
81
+ "size",
82
+ "etag",
83
+ "last_modified",
84
+ "prior_sha256",
85
+ "revision_preserved",
86
+ }
87
+ )
88
+
89
+
90
+ @dataclass(frozen=True)
91
+ class TopicRecord:
92
+ """The safe, typed projection of one content topic."""
93
+
94
+ source_key: str
95
+ source_id: str
96
+ topic_id: int
97
+ course_org_unit_id: int
98
+ course_code: str
99
+ course_name: str
100
+ term: str | None
101
+ title: str
102
+ kind: str
103
+ module_path: tuple[str, ...]
104
+ module_ids: tuple[int, ...]
105
+ view_url: str
106
+ outline_url: str | None
107
+ url_path: str | None
108
+ external_host: str | None
109
+ etag: str | None
110
+ last_modified: str | None
111
+ is_broken: bool
112
+ availability: str = "metadata_only"
113
+ source_path: str | None = None
114
+ path: str | None = None
115
+ sha256: str | None = None
116
+ size: int | None = None
117
+ stub_path: str | None = None
118
+ remote_size: int | None = None
119
+ next_action: str = "a2l fetch <topic-id>"
120
+ missing_since: str | None = None
121
+ withdrawn_at: str | None = None
122
+
123
+
124
+ @dataclass(frozen=True)
125
+ class CourseMetadata:
126
+ """Metadata and safe content projections for one selected course."""
127
+
128
+ course: CourseRef
129
+ directory: Path
130
+ topics: tuple[TopicRecord, ...]
131
+ module_tree: tuple[dict[str, object], ...]
132
+
133
+
134
+ @dataclass(frozen=True)
135
+ class MetadataReport:
136
+ """Result of the complete cheap metadata phase."""
137
+
138
+ courses: tuple[CourseMetadata, ...]
139
+ topic_count: int
140
+ deadline_count: int
141
+ errors: tuple[str, ...] = ()
142
+ exit_code: int = 0
143
+
144
+
145
+ @dataclass(frozen=True)
146
+ class FileReport:
147
+ """Result of a resumable source-file phase."""
148
+
149
+ downloaded: int = 0
150
+ skipped: int = 0
151
+ failed: int = 0
152
+ metadata_only: int = 0
153
+ download_gaps: int = 0
154
+ interrupted: bool = False
155
+ errors: tuple[str, ...] = ()
156
+ exit_code: int = 0
157
+
158
+
159
+ @dataclass(frozen=True)
160
+ class FetchReport:
161
+ """Result of resolving and fetching one stable topic identity."""
162
+
163
+ source_key: str
164
+ availability: str
165
+ source_path: str | None
166
+ citation_path: str | None
167
+ changed: bool
168
+ next_action: str | None = None
169
+
170
+
171
+ @dataclass(frozen=True)
172
+ class OutlineReport:
173
+ """The outline renderer's bounded result; implemented in :mod:`outlines`."""
174
+
175
+ rendered: int = 0
176
+ unavailable: int = 0
177
+ errors: tuple[str, ...] = ()
178
+
179
+
180
+ @dataclass(frozen=True)
181
+ class _PendingInstall:
182
+ """Durable proof that a validated download is waiting only on filesystem installation."""
183
+
184
+ marker: Path
185
+ part: Path
186
+ source_key: str
187
+ destination: str
188
+ sha256: str
189
+ size: int
190
+ etag: str | None
191
+ last_modified: str | None
192
+ prior_sha256: str | None
193
+ revision_preserved: bool
194
+ fetched_at: str | None = None
195
+
196
+
197
+ class IngestClient(Protocol):
198
+ """The small calibrated client surface required by the ingest pipeline.
199
+
200
+ Keeping this as a protocol lets the offline fixture client exercise the same code paths as the
201
+ real HTTP client without pretending that a test double is an authenticated ``api.Client``.
202
+ """
203
+
204
+ @property
205
+ def school(self) -> School: ...
206
+
207
+ lp_version: str | None
208
+ le_version: str | None
209
+ download_template: str | None
210
+
211
+ def get_json(self, path: str) -> object: ...
212
+
213
+ def download(
214
+ self,
215
+ url: str,
216
+ temp: Path,
217
+ *,
218
+ prior: ManifestEntry | None = None,
219
+ max_bytes: int | None = api.DEFAULT_MAX_BYTES,
220
+ is_html_topic: bool = False,
221
+ root: Path | None = None,
222
+ ) -> DownloadResult: ...
223
+
224
+
225
+ def ingest_metadata(
226
+ client: IngestClient,
227
+ vault: Vault,
228
+ school: School,
229
+ *,
230
+ term: str | None = None,
231
+ only: Iterable[int | str] | None = None,
232
+ include_grades: bool = False,
233
+ create_snapshot: bool = True,
234
+ ) -> MetadataReport:
235
+ """Fetch and persist complete typed metadata for the selected courses.
236
+
237
+ No file endpoint is touched here. The function is safe to call before the user chooses a file
238
+ scope, and every category writer merges stable IDs rather than deleting expired records. Direct
239
+ callers retain the historical metadata snapshot by default; the production pipeline disables
240
+ that interim write and creates its single snapshot after conversion and index reconciliation.
241
+ """
242
+
243
+ courses = _selected_courses(client, term=term, only=only)
244
+ reports: list[CourseMetadata] = []
245
+ errors: list[str] = []
246
+ deadline_count = 0
247
+
248
+ for course in courses:
249
+ course_dir = _course_directory(vault, school, course)
250
+ try:
251
+ paths.ensure_dir(course_dir, root=vault.root)
252
+ except ValueError as exc:
253
+ raise A2LError("vault course path contains a link component") from exc
254
+ try:
255
+ toc_payload, toc_complete, toc_error = _fetch_one(client, _toc_path(client, course))
256
+ if toc_error is not None:
257
+ errors.append(_safe_error("toc", toc_error))
258
+ records, module_tree, toc_valid = _topics_from_toc(
259
+ toc_payload, course=course, school=school
260
+ )
261
+ if not toc_valid and toc_error is None:
262
+ errors.append("toc: invalid response")
263
+ toc_complete = toc_complete and toc_valid
264
+ except SessionExpired:
265
+ raise
266
+ except Exception as exc:
267
+ errors.append(_safe_error("toc", exc))
268
+ records, module_tree, toc_complete = [], [], False
269
+
270
+ assignments, assignments_complete, assignments_error = _fetch_collection(
271
+ client, _endpoint_path(client, course, "dropbox/folders/")
272
+ )
273
+ if assignments_error is not None:
274
+ errors.append(_safe_error("assignments", assignments_error))
275
+ existing_map = _read_content_map(course_dir)
276
+ attachment_topics = _assignment_attachment_topics(assignments, course=course, school=school)
277
+ merged_topics = _merge_topic_records(
278
+ existing_map.get("topics", []),
279
+ [*records, *attachment_topics],
280
+ # Attachment identities come from Dropbox as well as the TOC. A failure in either
281
+ # listing must not mark a previously captured source absent.
282
+ complete=toc_complete and assignments_complete,
283
+ )
284
+ merged_topics = _materialize_external_stubs(
285
+ merged_topics, course_dir=course_dir, vault=vault, school=school, course=course
286
+ )
287
+
288
+ if toc_complete:
289
+ _write_toc(course_dir, module_tree, root=vault.root)
290
+ else:
291
+ try:
292
+ module_tree = _read_toc_modules(course_dir)
293
+ except A2LError as exc:
294
+ # A failed TOC fetch must not turn a corrupt cached tree into an apparently empty
295
+ # course. Keep the metadata phase usable, but make the coverage gap explicit.
296
+ errors.append(_safe_error("toc cache", exc))
297
+ module_tree = []
298
+
299
+ assignments_rows = _project_assignments(assignments)
300
+ active_assignments = {str(row["id"]) for row in assignments_rows}
301
+ previous_assignments = _read_list(course_dir / "_meta" / "assignments.json")
302
+ assignments_rows = _merge_rows(
303
+ previous_assignments,
304
+ assignments_rows,
305
+ id_field="id",
306
+ complete=assignments_complete,
307
+ )
308
+ assignment_directories = locations.assignment_directories(
309
+ vault,
310
+ course_dir,
311
+ school,
312
+ course,
313
+ assignments_rows,
314
+ previous_assignments,
315
+ active_assignments,
316
+ )
317
+ _write_list(course_dir / "_meta" / "assignments.json", assignments_rows, root=vault.root)
318
+ assignment_artifacts = _materialize_assignments(
319
+ assignments,
320
+ course_dir=course_dir,
321
+ vault=vault,
322
+ school=school,
323
+ course=course,
324
+ directories=assignment_directories,
325
+ )
326
+ for row in assignments_rows:
327
+ artifact = assignment_artifacts.get(str(row.get("id")))
328
+ if artifact is not None:
329
+ row.update(artifact)
330
+ _write_list(course_dir / "_meta" / "assignments.json", assignments_rows, root=vault.root)
331
+
332
+ news, news_complete, news_error = _fetch_collection(
333
+ client, _endpoint_path(client, course, "news/")
334
+ )
335
+ if news_error is not None:
336
+ errors.append(_safe_error("news", news_error))
337
+ news_rows = _project_news(news)
338
+ news_rows = _merge_rows(
339
+ _read_list(course_dir / "_meta" / "news.json"),
340
+ news_rows,
341
+ id_field="id",
342
+ complete=news_complete,
343
+ )
344
+ _write_list(course_dir / "_meta" / "news.json", news_rows, root=vault.root)
345
+ _write_announcements(
346
+ course_dir / "announcements" / "announcements.md", news_rows, root=vault.root
347
+ )
348
+
349
+ quizzes, quizzes_complete, quizzes_error = _fetch_collection(
350
+ client, _endpoint_path(client, course, "quizzes/")
351
+ )
352
+ if quizzes_error is not None:
353
+ errors.append(_safe_error("quizzes", quizzes_error))
354
+ quiz_rows = _project_quizzes(quizzes)
355
+ quiz_rows = _merge_rows(
356
+ _read_list(course_dir / "_meta" / "quizzes.json"),
357
+ quiz_rows,
358
+ id_field="id",
359
+ complete=quizzes_complete,
360
+ )
361
+ _write_list(course_dir / "_meta" / "quizzes.json", quiz_rows, root=vault.root)
362
+
363
+ if include_grades:
364
+ grades, grades_complete, grades_error = _fetch_collection(
365
+ client, _endpoint_path(client, course, "grades/values/myGradeValues/")
366
+ )
367
+ if grades_error is not None:
368
+ errors.append(_safe_error("grades", grades_error))
369
+ grades_path = course_dir / "_meta" / "my_grades.json"
370
+ if grades_complete:
371
+ grade_rows = _merge_rows(
372
+ _read_list(grades_path),
373
+ _project_grades(grades),
374
+ id_field="id",
375
+ complete=True,
376
+ )
377
+ _write_list(grades_path, grade_rows, root=vault.root)
378
+ else:
379
+ # Grade values are sensitive, but they still obey merge-not-replace: an
380
+ # incomplete response must never erase the last complete opt-in snapshot.
381
+ partial_grades = _project_grades(grades)
382
+ if partial_grades:
383
+ grade_rows = _merge_rows(
384
+ _read_list(grades_path),
385
+ partial_grades,
386
+ id_field="id",
387
+ complete=False,
388
+ )
389
+ _write_list(grades_path, grade_rows, root=vault.root)
390
+ errors.append("grades: incomplete response")
391
+
392
+ merged_topics = course_index.reconcile_content_map(vault, merged_topics)
393
+ _write_content_map(course_dir, merged_topics, root=vault.root)
394
+ _materialize_submission_only_readmes(
395
+ assignments,
396
+ artifacts=assignment_artifacts,
397
+ course_dir=course_dir,
398
+ directories=assignment_directories,
399
+ topics=merged_topics,
400
+ root=vault.root,
401
+ )
402
+ deadline_count += sum(1 for row in assignments_rows + quiz_rows if row.get("due_date"))
403
+ typed_topics = tuple(_topic_from_row(row, course=course) for row in merged_topics)
404
+ _write_index(course_dir, school=school, course=course, topics=typed_topics, root=vault.root)
405
+ reports.append(
406
+ CourseMetadata(
407
+ course=course,
408
+ directory=course_dir,
409
+ topics=typed_topics,
410
+ module_tree=tuple(module_tree),
411
+ )
412
+ )
413
+
414
+ if create_snapshot:
415
+ snapshot.write_snapshot(
416
+ vault,
417
+ [report.directory for report in reports],
418
+ include_grades=include_grades,
419
+ timestamp=_now(),
420
+ )
421
+ return MetadataReport(
422
+ courses=tuple(reports),
423
+ topic_count=sum(len(report.topics) for report in reports),
424
+ deadline_count=deadline_count,
425
+ errors=tuple(errors),
426
+ )
427
+
428
+
429
+ def load_metadata_topics(
430
+ vault: Vault, school: School, courses: Iterable[CourseRef]
431
+ ) -> tuple[TopicRecord, ...]:
432
+ """Load validated topic projections for a completed local metadata phase.
433
+
434
+ This is intentionally a local read. It lets a resumed onboarding run show an honest file
435
+ estimate without repeating the network metadata phase, while using the same row decoder and
436
+ filename/media rules as :func:`ingest_files`.
437
+ """
438
+
439
+ topics: list[TopicRecord] = []
440
+ for course in courses:
441
+ course_dir = _course_directory(vault, school, course)
442
+ content_map_path = course_dir / "_meta" / "content_map.json"
443
+ if paths.is_link(content_map_path) or not paths.long_path(content_map_path).is_file():
444
+ raise A2LError("course metadata is unavailable; run a2l init")
445
+ content_map = _read_content_map(course_dir)
446
+ topics.extend(_topic_from_row(row, course=course) for row in _map_topics(content_map))
447
+ return tuple(topics)
448
+
449
+
450
+ def load_metadata_report(
451
+ vault: Vault, school: School, courses: Iterable[CourseRef]
452
+ ) -> MetadataReport:
453
+ """Reconstruct a completed metadata report from validated local course projections."""
454
+ reports: list[CourseMetadata] = []
455
+ deadline_count = 0
456
+ for course in courses:
457
+ course_dir = _course_directory(vault, school, course)
458
+ topics = load_metadata_topics(vault, school, [course])
459
+ module_tree = tuple(_read_toc_modules(course_dir))
460
+ deadline_count += sum(
461
+ 1
462
+ for row in [
463
+ *_read_list(course_dir / "_meta" / "assignments.json"),
464
+ *_read_list(course_dir / "_meta" / "quizzes.json"),
465
+ ]
466
+ if row.get("due_date")
467
+ )
468
+ reports.append(
469
+ CourseMetadata(
470
+ course=course,
471
+ directory=course_dir,
472
+ topics=topics,
473
+ module_tree=module_tree,
474
+ )
475
+ )
476
+ return MetadataReport(
477
+ courses=tuple(reports),
478
+ topic_count=sum(len(report.topics) for report in reports),
479
+ deadline_count=deadline_count,
480
+ )
481
+
482
+
483
+ def is_downloadable_topic(topic: TopicRecord, *, include_media: bool = False) -> bool:
484
+ """Return whether a topic is eligible for the selected bulk file plan."""
485
+ return (
486
+ topic.availability != "external_link"
487
+ and topic.url_path is not None
488
+ and topic.kind.casefold() in _DOWNLOADABLE_KINDS
489
+ and not topic.is_broken
490
+ and not _is_office_lock(topic)
491
+ and (include_media or not _is_media(topic))
492
+ )
493
+
494
+
495
+ def select_priority_topics(
496
+ rows: Sequence[TopicRecord],
497
+ *,
498
+ budget: int = PRIORITY_BUDGET_BYTES,
499
+ include_media: bool = False,
500
+ ) -> tuple[TopicRecord, ...]:
501
+ """Return the deterministic priority subset used by ingest and onboarding estimates."""
502
+ if isinstance(budget, bool) or not isinstance(budget, int) or budget <= 0:
503
+ raise ValueError("priority budget must be a positive integer")
504
+ candidates = [
505
+ topic for topic in rows if is_downloadable_topic(topic, include_media=include_media)
506
+ ]
507
+ return tuple(_priority_rows(candidates, scope="priority", budget=budget))
508
+
509
+
510
+ def ingest_files(
511
+ client: IngestClient,
512
+ vault: Vault,
513
+ school: School,
514
+ *,
515
+ term: str | None = None,
516
+ only: Iterable[int | str] | None = None,
517
+ scope: Literal["all", "priority"] = DEFAULT_SCOPE,
518
+ include_media: bool = False,
519
+ priority_budget_bytes: int = PRIORITY_BUDGET_BYTES,
520
+ include_discussions: bool = False,
521
+ discussion_authors: bool = False,
522
+ ) -> FileReport:
523
+ """Download an explicit, resumable source scope after metadata is available."""
524
+
525
+ if scope not in {"all", "priority"}:
526
+ raise ValueError("scope must be 'all' or 'priority'")
527
+ if isinstance(priority_budget_bytes, bool) or not isinstance(priority_budget_bytes, int):
528
+ raise ValueError("priority_budget_bytes must be an integer")
529
+ if priority_budget_bytes <= 0:
530
+ raise ValueError("priority_budget_bytes must be positive")
531
+
532
+ selected = _selected_courses(client, term=term, only=only)
533
+ downloaded = skipped = failed = metadata_only = download_gaps = 0
534
+ errors: list[str] = []
535
+
536
+ for course in selected:
537
+ course_dir = _course_directory(vault, school, course)
538
+ content_map = _read_content_map(course_dir)
539
+ rows = [_topic_from_row(row, course=course) for row in _map_topics(content_map)]
540
+ if not paths.long_path(course_dir / "_meta" / "content_map.json").is_file():
541
+ # Keep the public entry point safe when called directly: metadata remains a separate
542
+ # phase, but a missing map is a configuration problem rather than a silent no-op.
543
+ raise A2LError("course metadata is unavailable; run ingest_metadata first")
544
+
545
+ planned = _plan_file_paths(rows, course_dir=course_dir, vault=vault, scope=scope)
546
+ chosen = (
547
+ list(
548
+ select_priority_topics(
549
+ planned,
550
+ budget=priority_budget_bytes,
551
+ include_media=include_media,
552
+ )
553
+ )
554
+ if scope == "priority"
555
+ else list(planned)
556
+ )
557
+ if include_discussions:
558
+ discussion_error = _ingest_discussions(
559
+ client,
560
+ course,
561
+ course_dir,
562
+ vault,
563
+ include_authors=discussion_authors,
564
+ )
565
+ if discussion_error is not None:
566
+ errors.append(discussion_error)
567
+
568
+ for topic in chosen:
569
+ if topic.availability == "external_link":
570
+ skipped += 1
571
+ continue
572
+ if topic.url_path is None or topic.kind.casefold() not in _DOWNLOADABLE_KINDS:
573
+ skipped += 1
574
+ metadata_only += 1
575
+ _update_row_state(
576
+ course_dir,
577
+ topic.source_key,
578
+ root=vault.root,
579
+ availability="metadata_only",
580
+ next_action="topic is metadata-only until explicitly fetched",
581
+ )
582
+ continue
583
+ if _is_office_lock(topic):
584
+ skipped += 1
585
+ metadata_only += 1
586
+ _update_row_state(
587
+ course_dir,
588
+ topic.source_key,
589
+ root=vault.root,
590
+ availability="metadata_only",
591
+ next_action="office lock file skipped",
592
+ )
593
+ continue
594
+ if _is_media(topic) and not include_media:
595
+ skipped += 1
596
+ metadata_only += 1
597
+ _update_row_state(
598
+ course_dir,
599
+ topic.source_key,
600
+ root=vault.root,
601
+ availability="metadata_only",
602
+ next_action="media excluded; rerun with --include-media",
603
+ )
604
+ continue
605
+ if topic.is_broken:
606
+ skipped += 1
607
+ metadata_only += 1
608
+ _update_row_state(
609
+ course_dir,
610
+ topic.source_key,
611
+ root=vault.root,
612
+ availability="metadata_only",
613
+ next_action="topic is marked broken in LEARN",
614
+ )
615
+ continue
616
+ if topic.remote_size is not None and topic.remote_size > api.DEFAULT_MAX_BYTES:
617
+ skipped += 1
618
+ metadata_only += 1
619
+ _update_row_state(
620
+ course_dir,
621
+ topic.source_key,
622
+ root=vault.root,
623
+ availability="metadata_only",
624
+ next_action=f"a2l fetch --allow-large {topic.source_id}",
625
+ )
626
+ continue
627
+
628
+ try:
629
+ result = _ingest_one_topic(client, vault, school, course_dir, topic)
630
+ except KeyboardInterrupt:
631
+ return FileReport(
632
+ downloaded=downloaded,
633
+ skipped=skipped,
634
+ failed=failed,
635
+ metadata_only=metadata_only,
636
+ download_gaps=download_gaps,
637
+ interrupted=True,
638
+ errors=tuple(errors),
639
+ exit_code=130,
640
+ )
641
+ except SessionExpired:
642
+ raise
643
+ except FileTooLarge:
644
+ skipped += 1
645
+ metadata_only += 1
646
+ _update_row_state(
647
+ course_dir,
648
+ topic.source_key,
649
+ root=vault.root,
650
+ availability="metadata_only",
651
+ next_action=f"a2l fetch --allow-large {topic.source_id}",
652
+ )
653
+ continue
654
+ except api.DiskSpaceExhausted as exc:
655
+ failed += 1
656
+ errors.append(_safe_error("download", exc))
657
+ continue
658
+ except DownloadError as exc:
659
+ download_gaps += 1
660
+ _update_row_state(
661
+ course_dir,
662
+ topic.source_key,
663
+ root=vault.root,
664
+ availability="download_gap",
665
+ next_action=(
666
+ f"download failed ({type(exc).__name__}); "
667
+ f"retry: a2l fetch {topic.source_id}"
668
+ ),
669
+ )
670
+ continue
671
+
672
+ if result == "downloaded":
673
+ downloaded += 1
674
+ else:
675
+ skipped += 1
676
+
677
+ return FileReport(
678
+ downloaded=downloaded,
679
+ skipped=skipped,
680
+ failed=failed,
681
+ metadata_only=metadata_only,
682
+ download_gaps=download_gaps,
683
+ errors=tuple(errors),
684
+ )
685
+
686
+
687
+ def fetch_topic(
688
+ client: IngestClient,
689
+ vault: Vault,
690
+ school: School,
691
+ topic: str,
692
+ *,
693
+ allow_large: bool = False,
694
+ confirm: Callable[[int | None], bool] | None = None,
695
+ ocr_words_per_page: int = DEFAULT_OCR_WORDS_PER_PAGE,
696
+ ) -> FetchReport:
697
+ """Resolve one stable topic ID/path/title and fetch only that source."""
698
+
699
+ _validate_threshold(ocr_words_per_page)
700
+ match = _resolve_topic(vault, topic)
701
+ if match is None:
702
+ raise A2LError(f"topic not found: {topic}")
703
+ course, record, course_dir = match
704
+ if record.availability == "external_link":
705
+ raise A2LError("external or licensed topics are link stubs and cannot be fetched")
706
+ known_oversized = record.remote_size is not None and record.remote_size > api.DEFAULT_MAX_BYTES
707
+ unbounded = False
708
+ if known_oversized or (record.remote_size is None and allow_large):
709
+ if not allow_large:
710
+ raise A2LError(
711
+ "source size exceeds the default limit; run: "
712
+ f"a2l fetch --allow-large {record.source_id}"
713
+ )
714
+ if confirm is None or not confirm(record.remote_size):
715
+ raise A2LError("large-file fetch cancelled")
716
+ unbounded = True
717
+
718
+ planned = _plan_file_paths([record], course_dir=course_dir, vault=vault, scope="all")
719
+ if not planned or planned[0].source_path is None:
720
+ raise A2LError("topic has no fetchable first-party source")
721
+ record = planned[0]
722
+
723
+ result = _ingest_one_topic(
724
+ client,
725
+ vault,
726
+ school,
727
+ course_dir,
728
+ record,
729
+ max_bytes=None if unbounded else api.DEFAULT_MAX_BYTES,
730
+ )
731
+ conversion = convert_vault(
732
+ vault, source_keys=(record.source_key,), ocr_words_per_page=ocr_words_per_page
733
+ )
734
+ refreshed = _topic_from_row(
735
+ _find_content_row(course_dir, record.source_key) or _topic_to_row(record), course=course
736
+ )
737
+ return FetchReport(
738
+ source_key=record.source_key,
739
+ availability=refreshed.availability,
740
+ source_path=refreshed.source_path,
741
+ citation_path=refreshed.path,
742
+ changed=result == "downloaded" or conversion.converted > 0,
743
+ next_action=refreshed.next_action,
744
+ )
745
+
746
+
747
+ def _selected_courses(
748
+ client: IngestClient,
749
+ *,
750
+ term: str | None,
751
+ only: Iterable[int | str] | None,
752
+ ) -> list[CourseRef]:
753
+ raw_courses: object = getattr(client, "courses", None)
754
+ if raw_courses is None:
755
+ calibration = getattr(client, "calibration", None)
756
+ if calibration is None:
757
+ try:
758
+ calibration = load_calibration()
759
+ except NotConfigured:
760
+ calibration = calibrate(client)
761
+ raw_courses = getattr(calibration, "courses", None)
762
+ if getattr(client, "lp_version", None) is None:
763
+ client.lp_version = getattr(calibration, "lp", None)
764
+ if getattr(client, "le_version", None) is None:
765
+ client.le_version = getattr(calibration, "le", None)
766
+ if client.download_template is None:
767
+ client.download_template = getattr(calibration, "download_template", None)
768
+
769
+ if not isinstance(raw_courses, Sequence) or isinstance(raw_courses, (str, bytes)):
770
+ raise A2LError("course metadata is unavailable; run calibration first")
771
+ selectors = {str(value).casefold() for value in only} if only is not None else None
772
+ courses: list[CourseRef] = []
773
+ for raw in raw_courses:
774
+ course = _course_ref(raw)
775
+ if not course.is_active:
776
+ continue
777
+ if term is not None and course.term != term:
778
+ continue
779
+ if selectors is not None and not (
780
+ str(course.org_unit_id).casefold() in selectors or course.code.casefold() in selectors
781
+ ):
782
+ continue
783
+ courses.append(course)
784
+ return sorted(
785
+ courses, key=lambda item: (item.term or "", item.code.casefold(), item.org_unit_id)
786
+ )
787
+
788
+
789
+ def _course_ref(raw: object) -> CourseRef:
790
+ if isinstance(raw, CourseRef):
791
+ return raw
792
+ if not isinstance(raw, Mapping):
793
+ raise A2LError("course metadata contains an invalid course")
794
+ try:
795
+ org_unit_id = raw["org_unit_id"]
796
+ code = raw["code"]
797
+ name = raw["name"]
798
+ term = raw.get("term")
799
+ is_active = raw["is_active"]
800
+ except KeyError as exc:
801
+ raise A2LError("course metadata contains an invalid course") from exc
802
+ if (
803
+ isinstance(org_unit_id, bool)
804
+ or not isinstance(org_unit_id, int)
805
+ or not isinstance(code, str)
806
+ or not code
807
+ or not isinstance(name, str)
808
+ or not name
809
+ or not isinstance(is_active, bool)
810
+ or term is not None
811
+ and not isinstance(term, str)
812
+ ):
813
+ raise A2LError("course metadata contains an invalid course")
814
+ return CourseRef(org_unit_id, code, name, term, is_active)
815
+
816
+
817
+ def _course_directory(vault: Vault, school: School, course: CourseRef) -> Path:
818
+ term_code = course.term or "unclassified"
819
+ if course.term is None:
820
+ term_label = "Unclassified"
821
+ else:
822
+ try:
823
+ term_label = school.term_label(course.term)
824
+ except ValueError:
825
+ term_label = f"Term {course.term}"
826
+ course_label = f"{course.code}_{term_code}" if course.code else f"Course-{course.org_unit_id}"
827
+ preferred = vault.root / paths.safe_name(term_label) / paths.safe_name(course_label)
828
+ return locations.course_directory(vault, school, course, preferred)
829
+
830
+
831
+ def _toc_path(client: IngestClient, course: CourseRef) -> str:
832
+ le = getattr(client, "le_version", None)
833
+ if not isinstance(le, str) or not le:
834
+ raise A2LError("LE API version is not calibrated")
835
+ return f"/d2l/api/le/{le}/{course.org_unit_id}/content/toc"
836
+
837
+
838
+ def _endpoint_path(client: IngestClient, course: CourseRef, endpoint: str) -> str:
839
+ le = getattr(client, "le_version", None)
840
+ if not isinstance(le, str) or not le:
841
+ raise A2LError("LE API version is not calibrated")
842
+ return f"/d2l/api/le/{le}/{course.org_unit_id}/{endpoint}"
843
+
844
+
845
+ def _fetch_one(client: IngestClient, path: str) -> tuple[object, bool, Exception | None]:
846
+ try:
847
+ return client.get_json(path), True, None
848
+ except SessionExpired:
849
+ raise
850
+ except Exception as exc:
851
+ return {}, False, exc
852
+
853
+
854
+ def _fetch_collection(
855
+ client: IngestClient, path: str
856
+ ) -> tuple[list[object], bool, Exception | None]:
857
+ values: list[object] = []
858
+ seen: set[str] = set()
859
+ current = path
860
+ for _ in range(_MAX_PAGES):
861
+ if current in seen:
862
+ return values, False, A2LError("metadata pagination repeated a page")
863
+ seen.add(current)
864
+ try:
865
+ payload = client.get_json(current)
866
+ except SessionExpired:
867
+ raise
868
+ except Exception as exc:
869
+ return values, False, exc
870
+ page, next_page, valid = _page_values(payload, current)
871
+ if not valid:
872
+ return values, False, A2LError("metadata endpoint returned an invalid page")
873
+ if any(not _collection_item_is_valid(item, current) for item in page):
874
+ return values, False, A2LError("metadata endpoint returned an invalid item")
875
+ values.extend(page)
876
+ if next_page is None:
877
+ return values, True, None
878
+ if not isinstance(next_page, str) or not next_page:
879
+ return values, False, A2LError("metadata pagination returned an invalid next route")
880
+ current = next_page
881
+ return values, False, A2LError("metadata pagination exceeded its limit")
882
+
883
+
884
+ def _page_values(payload: object, current: str) -> tuple[list[object], str | None, bool]:
885
+ if isinstance(payload, list):
886
+ return list(payload), None, True
887
+ if not isinstance(payload, dict):
888
+ return [], None, False
889
+ if isinstance(payload.get("Objects"), list):
890
+ next_page = payload.get("Next")
891
+ return list(payload["Objects"]), cast(str | None, next_page), True
892
+ if isinstance(payload.get("Items"), list):
893
+ paging = payload.get("PagingInfo")
894
+ if paging is None:
895
+ return list(payload["Items"]), cast(str | None, payload.get("Next")), True
896
+ if not isinstance(paging, dict) or not isinstance(paging.get("HasMoreItems"), bool):
897
+ return [], None, False
898
+ if not paging["HasMoreItems"]:
899
+ return list(payload["Items"]), None, True
900
+ bookmark = paging.get("Bookmark")
901
+ if not isinstance(bookmark, str) or not bookmark:
902
+ return [], None, False
903
+ # D2L's bookmark route is endpoint-specific; retain the path and add only the opaque
904
+ # bookmark value needed for the next request. It is never persisted in the vault.
905
+ next_route = _as_route_string(payload.get("Next")) or current
906
+ separator = "&" if "?" in next_route else "?"
907
+ return (
908
+ list(payload["Items"]),
909
+ f"{next_route}{separator}bookmark={bookmark}",
910
+ True,
911
+ )
912
+ return [], None, False
913
+
914
+
915
+ def _collection_item_is_valid(value: object, path: str) -> bool:
916
+ """Require stable IDs before a response can be considered complete for merge purposes."""
917
+ if not isinstance(value, dict):
918
+ return False
919
+ route = path.casefold()
920
+ if "dropbox/folders" in route or "/news/" in route:
921
+ identifier = value.get("Id")
922
+ if not isinstance(identifier, int) or isinstance(identifier, bool):
923
+ return False
924
+ if "dropbox/folders" in route:
925
+ return _assignment_attachments_are_valid(value)
926
+ return True
927
+ if "/quizzes/" in route:
928
+ identifier = value.get("QuizId")
929
+ return isinstance(identifier, int) and not isinstance(identifier, bool)
930
+ if "/grades/" in route:
931
+ identifier = value.get("GradeObjectIdentifier")
932
+ return (isinstance(identifier, int) and not isinstance(identifier, bool)) or (
933
+ isinstance(identifier, str) and bool(identifier)
934
+ )
935
+ if "/discussions/forums/" in route:
936
+ identifier = value.get("ForumId")
937
+ return isinstance(identifier, int) and not isinstance(identifier, bool)
938
+ return True
939
+
940
+
941
+ def _as_route_string(value: object) -> str:
942
+ return value if isinstance(value, str) and value else ""
943
+
944
+
945
+ def _topics_from_toc(
946
+ payload: object, *, course: CourseRef, school: School
947
+ ) -> tuple[list[TopicRecord], list[dict[str, object]], bool]:
948
+ if not isinstance(payload, dict) or not isinstance(payload.get("Modules"), list):
949
+ return [], [], False
950
+ records: list[TopicRecord] = []
951
+ valid = True
952
+
953
+ def walk(
954
+ modules: object, parent_titles: tuple[str, ...], parent_ids: tuple[int, ...]
955
+ ) -> list[dict[str, object]]:
956
+ nonlocal valid
957
+ if not isinstance(modules, list):
958
+ valid = False
959
+ return []
960
+ projected: list[dict[str, object]] = []
961
+ for module in modules:
962
+ if not isinstance(module, dict):
963
+ valid = False
964
+ continue
965
+ module_id = module.get("ModuleId")
966
+ title = module.get("Title")
967
+ children = module.get("Modules")
968
+ topics = module.get("Topics")
969
+ if (
970
+ isinstance(module_id, bool)
971
+ or not isinstance(module_id, int)
972
+ or not isinstance(title, str)
973
+ or not isinstance(children, list)
974
+ or not isinstance(topics, list)
975
+ ):
976
+ valid = False
977
+ continue
978
+ module_titles = parent_titles + (title,)
979
+ module_ids = parent_ids + (module_id,)
980
+ topic_projection: list[dict[str, object]] = []
981
+ for raw_topic in topics:
982
+ record = _topic_from_api(
983
+ raw_topic,
984
+ course=course,
985
+ school=school,
986
+ module_titles=module_titles,
987
+ module_ids=module_ids,
988
+ )
989
+ if record is None:
990
+ valid = False
991
+ continue
992
+ records.append(record)
993
+ topic_projection.append(_topic_projection(record))
994
+ projected.append(
995
+ {
996
+ "module_id": module_id,
997
+ "title": title,
998
+ "topics": topic_projection,
999
+ "modules": walk(children, module_titles, module_ids),
1000
+ }
1001
+ )
1002
+ return projected
1003
+
1004
+ modules = walk(payload["Modules"], (), ())
1005
+ records.sort(key=lambda record: record.source_key)
1006
+ return records, modules, valid
1007
+
1008
+
1009
+ def _topic_from_api(
1010
+ raw: object,
1011
+ *,
1012
+ course: CourseRef,
1013
+ school: School,
1014
+ module_titles: tuple[str, ...],
1015
+ module_ids: tuple[int, ...],
1016
+ ) -> TopicRecord | None:
1017
+ if not isinstance(raw, dict):
1018
+ return None
1019
+ topic_id = raw.get("TopicId")
1020
+ title = raw.get("Title")
1021
+ kind = raw.get("TypeIdentifier")
1022
+ raw_url = raw.get("Url")
1023
+ if (
1024
+ isinstance(topic_id, bool)
1025
+ or not isinstance(topic_id, int)
1026
+ or not isinstance(title, str)
1027
+ or not isinstance(kind, str)
1028
+ or raw_url is not None
1029
+ and not isinstance(raw_url, str)
1030
+ ):
1031
+ return None
1032
+ last_modified = raw.get("LastModifiedDate")
1033
+ if last_modified is not None and not isinstance(last_modified, str):
1034
+ last_modified = None
1035
+ is_broken = raw.get("IsBroken", False)
1036
+ if not isinstance(is_broken, bool):
1037
+ is_broken = False
1038
+ remote_size = raw.get("Size")
1039
+ if isinstance(remote_size, bool) or not isinstance(remote_size, int) or remote_size < 0:
1040
+ remote_size = None
1041
+
1042
+ key = f"{school.id}:{course.org_unit_id}:topic:{topic_id}"
1043
+ view_url = _view_url(school, course.org_unit_id, topic_id)
1044
+ external = bool(raw_url) and topic_is_excluded(kind, raw_url, school.topic_exclusion_policy())
1045
+ url_path = _first_party_path(raw_url, school.base_url)
1046
+ outline_url = _allowed_outline_url(raw_url, school)
1047
+ external_host = _safe_hostname(raw_url)
1048
+ if external or (raw_url and url_path is None and outline_url is None):
1049
+ return TopicRecord(
1050
+ source_key=key,
1051
+ source_id=str(topic_id),
1052
+ topic_id=topic_id,
1053
+ course_org_unit_id=course.org_unit_id,
1054
+ course_code=course.code,
1055
+ course_name=course.name,
1056
+ term=course.term,
1057
+ title=title,
1058
+ kind=kind,
1059
+ module_path=module_titles,
1060
+ module_ids=module_ids,
1061
+ view_url=view_url,
1062
+ outline_url=None,
1063
+ url_path=None,
1064
+ external_host=external_host,
1065
+ etag=_optional_text(raw.get("ETag")),
1066
+ last_modified=last_modified,
1067
+ is_broken=is_broken,
1068
+ availability="external_link",
1069
+ remote_size=remote_size,
1070
+ next_action="open the LEARN link manually",
1071
+ )
1072
+
1073
+ if is_broken:
1074
+ reason = "topic is marked broken in LEARN"
1075
+ elif url_path is None or kind.casefold() not in _DOWNLOADABLE_KINDS:
1076
+ reason = "fetch the topic explicitly"
1077
+ else:
1078
+ reason = f"a2l fetch {topic_id}"
1079
+ return TopicRecord(
1080
+ source_key=key,
1081
+ source_id=str(topic_id),
1082
+ topic_id=topic_id,
1083
+ course_org_unit_id=course.org_unit_id,
1084
+ course_code=course.code,
1085
+ course_name=course.name,
1086
+ term=course.term,
1087
+ title=title,
1088
+ kind=kind,
1089
+ module_path=module_titles,
1090
+ module_ids=module_ids,
1091
+ view_url=view_url,
1092
+ outline_url=outline_url,
1093
+ url_path=url_path,
1094
+ external_host=None,
1095
+ etag=_optional_text(raw.get("ETag")),
1096
+ last_modified=last_modified,
1097
+ is_broken=is_broken,
1098
+ availability="metadata_only",
1099
+ remote_size=remote_size,
1100
+ next_action=reason,
1101
+ )
1102
+
1103
+
1104
+ def _topic_projection(record: TopicRecord) -> dict[str, object]:
1105
+ return {
1106
+ "id": record.topic_id,
1107
+ "title": record.title,
1108
+ "kind": record.kind,
1109
+ "url_path": record.url_path,
1110
+ "external_host": record.external_host,
1111
+ "view_url": record.view_url,
1112
+ "last_modified": record.last_modified,
1113
+ "availability": record.availability,
1114
+ }
1115
+
1116
+
1117
+ def _merge_topic_records(
1118
+ existing: object,
1119
+ incoming: Sequence[TopicRecord],
1120
+ *,
1121
+ complete: bool,
1122
+ ) -> list[dict[str, object]]:
1123
+ old_rows = existing if isinstance(existing, list) else []
1124
+ by_key: dict[str, dict[str, object]] = {}
1125
+ for value in old_rows:
1126
+ if isinstance(value, dict) and isinstance(value.get("source_key"), str):
1127
+ by_key[value["source_key"]] = dict(value)
1128
+ incoming_keys: set[str] = set()
1129
+ for record in incoming:
1130
+ incoming_keys.add(record.source_key)
1131
+ prior = by_key.get(record.source_key, {})
1132
+ row = _topic_to_row(record)
1133
+ for field in ("source_path", "path", "sha256", "size", "stub_path", "source_sha256"):
1134
+ if row.get(field) is None and prior.get(field) is not None:
1135
+ row[field] = prior[field]
1136
+ if prior.get("availability") in {
1137
+ "source_only",
1138
+ "markdown_ready",
1139
+ "unsupported_format",
1140
+ "conversion_gap",
1141
+ "integrity_gap",
1142
+ "download_gap",
1143
+ }:
1144
+ row["availability"] = prior["availability"]
1145
+ row["next_action"] = prior.get("next_action", row["next_action"])
1146
+ row["missing_since"] = None
1147
+ row["withdrawn_at"] = None
1148
+ by_key[record.source_key] = row
1149
+
1150
+ if complete:
1151
+ now = _now()
1152
+ for key, row in by_key.items():
1153
+ if key in incoming_keys:
1154
+ continue
1155
+ if row.get("missing_since") is None:
1156
+ row["missing_since"] = now
1157
+ elif row.get("withdrawn_at") is None:
1158
+ row["withdrawn_at"] = now
1159
+ return sorted(by_key.values(), key=lambda row: str(row.get("source_key", "")))
1160
+
1161
+
1162
+ def _topic_to_row(record: TopicRecord) -> dict[str, object]:
1163
+ return {
1164
+ "source_key": record.source_key,
1165
+ "source_id": record.source_id,
1166
+ "topic_id": record.topic_id,
1167
+ "course_org_unit_id": record.course_org_unit_id,
1168
+ "course_code": record.course_code,
1169
+ "course_name": record.course_name,
1170
+ "term": record.term,
1171
+ "title": record.title,
1172
+ "kind": record.kind,
1173
+ "module_path": list(record.module_path),
1174
+ "module_ids": list(record.module_ids),
1175
+ "view_url": record.view_url,
1176
+ "outline_url": record.outline_url,
1177
+ "url_path": record.url_path,
1178
+ "external_host": record.external_host,
1179
+ "etag": record.etag,
1180
+ "last_modified": record.last_modified,
1181
+ "is_broken": record.is_broken,
1182
+ "availability": record.availability,
1183
+ "source_path": record.source_path,
1184
+ "path": record.path,
1185
+ "sha256": record.sha256,
1186
+ "source_sha256": record.sha256,
1187
+ "size": record.size,
1188
+ "stub_path": record.stub_path,
1189
+ "remote_size": record.remote_size,
1190
+ "next_action": record.next_action,
1191
+ "missing_since": record.missing_since,
1192
+ "withdrawn_at": record.withdrawn_at,
1193
+ }
1194
+
1195
+
1196
+ def _topic_from_row(row: object, *, course: CourseRef) -> TopicRecord:
1197
+ if not isinstance(row, dict):
1198
+ raise A2LError("content_map contains an invalid topic row")
1199
+ source_key = row.get("source_key")
1200
+ source_id = row.get("source_id")
1201
+ topic_id = row.get("topic_id")
1202
+ if (
1203
+ not isinstance(source_key, str)
1204
+ or not isinstance(source_id, str)
1205
+ or isinstance(topic_id, bool)
1206
+ or not isinstance(topic_id, int)
1207
+ ):
1208
+ raise A2LError("content_map contains an invalid topic identity")
1209
+ return TopicRecord(
1210
+ source_key=source_key,
1211
+ source_id=source_id,
1212
+ topic_id=topic_id,
1213
+ course_org_unit_id=course.org_unit_id,
1214
+ course_code=str(row.get("course_code", course.code)),
1215
+ course_name=str(row.get("course_name", course.name)),
1216
+ term=row.get("term") if isinstance(row.get("term"), str) else course.term,
1217
+ title=str(row.get("title", "untitled")),
1218
+ kind=str(row.get("kind", "File")),
1219
+ module_path=tuple(value for value in row.get("module_path", []) if isinstance(value, str)),
1220
+ module_ids=tuple(value for value in row.get("module_ids", []) if isinstance(value, int)),
1221
+ view_url=str(row.get("view_url", "")),
1222
+ outline_url=row.get("outline_url") if isinstance(row.get("outline_url"), str) else None,
1223
+ url_path=row.get("url_path") if isinstance(row.get("url_path"), str) else None,
1224
+ external_host=row.get("external_host")
1225
+ if isinstance(row.get("external_host"), str)
1226
+ else None,
1227
+ etag=row.get("etag") if isinstance(row.get("etag"), str) else None,
1228
+ last_modified=row.get("last_modified")
1229
+ if isinstance(row.get("last_modified"), str)
1230
+ else None,
1231
+ is_broken=bool(row.get("is_broken", False)),
1232
+ availability=str(row.get("availability", "metadata_only")),
1233
+ source_path=row.get("source_path") if isinstance(row.get("source_path"), str) else None,
1234
+ path=row.get("path") if isinstance(row.get("path"), str) else None,
1235
+ sha256=row.get("sha256") if isinstance(row.get("sha256"), str) else None,
1236
+ size=row.get("size") if isinstance(row.get("size"), int) else None,
1237
+ stub_path=row.get("stub_path") if isinstance(row.get("stub_path"), str) else None,
1238
+ remote_size=row.get("remote_size") if isinstance(row.get("remote_size"), int) else None,
1239
+ next_action=str(row.get("next_action", f"a2l fetch {source_id}")),
1240
+ missing_since=row.get("missing_since")
1241
+ if isinstance(row.get("missing_since"), str)
1242
+ else None,
1243
+ withdrawn_at=row.get("withdrawn_at") if isinstance(row.get("withdrawn_at"), str) else None,
1244
+ )
1245
+
1246
+
1247
+ def _assignment_attachment_topics(
1248
+ values: Sequence[object], *, course: CourseRef, school: School
1249
+ ) -> list[TopicRecord]:
1250
+ """Project only allowlisted first-party Dropbox attachments into the normal file pipeline."""
1251
+
1252
+ records: list[TopicRecord] = []
1253
+ for assignment in values:
1254
+ if (
1255
+ not isinstance(assignment, dict)
1256
+ or not isinstance(assignment.get("Id"), int)
1257
+ or isinstance(assignment.get("Id"), bool)
1258
+ ):
1259
+ continue
1260
+ assignment_id = assignment["Id"]
1261
+ assignment_title = _safe_text(assignment.get("Name")) or f"Assignment {assignment_id}"
1262
+ for attachment in _attachment_values(assignment):
1263
+ if not isinstance(attachment, dict):
1264
+ continue
1265
+ raw_id = attachment.get("Id", attachment.get("FileId"))
1266
+ if raw_id is None:
1267
+ continue
1268
+ attachment_id = str(raw_id)
1269
+ source_id = f"{assignment_id}-{attachment_id}".replace(":", "_")
1270
+ raw_url = _first_string(attachment, "Url", "URL", "Href", "DownloadUrl")
1271
+ url_path = _first_party_path(raw_url, school.base_url)
1272
+ if url_path is None:
1273
+ continue
1274
+ title = (
1275
+ _first_string(attachment, "FileName", "Name", "Title")
1276
+ or f"attachment-{attachment_id}"
1277
+ )
1278
+ remote_size = attachment.get("Size")
1279
+ if isinstance(remote_size, bool) or not isinstance(remote_size, int) or remote_size < 0:
1280
+ remote_size = None
1281
+ topic_id = (
1282
+ raw_id
1283
+ if isinstance(raw_id, int) and not isinstance(raw_id, bool)
1284
+ else _stable_numeric_id(source_id)
1285
+ )
1286
+ records.append(
1287
+ TopicRecord(
1288
+ source_key=f"{school.id}:{course.org_unit_id}:attachment:{source_id}",
1289
+ source_id=source_id,
1290
+ topic_id=topic_id,
1291
+ course_org_unit_id=course.org_unit_id,
1292
+ course_code=course.code,
1293
+ course_name=course.name,
1294
+ term=course.term,
1295
+ title=title,
1296
+ kind="File",
1297
+ module_path=("Assignments", assignment_title),
1298
+ module_ids=(),
1299
+ view_url=_view_url(school, course.org_unit_id, topic_id),
1300
+ outline_url=None,
1301
+ url_path=url_path,
1302
+ external_host=None,
1303
+ etag=_first_string(attachment, "ETag", "Etag"),
1304
+ last_modified=_first_string(attachment, "LastModifiedDate", "LastModified"),
1305
+ is_broken=False,
1306
+ remote_size=remote_size,
1307
+ next_action=f"a2l fetch {source_id}",
1308
+ )
1309
+ )
1310
+ return sorted(records, key=lambda item: item.source_key)
1311
+
1312
+
1313
+ def _attachment_values(assignment: Mapping[str, object]) -> list[object]:
1314
+ raw = assignment.get("Attachments", assignment.get("attachments", []))
1315
+ if isinstance(raw, dict):
1316
+ raw = raw.get("Items", raw.get("Objects", []))
1317
+ return list(raw) if isinstance(raw, list) else []
1318
+
1319
+
1320
+ def _assignment_attachments_are_valid(assignment: Mapping[str, object]) -> bool:
1321
+ """Reject malformed attachment containers before they can mark old files missing."""
1322
+ if "Attachments" not in assignment and "attachments" not in assignment:
1323
+ return True
1324
+ raw = assignment.get("Attachments", assignment.get("attachments"))
1325
+ if isinstance(raw, dict):
1326
+ raw = raw.get("Items", raw.get("Objects"))
1327
+ if not isinstance(raw, list):
1328
+ return False
1329
+ for attachment in raw:
1330
+ if not isinstance(attachment, dict):
1331
+ return False
1332
+ identifier = attachment.get("Id", attachment.get("FileId"))
1333
+ if isinstance(identifier, bool) or not isinstance(identifier, (int, str)):
1334
+ return False
1335
+ if isinstance(identifier, str) and not identifier:
1336
+ return False
1337
+ return True
1338
+
1339
+
1340
+ def _first_string(value: Mapping[str, object], *keys: str) -> str | None:
1341
+ for key in keys:
1342
+ candidate = value.get(key)
1343
+ if isinstance(candidate, str) and candidate:
1344
+ return candidate
1345
+ return None
1346
+
1347
+
1348
+ def _stable_numeric_id(value: str) -> int:
1349
+ return int.from_bytes(sha256(value.encode("utf-8")).digest()[:4], "big") & 0x7FFFFFFF
1350
+
1351
+
1352
+ def _materialize_assignments(
1353
+ values: Sequence[object],
1354
+ *,
1355
+ course_dir: Path,
1356
+ vault: Vault,
1357
+ school: School,
1358
+ course: CourseRef,
1359
+ directories: Mapping[str, Path],
1360
+ ) -> dict[str, dict[str, object]]:
1361
+ """Sanitize Dropbox RichText and persist a provenance-backed source/twin pair."""
1362
+
1363
+ artifacts: dict[str, dict[str, object]] = {}
1364
+ assignments = sorted(
1365
+ (value for value in values if isinstance(value, dict) and isinstance(value.get("Id"), int)),
1366
+ key=lambda value: int(value["Id"]),
1367
+ )
1368
+ for assignment in assignments:
1369
+ assignment_id = int(assignment["Id"])
1370
+ key = f"{school.id}:{course.org_unit_id}:dropbox:{assignment_id}"
1371
+ transactions.recover_generated(vault, key)
1372
+ richtext = _assignment_richtext(assignment)
1373
+ if richtext is None:
1374
+ continue
1375
+ raw_html, raw_text = richtext
1376
+ canonical_html = _sanitize_richtext(raw_html or raw_text or "", school.base_url)
1377
+ if not canonical_html.strip():
1378
+ continue
1379
+ html_bytes = canonical_html.encode("utf-8")
1380
+ source_hash = sha256(html_bytes).hexdigest()
1381
+ prior = vault.entry(key)
1382
+ title = _safe_text(assignment.get("Name")) or f"Assignment {assignment_id}"
1383
+ if prior is not None:
1384
+ source_destination = vault.materialized(prior)
1385
+ else:
1386
+ assignment_directory = directories[str(assignment_id)]
1387
+ source_destination = paths.unique_path(
1388
+ assignment_directory / "instructions.html", reserved=vault.claimed_paths()
1389
+ )
1390
+
1391
+ prior_artifact = prior.derived.get("markdown") if prior is not None else None
1392
+ if prior_artifact is not None:
1393
+ markdown_destination = vault.materialized(
1394
+ ManifestEntry(
1395
+ path=prior_artifact.path,
1396
+ sha256=prior_artifact.sha256,
1397
+ source_id="derived",
1398
+ etag=None,
1399
+ last_modified=None,
1400
+ size=0,
1401
+ fetched_at=_now(),
1402
+ )
1403
+ )
1404
+ else:
1405
+ markdown_destination = source_destination.with_suffix(".md")
1406
+ markdown_destination = vault.derived_destination(key, markdown_destination)
1407
+
1408
+ source_changed = prior is not None and prior.sha256 != source_hash
1409
+ twin_modified = bool(
1410
+ prior_artifact is not None
1411
+ and paths.long_path(markdown_destination).is_file()
1412
+ and _hash_file(markdown_destination)[0] != prior_artifact.sha256
1413
+ )
1414
+ markdown_bytes = _richtext_markdown(title, canonical_html).encode("utf-8")
1415
+ derived = DerivedArtifact(
1416
+ path=paths.rel_posix(markdown_destination, vault.root),
1417
+ sha256=sha256(markdown_bytes).hexdigest(),
1418
+ source_sha256=source_hash,
1419
+ tool="richtext-sanitizer",
1420
+ tool_version="1",
1421
+ created_at=_now(),
1422
+ )
1423
+ entry = ManifestEntry(
1424
+ path=paths.rel_posix(source_destination, vault.root),
1425
+ sha256=source_hash,
1426
+ source_id=str(assignment_id),
1427
+ etag=None,
1428
+ last_modified=_optional_text(assignment.get("LastModifiedDate")),
1429
+ size=len(html_bytes),
1430
+ fetched_at=_now(),
1431
+ derived={"markdown": derived},
1432
+ )
1433
+ transactions.install_generated(
1434
+ vault,
1435
+ key,
1436
+ entry,
1437
+ html_bytes,
1438
+ {"markdown": markdown_bytes},
1439
+ preserve=source_changed or twin_modified,
1440
+ )
1441
+
1442
+ _write_assignment_readme(
1443
+ source_destination.parent,
1444
+ title=title,
1445
+ entry=entry,
1446
+ attachments=_assignment_attachment_display(assignment, school),
1447
+ root=vault.root,
1448
+ )
1449
+ artifacts[str(assignment_id)] = {
1450
+ "instructions_html": entry.path,
1451
+ "instructions_md": derived.path,
1452
+ "instructions_sha256": source_hash,
1453
+ }
1454
+ return artifacts
1455
+
1456
+
1457
+ def _assignment_richtext(assignment: Mapping[str, object]) -> tuple[str | None, str | None] | None:
1458
+ for field in ("CustomInstructions", "Description", "Instructions", "RichText", "Body"):
1459
+ value = assignment.get(field)
1460
+ if isinstance(value, str):
1461
+ return value, None
1462
+ if isinstance(value, dict):
1463
+ html_value = _first_string(value, "Html", "HTML", "html")
1464
+ text_value = _first_string(value, "Text", "text")
1465
+ if html_value is not None or text_value is not None:
1466
+ return html_value, text_value
1467
+ return None
1468
+
1469
+
1470
+ def _assignment_attachment_display(
1471
+ assignment: Mapping[str, object], school: School
1472
+ ) -> list[dict[str, str]]:
1473
+ rows: list[dict[str, str]] = []
1474
+ assignment_id = assignment.get("Id")
1475
+ for attachment in _attachment_values(assignment):
1476
+ if not isinstance(attachment, dict):
1477
+ continue
1478
+ raw_id = attachment.get("Id", attachment.get("FileId"))
1479
+ if raw_id is None:
1480
+ continue
1481
+ attachment_id = str(raw_id)
1482
+ name = (
1483
+ _first_string(attachment, "FileName", "Name", "Title") or f"attachment-{attachment_id}"
1484
+ )
1485
+ raw_url = _first_string(attachment, "Url", "URL", "Href", "DownloadUrl")
1486
+ path = _first_party_path(raw_url, school.base_url)
1487
+ if path is not None:
1488
+ rows.append(
1489
+ {
1490
+ "name": name,
1491
+ "action": f"a2l fetch {assignment_id}-{attachment_id}",
1492
+ }
1493
+ )
1494
+ else:
1495
+ host = _safe_hostname(raw_url) or "unknown-host"
1496
+ rows.append({"name": name, "action": f"external link · {host}"})
1497
+ return rows
1498
+
1499
+
1500
+ def _write_assignment_readme(
1501
+ directory: Path,
1502
+ *,
1503
+ title: str,
1504
+ entry: ManifestEntry,
1505
+ attachments: Sequence[Mapping[str, str]],
1506
+ root: Path | None = None,
1507
+ ) -> None:
1508
+ markdown_path = entry.derived["markdown"].path
1509
+ lines = [
1510
+ f"# {title}",
1511
+ "",
1512
+ f"- Instructions source: `{PurePosixPath(entry.path).name}`",
1513
+ f"- Instructions twin: `{PurePosixPath(markdown_path).name}`",
1514
+ f"- Source SHA-256: `{entry.sha256}`",
1515
+ "",
1516
+ "## Attachments",
1517
+ "",
1518
+ ]
1519
+ if attachments:
1520
+ lines.extend(f"- {row['name']} — {row['action']}" for row in attachments)
1521
+ else:
1522
+ lines.append("- None recorded.")
1523
+ lines.append("")
1524
+ paths.atomic_write_text(directory / "README.md", "\n".join(lines), root=root)
1525
+
1526
+
1527
+ def _materialize_submission_only_readmes(
1528
+ values: Sequence[object],
1529
+ *,
1530
+ artifacts: Mapping[str, Mapping[str, object]],
1531
+ course_dir: Path,
1532
+ directories: Mapping[str, Path],
1533
+ topics: Sequence[Mapping[str, object]],
1534
+ root: Path | None = None,
1535
+ ) -> None:
1536
+ """Give a Dropbox folder without RichText a generated navigation hub.
1537
+
1538
+ This is deliberately a metadata-only projection: it creates no fetch route and does not copy
1539
+ a prompt into the README. Exact display-title matches are only navigation cross-links; topic
1540
+ resolution itself remains stable-ID/manifest based in :mod:`agent2learn.index`.
1541
+ """
1542
+ for assignment in sorted(
1543
+ (row for row in values if isinstance(row, Mapping) and isinstance(row.get("Id"), int)),
1544
+ key=lambda row: int(row["Id"]),
1545
+ ):
1546
+ assignment_id = str(assignment["Id"])
1547
+ if assignment_id in artifacts:
1548
+ continue
1549
+ title = _safe_text(assignment.get("Name")) or f"Assignment {assignment_id}"
1550
+ directory = directories[assignment_id]
1551
+ paths.ensure_dir(directory, root=root)
1552
+ links: list[tuple[str, str]] = []
1553
+ for topic in topics:
1554
+ if str(topic.get("title", "")).casefold() != title.casefold():
1555
+ continue
1556
+ target = topic.get("path") or topic.get("source_path") or topic.get("stub_path")
1557
+ source_id = topic.get("source_id")
1558
+ if isinstance(target, str) and isinstance(source_id, str):
1559
+ links.append((source_id, _course_relative_link(target, course_dir)))
1560
+ course_index.write_submission_readme(directory, title=title, content_links=links, root=root)
1561
+
1562
+
1563
+ def _materialize_external_stubs(
1564
+ rows: list[dict[str, object]],
1565
+ *,
1566
+ course_dir: Path,
1567
+ vault: Vault,
1568
+ school: School,
1569
+ course: CourseRef,
1570
+ ) -> list[dict[str, object]]:
1571
+ reserved: set[str] = set()
1572
+ for row in sorted(rows, key=lambda value: str(value.get("source_key", ""))):
1573
+ if row.get("availability") != "external_link":
1574
+ continue
1575
+ prior = row.get("stub_path")
1576
+ if isinstance(prior, str):
1577
+ stub = _vault_relative_path(vault, prior)
1578
+ else:
1579
+ record = _topic_from_row(row, course=course)
1580
+ destination = _content_directory(course_dir, record.module_path) / (
1581
+ f"{paths.safe_name(record.title)}.url.txt"
1582
+ )
1583
+ stub = _unique_reserved(destination, reserved)
1584
+ row["stub_path"] = paths.rel_posix(stub, vault.root)
1585
+ paths.ensure_dir(stub.parent, root=vault.root)
1586
+ if paths.is_link(stub):
1587
+ raise A2LError("external-topic stub path contains a symlink or junction")
1588
+ if not paths.long_path(stub).is_file():
1589
+ record = _topic_from_row(row, course=course)
1590
+ text = (
1591
+ "Agent2Learn external-topic stub\n"
1592
+ f"topic: {record.title}\n"
1593
+ f"view in LEARN: {record.view_url}\n"
1594
+ f"destination host: {record.external_host or 'unknown-host'}\n"
1595
+ )
1596
+ # Keep the ordinary path as the value passed between layers; long_path belongs only
1597
+ # at filesystem boundaries, and atomic_write_text applies it internally.
1598
+ paths.atomic_write_text(stub, text, root=vault.root)
1599
+ return rows
1600
+
1601
+
1602
+ def _plan_file_paths(
1603
+ rows: Sequence[TopicRecord], *, course_dir: Path, vault: Vault, scope: str
1604
+ ) -> list[TopicRecord]:
1605
+ del scope
1606
+ manifest = Vault(vault.root).manifest()
1607
+ installed = _installed_pending_paths(
1608
+ vault, course_dir, {row.source_key for row in rows if row.source_key not in manifest}
1609
+ )
1610
+ reserved_by_folder: dict[str, set[str]] = defaultdict(set)
1611
+ for reserved_entry in manifest.values():
1612
+ for relative in (
1613
+ reserved_entry.path,
1614
+ *(artifact.path for artifact in reserved_entry.derived.values()),
1615
+ ):
1616
+ occupied = vault.root / PurePosixPath(relative)
1617
+ reserved_by_folder[occupied.parent.as_posix()].add(occupied.name.casefold())
1618
+ planned: list[TopicRecord] = []
1619
+ for topic in sorted(rows, key=lambda item: item.source_key):
1620
+ if topic.availability == "external_link":
1621
+ planned.append(topic)
1622
+ continue
1623
+ entry = manifest.get(topic.source_key)
1624
+ if entry is not None:
1625
+ planned.append(
1626
+ replace(topic, source_path=entry.path, path=_derived_path(manifest, entry))
1627
+ )
1628
+ continue
1629
+ if topic.url_path is None or topic.kind.casefold() not in _DOWNLOADABLE_KINDS:
1630
+ planned.append(topic)
1631
+ continue
1632
+ if topic.source_key in installed:
1633
+ planned.append(
1634
+ replace(topic, source_path=paths.rel_posix(installed[topic.source_key], vault.root))
1635
+ )
1636
+ continue
1637
+ destination = _content_directory(course_dir, topic.module_path) / _topic_filename(topic)
1638
+ folder_key = destination.parent.as_posix()
1639
+ candidate = _unique_reserved(destination, reserved_by_folder[folder_key])
1640
+ planned.append(replace(topic, source_path=paths.rel_posix(candidate, vault.root)))
1641
+ return planned
1642
+
1643
+
1644
+ def _priority_rows(
1645
+ rows: Sequence[TopicRecord], *, scope: Literal["all", "priority"], budget: int
1646
+ ) -> list[TopicRecord]:
1647
+ if scope == "all":
1648
+ return list(rows)
1649
+ ranked = sorted(
1650
+ rows,
1651
+ key=lambda topic: (
1652
+ 0 if _is_priority(topic) else 1,
1653
+ 0 if topic.last_modified is not None else 1,
1654
+ -(int(_timestamp_sort(topic.last_modified)) if topic.last_modified else 0),
1655
+ topic.source_key,
1656
+ ),
1657
+ )
1658
+ selected: list[TopicRecord] = []
1659
+ total = 0
1660
+ for topic in ranked:
1661
+ # An unknown size cannot be charged against a hard byte budget without inventing a
1662
+ # bound. Leave it for the full plan (which still has a per-file ceiling) or explicit
1663
+ # ``a2l fetch``; priority must remain provably byte-bounded.
1664
+ if topic.remote_size is None:
1665
+ continue
1666
+ if total + topic.remote_size > budget:
1667
+ continue
1668
+ total += topic.remote_size
1669
+ selected.append(topic)
1670
+ return selected
1671
+
1672
+
1673
+ def _is_priority(topic: TopicRecord) -> bool:
1674
+ value = f"{topic.title} {'/'.join(topic.module_path)}".casefold()
1675
+ return "assignment" in value or "outline" in value or "syllabus" in value
1676
+
1677
+
1678
+ def _ingest_one_topic(
1679
+ client: IngestClient,
1680
+ vault: Vault,
1681
+ school: School,
1682
+ course_dir: Path,
1683
+ topic: TopicRecord,
1684
+ *,
1685
+ max_bytes: int | None = api.DEFAULT_MAX_BYTES,
1686
+ ) -> Literal["downloaded", "skipped"]:
1687
+ if topic.url_path is None:
1688
+ raise DownloadError("topic has no first-party download route")
1689
+ key = topic.source_key
1690
+ transactions.recover_generated(vault, key)
1691
+ manifest = vault.manifest()
1692
+ prior = manifest.get(key)
1693
+ destination = _destination_for_topic(vault, course_dir, topic, prior)
1694
+ paths.ensure_dir(destination.parent, root=vault.root)
1695
+ pending = _find_pending_install(vault, destination, topic)
1696
+ if pending is not None:
1697
+ persisted = Vault(vault.root).entry(key)
1698
+ retried = _retry_pending_install(
1699
+ vault, course_dir, school, topic, destination, persisted, pending
1700
+ )
1701
+ prior = vault.entry(key)
1702
+ if retried is not None:
1703
+ return retried
1704
+ if prior is not None and _unchanged_local(prior, topic, vault):
1705
+ _mark_topic_source_only(vault, course_dir, topic, school)
1706
+ return "skipped"
1707
+
1708
+ # Revalidate immediately before reserving the download sibling; the network and manifest
1709
+ # checks above must not leave a race window where a replaced parent redirects the part.
1710
+ if paths.has_link_component(destination.parent, root=vault.root):
1711
+ raise A2LError("download parent contains a link component")
1712
+ fd, raw_temp = tempfile.mkstemp(
1713
+ prefix=f".{destination.name}.",
1714
+ suffix=".part",
1715
+ dir=os.fspath(paths.long_path(destination.parent)),
1716
+ )
1717
+ os.close(fd)
1718
+ temporary = paths.plain_path(Path(raw_temp))
1719
+ install_attempted = False
1720
+ installed = False
1721
+ try:
1722
+ # ``mkstemp`` is intentionally used to reserve the sibling before the network call. The
1723
+ # reservation is a separate filesystem boundary, so revalidate it before any download or
1724
+ # later writer can use a parent that was swapped to a link in the meantime.
1725
+ if paths.has_link_component(temporary, root=vault.root):
1726
+ raise A2LError("download temporary path contains a link component")
1727
+ conditional = prior if prior is not None and _source_bytes_match(prior, vault) else None
1728
+ result = _download_with_candidates(
1729
+ client,
1730
+ school,
1731
+ topic,
1732
+ temporary,
1733
+ prior=conditional,
1734
+ max_bytes=max_bytes,
1735
+ root=vault.root,
1736
+ )
1737
+ if result.not_modified:
1738
+ if prior is not None and _source_bytes_match(prior, vault):
1739
+ _mark_topic_source_only(vault, course_dir, topic, school)
1740
+ return "skipped"
1741
+ result = _download_with_candidates(
1742
+ client, school, topic, temporary, prior=None, max_bytes=max_bytes, root=vault.root
1743
+ )
1744
+ if result.not_modified:
1745
+ raise DownloadError("server returned 304 without a verified local source")
1746
+ if result.temp is None or not paths.long_path(result.temp).is_file():
1747
+ raise DownloadError("download did not produce a source file")
1748
+ actual_hash, actual_size = _hash_file(result.temp)
1749
+ if actual_size <= 0 or result.sha256 != actual_hash or result.size != actual_size:
1750
+ raise DownloadError("download integrity validation failed")
1751
+ pending = _PendingInstall(
1752
+ marker=_pending_marker_path(temporary),
1753
+ part=temporary,
1754
+ source_key=key,
1755
+ destination=paths.rel_posix(destination, vault.root),
1756
+ sha256=actual_hash,
1757
+ size=actual_size,
1758
+ etag=result.etag or topic.etag or (prior.etag if prior else None),
1759
+ last_modified=result.last_modified
1760
+ or topic.last_modified
1761
+ or (prior.last_modified if prior else None),
1762
+ prior_sha256=(
1763
+ prior.sha256 if prior is not None and actual_hash != prior.sha256 else None
1764
+ ),
1765
+ revision_preserved=prior is None or actual_hash == prior.sha256,
1766
+ fetched_at=_now(),
1767
+ )
1768
+ _write_pending_install(pending, root=vault.root)
1769
+ install_attempted = True
1770
+ if pending.prior_sha256 is not None:
1771
+ if prior is None:
1772
+ raise A2LError("pending source revision has no current manifest entry")
1773
+ preserved = vault.preserve_revision(key, changed_at=clock.now())
1774
+ if preserved is None and paths.long_path(vault.materialized(prior)).exists():
1775
+ raise A2LError("current source could not be preserved; refusing replacement")
1776
+ pending = replace(pending, revision_preserved=True)
1777
+ _write_pending_install(pending, root=vault.root)
1778
+
1779
+ paths.atomic_install_temp(destination, temporary, root=vault.root)
1780
+ installed = True
1781
+ entry = _manifest_entry_for_install(
1782
+ vault,
1783
+ topic,
1784
+ destination=destination,
1785
+ prior=prior,
1786
+ sha256=actual_hash,
1787
+ size=actual_size,
1788
+ etag=pending.etag,
1789
+ last_modified=pending.last_modified,
1790
+ fetched_at=pending.fetched_at,
1791
+ )
1792
+ vault.mark(key, entry)
1793
+ vault.save_manifest()
1794
+ _mark_topic_source_only(vault, course_dir, topic, school)
1795
+ _remove_pending_install(pending)
1796
+ return "downloaded"
1797
+ except BaseException:
1798
+ # A failed transfer has an incomplete part and may be restarted from byte zero. A
1799
+ # completed part whose *install* failed is different: paths.atomic_install_temp owns the
1800
+ # deliberate retention guarantee, so do not clean it here once installation was attempted.
1801
+ if not install_attempted:
1802
+ _remove_quietly(temporary)
1803
+ raise
1804
+ finally:
1805
+ if installed:
1806
+ _remove_quietly(temporary)
1807
+
1808
+
1809
+ def _pending_marker_path(part: Path) -> Path:
1810
+ return part.with_name(part.name + _PENDING_MARKER_SUFFIX)
1811
+
1812
+
1813
+ def _write_pending_install(pending: _PendingInstall, *, root: Path | None = None) -> None:
1814
+ payload = {
1815
+ "version": 2,
1816
+ "fetched_at": pending.fetched_at or _now(),
1817
+ "source_key": pending.source_key,
1818
+ "destination": pending.destination,
1819
+ "sha256": pending.sha256,
1820
+ "size": pending.size,
1821
+ "etag": pending.etag,
1822
+ "last_modified": pending.last_modified,
1823
+ "prior_sha256": pending.prior_sha256,
1824
+ "revision_preserved": pending.revision_preserved,
1825
+ }
1826
+ paths.atomic_write_text(
1827
+ pending.marker,
1828
+ json.dumps(payload, ensure_ascii=False, sort_keys=True, separators=(",", ":")) + "\n",
1829
+ root=root,
1830
+ )
1831
+
1832
+
1833
+ def _read_pending_install(marker: Path, part: Path) -> _PendingInstall | None:
1834
+ if not _is_safe_local_file(marker):
1835
+ return None
1836
+ try:
1837
+ with open(os.fspath(paths.long_path(marker)), encoding="utf-8", newline="") as handle:
1838
+ raw: Any = json.load(handle)
1839
+ except (OSError, json.JSONDecodeError, UnicodeError):
1840
+ return None
1841
+ if not isinstance(raw, dict):
1842
+ return None
1843
+ version = raw.get("version")
1844
+ if isinstance(version, bool) or not isinstance(version, int) or version not in {1, 2}:
1845
+ return None
1846
+ expected_keys = _PENDING_INSTALL_KEYS | ({"fetched_at"} if version == 2 else set())
1847
+ if set(raw) != expected_keys:
1848
+ return None
1849
+ fetched_at = raw.get("fetched_at")
1850
+ if version == 2:
1851
+ if not isinstance(fetched_at, str):
1852
+ return None
1853
+ try:
1854
+ parse_api_timestamp(fetched_at)
1855
+ except (TypeError, ValueError):
1856
+ return None
1857
+ source_key = raw.get("source_key")
1858
+ destination = raw.get("destination")
1859
+ sha256_value = raw.get("sha256")
1860
+ size = raw.get("size")
1861
+ etag = raw.get("etag")
1862
+ last_modified = raw.get("last_modified")
1863
+ prior_sha256 = raw.get("prior_sha256")
1864
+ revision_preserved = raw.get("revision_preserved")
1865
+ if (
1866
+ not isinstance(source_key, str)
1867
+ or not source_key
1868
+ or not isinstance(destination, str)
1869
+ or not destination
1870
+ or not isinstance(sha256_value, str)
1871
+ or re.fullmatch(r"[0-9a-f]{64}", sha256_value) is None
1872
+ or isinstance(size, bool)
1873
+ or not isinstance(size, int)
1874
+ or size <= 0
1875
+ or (etag is not None and not isinstance(etag, str))
1876
+ or (last_modified is not None and not isinstance(last_modified, str))
1877
+ or (
1878
+ prior_sha256 is not None
1879
+ and (
1880
+ not isinstance(prior_sha256, str)
1881
+ or re.fullmatch(r"[0-9a-f]{64}", prior_sha256) is None
1882
+ )
1883
+ )
1884
+ or not isinstance(revision_preserved, bool)
1885
+ ):
1886
+ return None
1887
+ return _PendingInstall(
1888
+ marker=marker,
1889
+ part=part,
1890
+ source_key=source_key,
1891
+ destination=destination,
1892
+ sha256=sha256_value,
1893
+ size=size,
1894
+ etag=etag,
1895
+ last_modified=last_modified,
1896
+ prior_sha256=prior_sha256,
1897
+ revision_preserved=revision_preserved,
1898
+ fetched_at=fetched_at,
1899
+ )
1900
+
1901
+
1902
+ def _is_safe_local_file(path: Path) -> bool:
1903
+ try:
1904
+ file_stat = os.lstat(os.fspath(paths.long_path(path)))
1905
+ except OSError:
1906
+ return False
1907
+ return (
1908
+ not paths.is_link(path)
1909
+ and stat.S_ISREG(file_stat.st_mode)
1910
+ and getattr(file_stat, "st_nlink", 1) == 1
1911
+ )
1912
+
1913
+
1914
+ def _remove_pending_install(pending: _PendingInstall) -> None:
1915
+ # The marker is removed first so an orphaned part is never treated as a validated download.
1916
+ _remove_pending_paths(pending.marker, pending.part)
1917
+
1918
+
1919
+ def _remove_pending_paths(marker: Path, part: Path) -> None:
1920
+ paths.remove_tree(marker, ignore_errors=True)
1921
+ paths.remove_tree(part, ignore_errors=True)
1922
+
1923
+
1924
+ def _pending_matches_topic(pending: _PendingInstall, topic: TopicRecord) -> bool:
1925
+ # A stable key and matching byte count do not prove that a remote file with no validator is
1926
+ # unchanged. Revalidate it over the network rather than replaying a potentially stale part.
1927
+ if topic.etag is None and topic.last_modified is None:
1928
+ return False
1929
+ if topic.etag is not None and pending.etag != topic.etag:
1930
+ return False
1931
+ if topic.last_modified is not None and pending.last_modified != topic.last_modified:
1932
+ return False
1933
+ return topic.remote_size is None or pending.size == topic.remote_size
1934
+
1935
+
1936
+ def _installed_pending_matches(pending: _PendingInstall, destination: Path) -> bool:
1937
+ return (
1938
+ pending.revision_preserved
1939
+ and not paths.collides(pending.part)
1940
+ and _is_safe_local_file(destination)
1941
+ and _hash_file(destination) == (pending.sha256, pending.size)
1942
+ )
1943
+
1944
+
1945
+ def _installed_pending_paths(vault: Vault, course_dir: Path, wanted: set[str]) -> dict[str, Path]:
1946
+ if not wanted:
1947
+ return {}
1948
+ content = course_dir / _COURSE_CONTENT
1949
+ if paths.has_link_component(content, root=vault.root):
1950
+ raise A2LError("pending download directory contains a link component")
1951
+ if not paths.long_path(content).is_dir():
1952
+ return {}
1953
+ result: dict[str, Path] = {}
1954
+ for marker in paths.walk(content):
1955
+ if not marker.name.startswith(".") or not marker.name.endswith(_PENDING_INSTALL_SUFFIX):
1956
+ continue
1957
+ part = marker.with_name(marker.name[: -len(_PENDING_MARKER_SUFFIX)])
1958
+ pending = _read_pending_install(marker, part)
1959
+ if pending is None or pending.source_key not in wanted or pending.prior_sha256 is not None:
1960
+ continue
1961
+ try:
1962
+ destination = _vault_relative_path(vault, pending.destination)
1963
+ except (A2LError, ValueError):
1964
+ continue
1965
+ if destination.parent != marker.parent or not destination.is_relative_to(content):
1966
+ continue
1967
+ if not marker.name.startswith(f".{destination.name}."):
1968
+ continue
1969
+ if not _installed_pending_matches(pending, destination):
1970
+ continue
1971
+ if pending.source_key in result and result[pending.source_key] != destination:
1972
+ raise A2LError("multiple installed revisions require recovery for one source")
1973
+ result[pending.source_key] = destination
1974
+ return result
1975
+
1976
+
1977
+ def _find_pending_install(
1978
+ vault: Vault, destination: Path, topic: TopicRecord
1979
+ ) -> _PendingInstall | None:
1980
+ """Find a validated install left by a prior sync, rejecting stale or untrusted markers."""
1981
+ expected_destination = paths.rel_posix(destination, vault.root)
1982
+ prefix = f".{destination.name}."
1983
+ try:
1984
+ with os.scandir(os.fspath(paths.long_path(destination.parent))) as iterator:
1985
+ candidates = sorted(entry.name for entry in iterator)
1986
+ except FileNotFoundError:
1987
+ return None
1988
+ except OSError as exc:
1989
+ raise A2LError("pending download state is unreadable") from exc
1990
+
1991
+ for name in candidates:
1992
+ if not name.startswith(prefix) or not name.endswith(_PENDING_INSTALL_SUFFIX):
1993
+ continue
1994
+ marker = destination.parent / name
1995
+ part = destination.parent / name[: -len(_PENDING_MARKER_SUFFIX)]
1996
+ pending = _read_pending_install(marker, part)
1997
+ if pending is None:
1998
+ _remove_pending_paths(marker, part)
1999
+ continue
2000
+ if pending.destination == expected_destination and pending.source_key != topic.source_key:
2001
+ _remove_pending_install(pending)
2002
+ continue
2003
+ if pending.source_key != topic.source_key or pending.destination != expected_destination:
2004
+ continue
2005
+ if _installed_pending_matches(pending, destination):
2006
+ return pending
2007
+ if not _pending_matches_topic(pending, topic) or not _is_safe_local_file(pending.part):
2008
+ _remove_pending_install(pending)
2009
+ continue
2010
+ actual_hash, actual_size = _hash_file(pending.part)
2011
+ if actual_hash != pending.sha256 or actual_size != pending.size:
2012
+ _remove_pending_install(pending)
2013
+ continue
2014
+ return pending
2015
+ return None
2016
+
2017
+
2018
+ def _retry_pending_install(
2019
+ vault: Vault,
2020
+ course_dir: Path,
2021
+ school: School,
2022
+ topic: TopicRecord,
2023
+ destination: Path,
2024
+ prior: ManifestEntry | None,
2025
+ pending: _PendingInstall,
2026
+ ) -> Literal["downloaded"] | None:
2027
+ """Install a previously validated part; return ``None`` when it is stale and was removed."""
2028
+ already_installed = _installed_pending_matches(pending, destination)
2029
+ if already_installed and prior is not None and prior.sha256 == pending.sha256:
2030
+ vault.mark(topic.source_key, prior)
2031
+ _remove_pending_install(pending)
2032
+ return None
2033
+ if pending.prior_sha256 is None:
2034
+ if prior is not None and pending.sha256 != prior.sha256:
2035
+ _remove_pending_install(pending)
2036
+ return None
2037
+ elif prior is None or pending.prior_sha256 != prior.sha256:
2038
+ _remove_pending_install(pending)
2039
+ return None
2040
+
2041
+ if pending.prior_sha256 is not None and not pending.revision_preserved:
2042
+ if prior is None:
2043
+ _remove_pending_install(pending)
2044
+ return None
2045
+ preserved = vault.preserve_revision(topic.source_key, changed_at=clock.now())
2046
+ if preserved is None and paths.long_path(vault.materialized(prior)).exists():
2047
+ raise A2LError("current source could not be preserved; refusing replacement")
2048
+ pending = replace(pending, revision_preserved=True)
2049
+ _write_pending_install(pending, root=vault.root)
2050
+
2051
+ if not already_installed:
2052
+ paths.atomic_install_temp(destination, pending.part, root=vault.root)
2053
+ entry = _manifest_entry_for_install(
2054
+ vault,
2055
+ topic,
2056
+ destination=destination,
2057
+ prior=prior,
2058
+ sha256=pending.sha256,
2059
+ size=pending.size,
2060
+ etag=pending.etag,
2061
+ last_modified=pending.last_modified,
2062
+ fetched_at=pending.fetched_at,
2063
+ )
2064
+ vault.mark(topic.source_key, entry)
2065
+ vault.save_manifest()
2066
+ _mark_topic_source_only(vault, course_dir, topic, school)
2067
+ _remove_pending_install(pending)
2068
+ return (
2069
+ None if already_installed and not _pending_matches_topic(pending, topic) else "downloaded"
2070
+ )
2071
+
2072
+
2073
+ def _manifest_entry_for_install(
2074
+ vault: Vault,
2075
+ topic: TopicRecord,
2076
+ *,
2077
+ destination: Path,
2078
+ prior: ManifestEntry | None,
2079
+ sha256: str,
2080
+ size: int,
2081
+ etag: str | None,
2082
+ last_modified: str | None,
2083
+ fetched_at: str | None = None,
2084
+ ) -> ManifestEntry:
2085
+ return ManifestEntry(
2086
+ path=paths.rel_posix(destination, vault.root),
2087
+ sha256=sha256,
2088
+ source_id=topic.source_id,
2089
+ etag=etag,
2090
+ last_modified=last_modified,
2091
+ size=size,
2092
+ fetched_at=fetched_at or _now(),
2093
+ derived=prior.derived if prior is not None and sha256 == prior.sha256 else {},
2094
+ )
2095
+
2096
+
2097
+ def _download_with_candidates(
2098
+ client: IngestClient,
2099
+ school: School,
2100
+ topic: TopicRecord,
2101
+ temporary: Path,
2102
+ *,
2103
+ prior: ManifestEntry | None,
2104
+ max_bytes: int | None,
2105
+ root: Path | None = None,
2106
+ ) -> DownloadResult:
2107
+ candidates = _download_candidates(client, school, topic)
2108
+ last_error: DownloadError | None = None
2109
+ for candidate in candidates:
2110
+ try:
2111
+ kwargs: dict[str, object] = {
2112
+ "prior": prior,
2113
+ "is_html_topic": topic.kind.casefold() == "html"
2114
+ or (topic.url_path or "").casefold().endswith((".html", ".htm")),
2115
+ }
2116
+ kwargs["max_bytes"] = max_bytes
2117
+ kwargs["root"] = root
2118
+ return client.download(candidate, temporary, **kwargs) # type: ignore[arg-type]
2119
+ except SessionExpired:
2120
+ raise
2121
+ except FileTooLarge:
2122
+ # A size refusal is a safety decision, not a route-health failure. Trying another
2123
+ # provider route could bypass the exact response-size check that protected the vault.
2124
+ raise
2125
+ except api.DiskSpaceExhausted:
2126
+ # Exhausted local disk space is fatal for every route; a later route's ordinary
2127
+ # failure must not downgrade it to a retryable download error.
2128
+ raise
2129
+ except DownloadError as exc:
2130
+ last_error = exc
2131
+ _remove_quietly(temporary)
2132
+ except RequestException:
2133
+ last_error = DownloadError("download route returned an unusable response")
2134
+ _remove_quietly(temporary)
2135
+ if last_error is not None:
2136
+ raise last_error
2137
+ raise DownloadError("no first-party download route was available")
2138
+
2139
+
2140
+ def _download_candidates(client: IngestClient, school: School, topic: TopicRecord) -> list[str]:
2141
+ base = school.base_url.rstrip("/")
2142
+ le = getattr(client, "le_version", None)
2143
+ ou = topic.course_org_unit_id
2144
+ tid = topic.topic_id
2145
+ candidates: list[str] = []
2146
+ if ":attachment:" in topic.source_key and topic.url_path is not None:
2147
+ return [urljoin(base + "/", topic.url_path.lstrip("/"))]
2148
+ template = client.download_template
2149
+ if isinstance(template, str) and template:
2150
+ with suppress(KeyError, ValueError):
2151
+ candidates.append(template.format(base=base, le=le or "", ou=ou, tid=tid))
2152
+ candidates.extend(
2153
+ [
2154
+ f"{base}/d2l/le/content/{ou}/topics/files/download/{tid}/DirectFileTopicDownload",
2155
+ f"{base}/d2l/api/le/{le}/{ou}/content/topics/{tid}/file"
2156
+ if isinstance(le, str) and le
2157
+ else "",
2158
+ ]
2159
+ )
2160
+ if topic.url_path is not None:
2161
+ candidates.append(urljoin(base + "/", topic.url_path.lstrip("/")))
2162
+ seen: set[str] = set()
2163
+ unique: list[str] = []
2164
+ for candidate in candidates:
2165
+ if candidate and candidate not in seen:
2166
+ seen.add(candidate)
2167
+ unique.append(candidate)
2168
+ return unique
2169
+
2170
+
2171
+ def _destination_for_topic(
2172
+ vault: Vault, course_dir: Path, topic: TopicRecord, prior: ManifestEntry | None
2173
+ ) -> Path:
2174
+ if prior is not None:
2175
+ return vault.materialized(prior)
2176
+ if topic.source_path is None:
2177
+ raise A2LError("topic has no allocated source path")
2178
+ return _vault_relative_path(vault, topic.source_path)
2179
+
2180
+
2181
+ def _unchanged_local(entry: ManifestEntry, topic: TopicRecord, vault: Vault) -> bool:
2182
+ if topic.etag is None and topic.last_modified is None:
2183
+ return False
2184
+ if topic.etag is not None and entry.etag != topic.etag:
2185
+ return False
2186
+ if topic.last_modified is not None and entry.last_modified != topic.last_modified:
2187
+ return False
2188
+ return _source_bytes_match(entry, vault)
2189
+
2190
+
2191
+ def _source_bytes_match(entry: ManifestEntry, vault: Vault) -> bool:
2192
+ source = vault.materialized(entry)
2193
+ try:
2194
+ return _hash_file(source) == (entry.sha256, entry.size)
2195
+ except (FileNotFoundError, IsADirectoryError):
2196
+ return False
2197
+
2198
+
2199
+ def _mark_topic_source_only(
2200
+ vault: Vault,
2201
+ course_dir: Path,
2202
+ topic: TopicRecord,
2203
+ school: School,
2204
+ ) -> None:
2205
+ rows = _map_topics(_read_content_map(course_dir))
2206
+ # A manifest artifact record is not proof that the current twin bytes are still trusted.
2207
+ # Reconcile through the same source-and-derived hash checks used by metadata sync.
2208
+ reconciled = course_index.reconcile_content_map(vault, rows)
2209
+ _write_content_map(course_dir, reconciled, root=vault.root)
2210
+ course = CourseRef(
2211
+ topic.course_org_unit_id,
2212
+ topic.course_code,
2213
+ topic.course_name,
2214
+ topic.term,
2215
+ True,
2216
+ )
2217
+ topics = tuple(
2218
+ _topic_from_row(row, course=course) for row in reconciled if isinstance(row, dict)
2219
+ )
2220
+ _write_index(course_dir, school=school, course=course, topics=topics, root=vault.root)
2221
+
2222
+
2223
+ def _update_row_state(
2224
+ course_dir: Path, source_key: str, *, root: Path | None = None, **updates: object
2225
+ ) -> None:
2226
+ content_map = _read_content_map(course_dir)
2227
+ rows = _map_topics(content_map)
2228
+ for row in rows:
2229
+ if isinstance(row, dict) and row.get("source_key") == source_key:
2230
+ row.update(updates)
2231
+ _write_content_map(course_dir, rows, root=root)
2232
+
2233
+
2234
+ def _find_content_row(course_dir: Path, source_key: str) -> dict[str, object] | None:
2235
+ for row in _map_topics(_read_content_map(course_dir)):
2236
+ if isinstance(row, dict) and row.get("source_key") == source_key:
2237
+ return row
2238
+ return None
2239
+
2240
+
2241
+ def _resolve_topic(vault: Vault, query: str) -> tuple[CourseRef, TopicRecord, Path] | None:
2242
+ if not isinstance(query, str) or not query.strip():
2243
+ raise A2LError("topic selector must not be empty")
2244
+ exact: list[tuple[CourseRef, TopicRecord, Path]] = []
2245
+ fuzzy: list[tuple[CourseRef, TopicRecord, Path]] = []
2246
+ folded = query.casefold()
2247
+ for map_path in sorted(
2248
+ path for path in paths.walk(vault.root) if path.name == "content_map.json"
2249
+ ):
2250
+ course_dir = map_path.parent.parent
2251
+ raw = _read_content_map(course_dir)
2252
+ for row in _map_topics(raw):
2253
+ if not isinstance(row, dict):
2254
+ continue
2255
+ try:
2256
+ course = CourseRef(
2257
+ int(row["course_org_unit_id"]),
2258
+ str(row["course_code"]),
2259
+ str(row["course_name"]),
2260
+ row.get("term") if isinstance(row.get("term"), str) else None,
2261
+ True,
2262
+ )
2263
+ record = _topic_from_row(row, course=course)
2264
+ except (KeyError, TypeError, ValueError, A2LError):
2265
+ continue
2266
+ candidate = (course, record, course_dir)
2267
+ if query in {record.source_key, record.source_id} or query == record.source_path:
2268
+ exact.append(candidate)
2269
+ elif folded in record.title.casefold() or (
2270
+ record.path is not None and folded in record.path.casefold()
2271
+ ):
2272
+ fuzzy.append(candidate)
2273
+ if exact:
2274
+ if len(exact) > 1:
2275
+ raise A2LError(f"ambiguous topic selector: {query}")
2276
+ return exact[0]
2277
+ if len(fuzzy) > 1:
2278
+ raise A2LError(f"ambiguous topic selector: {query}")
2279
+ return fuzzy[0] if fuzzy else None
2280
+
2281
+
2282
+ def _read_content_map(course_dir: Path) -> dict[str, object]:
2283
+ return course_index.read_content_map(course_dir)
2284
+
2285
+
2286
+ def _map_topics(content_map: Mapping[str, object]) -> list[object]:
2287
+ topics = content_map.get("topics")
2288
+ if not isinstance(topics, list):
2289
+ raise A2LError("content_map.json topics must be an array")
2290
+ return topics
2291
+
2292
+
2293
+ def _write_content_map(
2294
+ course_dir: Path, rows: Sequence[object], *, root: Path | None = None
2295
+ ) -> None:
2296
+ course_index.write_content_map(course_dir, rows, root=root)
2297
+
2298
+
2299
+ def _write_toc(
2300
+ course_dir: Path, modules: Sequence[dict[str, object]], *, root: Path | None = None
2301
+ ) -> None:
2302
+ _write_json(
2303
+ course_dir / "_meta" / "toc.json",
2304
+ {"schema_version": 1, "modules": list(modules)},
2305
+ root=root,
2306
+ )
2307
+
2308
+
2309
+ def _read_toc_modules(course_dir: Path) -> list[dict[str, object]]:
2310
+ destination = course_dir / "_meta" / "toc.json"
2311
+ try:
2312
+ with open(os.fspath(paths.long_path(destination)), encoding="utf-8", newline="") as handle:
2313
+ raw: Any = json.load(handle)
2314
+ except FileNotFoundError:
2315
+ return []
2316
+ except (OSError, UnicodeError, json.JSONDecodeError) as exc:
2317
+ raise A2LError(f"{destination.name} is unreadable") from exc
2318
+ if not isinstance(raw, dict):
2319
+ raise A2LError(f"{destination.name} must contain an object")
2320
+ modules = raw.get("modules")
2321
+ if not isinstance(modules, list):
2322
+ raise A2LError(f"{destination.name} must contain a modules list")
2323
+ if any(not isinstance(value, dict) for value in modules):
2324
+ raise A2LError(f"{destination.name} contains an invalid module")
2325
+ return [cast(dict[str, object], value) for value in modules]
2326
+
2327
+
2328
+ def _write_json(destination: Path, payload: object, *, root: Path | None = None) -> None:
2329
+ paths.ensure_dir(destination.parent, root=root)
2330
+ text = (
2331
+ json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2, separators=(",", ": "))
2332
+ + "\n"
2333
+ )
2334
+ paths.atomic_write_text(destination, text, root=root)
2335
+
2336
+
2337
+ def _read_list(destination: Path) -> list[dict[str, object]]:
2338
+ try:
2339
+ with open(os.fspath(paths.long_path(destination)), encoding="utf-8", newline="") as handle:
2340
+ raw: Any = json.load(handle)
2341
+ except FileNotFoundError:
2342
+ return []
2343
+ except (OSError, UnicodeError, json.JSONDecodeError) as exc:
2344
+ raise A2LError(f"{destination.name} is unreadable") from exc
2345
+ if not isinstance(raw, list):
2346
+ raise A2LError(f"{destination.name} must contain a list")
2347
+ if any(not isinstance(row, dict) for row in raw):
2348
+ raise A2LError(f"{destination.name} contains an invalid item")
2349
+ return [cast(dict[str, object], row) for row in raw]
2350
+
2351
+
2352
+ def _write_list(
2353
+ destination: Path, rows: Sequence[Mapping[str, object]], *, root: Path | None = None
2354
+ ) -> None:
2355
+ _write_json(destination, list(rows), root=root)
2356
+
2357
+
2358
+ def _merge_rows(
2359
+ existing: Sequence[Mapping[str, object]],
2360
+ incoming: Sequence[Mapping[str, object]],
2361
+ *,
2362
+ id_field: str,
2363
+ complete: bool,
2364
+ ) -> list[dict[str, object]]:
2365
+ by_id: dict[str, dict[str, object]] = {}
2366
+ for row in existing:
2367
+ value = row.get(id_field)
2368
+ if value is not None:
2369
+ by_id[str(value)] = dict(row)
2370
+ incoming_ids: set[str] = set()
2371
+ for row in incoming:
2372
+ value = row.get(id_field)
2373
+ if value is None:
2374
+ continue
2375
+ key = str(value)
2376
+ incoming_ids.add(key)
2377
+ merged = dict(by_id.get(key, {}))
2378
+ merged.update(row)
2379
+ merged["missing_since"] = None
2380
+ merged["withdrawn_at"] = None
2381
+ by_id[key] = merged
2382
+ if complete:
2383
+ now = _now()
2384
+ for key, row in by_id.items():
2385
+ if key in incoming_ids:
2386
+ continue
2387
+ if row.get("missing_since") is None:
2388
+ row["missing_since"] = now
2389
+ elif row.get("withdrawn_at") is None:
2390
+ row["withdrawn_at"] = now
2391
+ return sorted(
2392
+ by_id.values(), key=lambda row: (_date_key(row.get("date")), str(row.get(id_field)))
2393
+ )
2394
+
2395
+
2396
+ def _project_assignments(values: Sequence[object]) -> list[dict[str, object]]:
2397
+ rows: list[dict[str, object]] = []
2398
+ for value in values:
2399
+ if (
2400
+ not isinstance(value, dict)
2401
+ or not isinstance(value.get("Id"), int)
2402
+ or isinstance(value.get("Id"), bool)
2403
+ ):
2404
+ continue
2405
+ availability = value.get("Availability")
2406
+ rows.append(
2407
+ {
2408
+ "id": value["Id"],
2409
+ "title": _safe_text(value.get("Name")),
2410
+ "due_date": _optional_text(value.get("DueDate")),
2411
+ "start_date": _nested_optional_text(availability, "StartDate"),
2412
+ "end_date": _nested_optional_text(availability, "EndDate"),
2413
+ "grade_item": isinstance(value.get("GradeItemId"), int),
2414
+ "group": isinstance(value.get("GroupTypeId"), int),
2415
+ "date": _optional_text(value.get("DueDate")) or "",
2416
+ }
2417
+ )
2418
+ return rows
2419
+
2420
+
2421
+ def _project_news(values: Sequence[object]) -> list[dict[str, object]]:
2422
+ rows: list[dict[str, object]] = []
2423
+ for value in values:
2424
+ if (
2425
+ not isinstance(value, dict)
2426
+ or not isinstance(value.get("Id"), int)
2427
+ or isinstance(value.get("Id"), bool)
2428
+ ):
2429
+ continue
2430
+ body = value.get("Body")
2431
+ body_text = body.get("Text") if isinstance(body, dict) else None
2432
+ body_html = body.get("Html") if isinstance(body, dict) else None
2433
+ start = _optional_text(value.get("StartDate")) or ""
2434
+ rows.append(
2435
+ {
2436
+ "id": value["Id"],
2437
+ "title": _safe_text(value.get("Title")),
2438
+ "text": _safe_text(body_text),
2439
+ "html": _sanitize_html_text(body_html) if isinstance(body_html, str) else None,
2440
+ "start_date": _optional_text(value.get("StartDate")),
2441
+ "end_date": _optional_text(value.get("EndDate")),
2442
+ "published": bool(value.get("IsPublished", False)),
2443
+ "date": start,
2444
+ }
2445
+ )
2446
+ return rows
2447
+
2448
+
2449
+ def _project_quizzes(values: Sequence[object]) -> list[dict[str, object]]:
2450
+ rows: list[dict[str, object]] = []
2451
+ for value in values:
2452
+ if (
2453
+ not isinstance(value, dict)
2454
+ or not isinstance(value.get("QuizId"), int)
2455
+ or isinstance(value.get("QuizId"), bool)
2456
+ ):
2457
+ continue
2458
+ due = _optional_text(value.get("DueDate"))
2459
+ rows.append(
2460
+ {
2461
+ "id": value["QuizId"],
2462
+ "title": _safe_text(value.get("Name")),
2463
+ "due_date": due,
2464
+ "start_date": _optional_text(value.get("StartDate")),
2465
+ "end_date": _optional_text(value.get("EndDate")),
2466
+ "active": bool(value.get("IsActive", False)),
2467
+ "date": due or "",
2468
+ }
2469
+ )
2470
+ return rows
2471
+
2472
+
2473
+ def _project_grades(values: Sequence[object]) -> list[dict[str, object]]:
2474
+ # Grades are opt-in, but the projection still excludes all unrelated response fields.
2475
+ rows: list[dict[str, object]] = []
2476
+ for value in values:
2477
+ if not isinstance(value, dict):
2478
+ continue
2479
+ identifier = value.get("GradeObjectIdentifier")
2480
+ if identifier is None:
2481
+ continue
2482
+ rows.append(
2483
+ {
2484
+ "id": str(identifier),
2485
+ "name": _safe_text(value.get("GradeObjectName")),
2486
+ "type": _safe_text(value.get("GradeObjectType")),
2487
+ "numerator": value.get("PointsNumerator"),
2488
+ "denominator": value.get("PointsDenominator"),
2489
+ "displayed": _safe_text(value.get("DisplayedGrade")),
2490
+ }
2491
+ )
2492
+ return rows
2493
+
2494
+
2495
+ def _write_announcements(
2496
+ destination: Path,
2497
+ rows: Sequence[Mapping[str, object]],
2498
+ *,
2499
+ root: Path | None = None,
2500
+ ) -> None:
2501
+ paths.ensure_dir(destination.parent, root=root)
2502
+ lines = ["# Announcements", ""]
2503
+ for row in rows:
2504
+ title = str(row.get("title") or "Untitled")
2505
+ date = str(row.get("start_date") or "")
2506
+ lines.extend([f"## {title}", "", f"Date: {date}" if date else "", ""])
2507
+ if row.get("withdrawn_at"):
2508
+ lines.extend(["> No longer posted in LEARN.", ""])
2509
+ text = str(row.get("text") or "").strip()
2510
+ if text:
2511
+ lines.extend([text, ""])
2512
+ paths.atomic_write_text(destination, "\n".join(lines).rstrip() + "\n", root=root)
2513
+
2514
+
2515
+ def _write_index(
2516
+ course_dir: Path,
2517
+ *,
2518
+ school: School | None,
2519
+ course: CourseRef | None,
2520
+ topics: Sequence[TopicRecord] | None,
2521
+ root: Path | None = None,
2522
+ ) -> None:
2523
+ if school is None or course is None or topics is None:
2524
+ # A file-only checkpoint updates the coverage tree without needing to retain a second
2525
+ # network response in memory. The existing index remains valid until metadata runs again.
2526
+ return
2527
+ term_label = school.term_label(course.term) if course.term else "Unclassified"
2528
+ assignments = _read_list(course_dir / "_meta" / "assignments.json")
2529
+ quizzes = _read_list(course_dir / "_meta" / "quizzes.json")
2530
+ deadlines = [
2531
+ (str(row.get("due_date")), str(row.get("title") or "Untitled"), "assignment")
2532
+ for row in assignments
2533
+ if row.get("due_date")
2534
+ ] + [
2535
+ (str(row.get("due_date")), str(row.get("title") or "Untitled"), "quiz")
2536
+ for row in quizzes
2537
+ if row.get("due_date")
2538
+ ]
2539
+ course_index.write_course_index(
2540
+ course_dir,
2541
+ course_code=course.code,
2542
+ course_name=course.name,
2543
+ term_label=term_label,
2544
+ term_code=course.term,
2545
+ topics=[_topic_to_row(topic) for topic in topics],
2546
+ deadlines=deadlines,
2547
+ root=root,
2548
+ )
2549
+
2550
+
2551
+ def _course_relative_link(value: str, course_dir: Path) -> str:
2552
+ parts = list(PurePosixPath(value).parts)
2553
+ try:
2554
+ index = parts.index(course_dir.name)
2555
+ except ValueError:
2556
+ return PurePosixPath(value).as_posix()
2557
+ relative = parts[index + 1 :]
2558
+ return PurePosixPath(*relative).as_posix() if relative else "."
2559
+
2560
+
2561
+ def _content_directory(course_dir: Path, module_path: Sequence[str]) -> Path:
2562
+ directory = course_dir / _COURSE_CONTENT
2563
+ for component in module_path:
2564
+ directory /= paths.safe_name(component)
2565
+ return directory
2566
+
2567
+
2568
+ def _extension(name: str) -> str:
2569
+ """Return a usable extension, treating a bare trailing dot as no extension.
2570
+
2571
+ Python 3.14 changed ``PurePath.suffix``: ``'Reading list.'`` now reports ``'.'`` where
2572
+ earlier versions report ``''``. Taken at face value that lone dot looks like an
2573
+ extension, so a topic titled with a trailing dot keeps no real extension and the file
2574
+ lands with none at all — a different vault on 3.14 than on 3.11. A single dot is not an
2575
+ extension on any version, so it is normalized away here rather than at each call site.
2576
+ """
2577
+ suffix = PurePosixPath(name).suffix
2578
+ return suffix if len(suffix) > 1 else ""
2579
+
2580
+
2581
+ def _topic_filename(topic: TopicRecord) -> str:
2582
+ title = topic.title.strip() or "untitled"
2583
+ title_path = PurePosixPath(title).name
2584
+ url_name = PurePosixPath(urlsplit(topic.url_path or "").path).name
2585
+ title_suffix = _extension(title_path)
2586
+ suffix = title_suffix or _extension(url_name)
2587
+ base = title_path if title_suffix else f"{title_path}{suffix}"
2588
+ return paths.safe_name(base)
2589
+
2590
+
2591
+ def _unique_reserved(destination: Path, reserved: set[str]) -> Path:
2592
+ candidate = paths.unique_path(destination)
2593
+ for number in range(1, 100_000):
2594
+ if _canonical_name(candidate) not in reserved and not paths.collides(candidate):
2595
+ reserved.add(_canonical_name(candidate))
2596
+ return candidate
2597
+ stem, extension = _split_name(candidate.name)
2598
+ suffix = "" if number == 1 else f"_{number}"
2599
+ candidate = candidate.with_name(paths.safe_name(f"{stem}{suffix}{extension}"))
2600
+ raise A2LError("could not allocate a unique content path")
2601
+
2602
+
2603
+ def _canonical_name(path: Path) -> str:
2604
+ return path.name.casefold()
2605
+
2606
+
2607
+ def _split_name(name: str) -> tuple[str, str]:
2608
+ # ``_extension`` rather than ``.suffix``: a collision suffix must be inserted at the
2609
+ # same place on every Python version (see the 3.14 note there).
2610
+ suffix = _extension(name)
2611
+ return (name[: -len(suffix)], suffix) if suffix else (name, "")
2612
+
2613
+
2614
+ def _vault_relative_path(vault: Vault, value: str) -> Path:
2615
+ if (
2616
+ not isinstance(value, str)
2617
+ or not value
2618
+ or "\\" in value
2619
+ or re.match(r"^[A-Za-z]:[\\/]", value) is not None
2620
+ or PurePosixPath(value).is_absolute()
2621
+ or any(part in {"", ".", ".."} for part in PurePosixPath(value).parts)
2622
+ ):
2623
+ raise A2LError("content path must be a normalized relative POSIX path")
2624
+ candidate = (vault.root / Path(*PurePosixPath(value).parts)).resolve()
2625
+ try:
2626
+ candidate.relative_to(vault.root)
2627
+ except ValueError as exc:
2628
+ raise A2LError("content path escapes the vault root") from exc
2629
+ return candidate
2630
+
2631
+
2632
+ def _derived_path(manifest: Mapping[str, ManifestEntry], entry: ManifestEntry) -> str | None:
2633
+ artifact = entry.derived.get("markdown")
2634
+ if artifact is None:
2635
+ return None
2636
+ return artifact.path if artifact.path else None
2637
+
2638
+
2639
+ def _is_media(topic: TopicRecord) -> bool:
2640
+ return PurePosixPath(_topic_filename(topic)).suffix.casefold() in _MEDIA_SUFFIXES
2641
+
2642
+
2643
+ def is_media_topic(topic: TopicRecord) -> bool:
2644
+ """Return the canonical media classification used by file ingestion and previews."""
2645
+
2646
+ return _is_media(topic)
2647
+
2648
+
2649
+ def _is_office_lock(topic: TopicRecord) -> bool:
2650
+ filename = _topic_filename(topic)
2651
+ url_name = PurePosixPath(urlsplit(topic.url_path or "").path).name
2652
+ return bool(_OFFICE_LOCK.match(filename) or _OFFICE_LOCK.match(url_name))
2653
+
2654
+
2655
+ def _safe_hostname(value: str | None) -> str | None:
2656
+ if not value:
2657
+ return None
2658
+ try:
2659
+ hostname = urlsplit(value).hostname
2660
+ if hostname is None:
2661
+ return None
2662
+ return hostname.encode("idna").decode("ascii").casefold().rstrip(".")
2663
+ except (UnicodeError, ValueError):
2664
+ return None
2665
+
2666
+
2667
+ def _first_party_path(value: str | None, base_url: str) -> str | None:
2668
+ if not value:
2669
+ return None
2670
+ try:
2671
+ candidate = urljoin(base_url.rstrip("/") + "/", value)
2672
+ parsed = urlsplit(candidate)
2673
+ base = urlsplit(base_url)
2674
+ if parsed.scheme.casefold() not in {"http", "https"}:
2675
+ return None
2676
+ if parsed.username is not None or parsed.password is not None:
2677
+ return None
2678
+ host = _safe_hostname(candidate)
2679
+ base_host = _safe_hostname(base_url)
2680
+ if host != base_host or parsed.port != base.port:
2681
+ return None
2682
+ path = parsed.path or "/"
2683
+ return urlunsplit(("", "", path, "", ""))
2684
+ except (TypeError, ValueError):
2685
+ return None
2686
+
2687
+
2688
+ def _allowed_outline_url(value: str | None, school: School) -> str | None:
2689
+ if not value:
2690
+ return None
2691
+ try:
2692
+ parsed = urlsplit(value)
2693
+ if (
2694
+ parsed.scheme.casefold() != "https"
2695
+ or parsed.username is not None
2696
+ or parsed.password is not None
2697
+ or not hostname_matches_suffix(value, school.outline_hosts())
2698
+ ):
2699
+ return None
2700
+ return urlunsplit(("https", parsed.netloc, parsed.path or "/", "", ""))
2701
+ except (TypeError, ValueError):
2702
+ return None
2703
+
2704
+
2705
+ def _view_url(school: School, org_unit_id: int, topic_id: int) -> str:
2706
+ return f"{school.base_url.rstrip('/')}/d2l/le/content/{org_unit_id}/viewContent/{topic_id}/View"
2707
+
2708
+
2709
+ def _safe_text(value: object) -> str:
2710
+ return value if isinstance(value, str) else ""
2711
+
2712
+
2713
+ def _optional_text(value: object) -> str | None:
2714
+ return value if isinstance(value, str) else None
2715
+
2716
+
2717
+ def _nested_optional_text(value: object, key: str) -> str | None:
2718
+ return _optional_text(value.get(key)) if isinstance(value, dict) else None
2719
+
2720
+
2721
+ class _RichTextSanitizer(HTMLParser):
2722
+ """Render inert, deterministic HTML without executing or retaining active attributes."""
2723
+
2724
+ _ALLOWED_TAGS = frozenset(
2725
+ {
2726
+ "a",
2727
+ "b",
2728
+ "blockquote",
2729
+ "br",
2730
+ "code",
2731
+ "dd",
2732
+ "div",
2733
+ "em",
2734
+ "h1",
2735
+ "h2",
2736
+ "h3",
2737
+ "h4",
2738
+ "h5",
2739
+ "h6",
2740
+ "hr",
2741
+ "img",
2742
+ "li",
2743
+ "ol",
2744
+ "p",
2745
+ "pre",
2746
+ "section",
2747
+ "span",
2748
+ "strong",
2749
+ "table",
2750
+ "tbody",
2751
+ "td",
2752
+ "th",
2753
+ "thead",
2754
+ "tr",
2755
+ "u",
2756
+ "ul",
2757
+ }
2758
+ )
2759
+ _VOID_TAGS = frozenset({"br", "hr", "img"})
2760
+
2761
+ def __init__(self, base_url: str) -> None:
2762
+ super().__init__(convert_charrefs=False)
2763
+ self.base_url = base_url
2764
+ self.parts: list[str] = []
2765
+ self._skip_depth = 0
2766
+
2767
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
2768
+ tag = tag.casefold()
2769
+ if self._skip_depth:
2770
+ if tag not in self._VOID_TAGS:
2771
+ self._skip_depth += 1
2772
+ return
2773
+ if tag in {"script", "style", "form", "iframe", "object", "embed", "template"}:
2774
+ self._skip_depth = 1
2775
+ return
2776
+ if tag not in self._ALLOWED_TAGS:
2777
+ return
2778
+ safe_attrs: list[str] = []
2779
+ for name, value in attrs:
2780
+ name = name.casefold()
2781
+ if name == "href" and tag == "a" and value is not None:
2782
+ safe_url = _safe_richtext_url(value, self.base_url)
2783
+ if safe_url is not None:
2784
+ safe_attrs.append(f'href="{escape(safe_url, quote=True)}"')
2785
+ elif name in {"alt", "title"} and value is not None and tag in {"a", "img"}:
2786
+ safe_attrs.append(f'{name}="{escape(value, quote=True)}"')
2787
+ suffix = "" if not safe_attrs else " " + " ".join(safe_attrs)
2788
+ self.parts.append(f"<{tag}{suffix}>")
2789
+
2790
+ def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
2791
+ self.handle_starttag(tag, attrs)
2792
+
2793
+ def handle_endtag(self, tag: str) -> None:
2794
+ tag = tag.casefold()
2795
+ if self._skip_depth:
2796
+ self._skip_depth -= 1
2797
+ return
2798
+ if tag in self._ALLOWED_TAGS and tag not in self._VOID_TAGS:
2799
+ self.parts.append(f"</{tag}>")
2800
+
2801
+ def handle_data(self, data: str) -> None:
2802
+ if not self._skip_depth:
2803
+ self.parts.append(escape(data))
2804
+
2805
+ def handle_entityref(self, name: str) -> None:
2806
+ if not self._skip_depth:
2807
+ self.parts.append(f"&{name};")
2808
+
2809
+ def handle_charref(self, name: str) -> None:
2810
+ if not self._skip_depth:
2811
+ self.parts.append(f"&#{name};")
2812
+
2813
+ def handle_comment(self, _data: str) -> None:
2814
+ return
2815
+
2816
+
2817
+ def _sanitize_richtext(value: str, base_url: str) -> str:
2818
+ parser = _RichTextSanitizer(base_url)
2819
+ try:
2820
+ parser.feed(value)
2821
+ parser.close()
2822
+ except (TypeError, ValueError):
2823
+ return ""
2824
+ return "".join(parser.parts).strip()
2825
+
2826
+
2827
+ def _safe_richtext_url(value: str, base_url: str) -> str | None:
2828
+ try:
2829
+ candidate = urljoin(base_url.rstrip("/") + "/", value)
2830
+ parsed = urlsplit(candidate)
2831
+ if parsed.scheme.casefold() not in {"http", "https"}:
2832
+ return None
2833
+ if parsed.username is not None or parsed.password is not None or parsed.hostname is None:
2834
+ return None
2835
+ # RichText links are inert and query-free. They are never followed by conversion or sync.
2836
+ return urlunsplit((parsed.scheme.casefold(), parsed.netloc, parsed.path or "/", "", ""))
2837
+ except (TypeError, ValueError):
2838
+ return None
2839
+
2840
+
2841
+ def _richtext_markdown(title: str, sanitized_html: str) -> str:
2842
+ body = re.sub(r"(?s)<br\s*/?>", "\n", sanitized_html, flags=re.IGNORECASE)
2843
+ body = re.sub(r"(?s)</(?:p|div|h[1-6]|li|tr|blockquote)>", "\n", body, flags=re.IGNORECASE)
2844
+ body = re.sub(r"(?s)<[^>]+>", "", body)
2845
+ body = re.sub(r"\n{3,}", "\n\n", body)
2846
+ body = re.sub(r"[ \t]+", " ", body)
2847
+ body = body.strip()
2848
+ return f"# {title}\n\n{body}\n"
2849
+
2850
+
2851
+ def _sanitize_html_text(value: str) -> str:
2852
+ # News remains a typed projection; unlike assignment instructions it does not become an
2853
+ # evidence-bearing source/twin, so retain a compact plain-text rendering only.
2854
+ text = re.sub(r"<[^>]*>", " ", value)
2855
+ return re.sub(r"\s+", " ", text).strip()
2856
+
2857
+
2858
+ def _now() -> str:
2859
+ return clock.stamp()
2860
+
2861
+
2862
+ def _date_key(value: object) -> str:
2863
+ return value if isinstance(value, str) else ""
2864
+
2865
+
2866
+ def _timestamp_sort(value: str | None) -> float:
2867
+ if value is None:
2868
+ return 0.0
2869
+ try:
2870
+ return parse_api_timestamp(value).timestamp()
2871
+ except ValueError:
2872
+ return 0.0
2873
+
2874
+
2875
+ def _hash_file(path: Path) -> tuple[str, int]:
2876
+ digest = sha256()
2877
+ size = 0
2878
+ with open(os.fspath(paths.long_path(path)), "rb") as handle:
2879
+ while chunk := handle.read(1024 * 1024):
2880
+ digest.update(chunk)
2881
+ size += len(chunk)
2882
+ return digest.hexdigest(), size
2883
+
2884
+
2885
+ def _remove_quietly(path: Path) -> None:
2886
+ try:
2887
+ os.unlink(os.fspath(paths.long_path(path)))
2888
+ except OSError:
2889
+ return
2890
+
2891
+
2892
+ def _safe_error(stage: str, exc: BaseException) -> str:
2893
+ # Reports are intentionally category-level. Never include URL, title, path, or response text.
2894
+ return f"{stage}: {type(exc).__name__}"
2895
+
2896
+
2897
+ def _ingest_discussions(
2898
+ client: IngestClient,
2899
+ course: CourseRef,
2900
+ course_dir: Path,
2901
+ vault: Vault,
2902
+ *,
2903
+ include_authors: bool,
2904
+ ) -> str | None:
2905
+ """Fetch opt-in discussions while keeping raw author identity out of the vault."""
2906
+
2907
+ values, complete, error = _fetch_collection(
2908
+ client, _endpoint_path(client, course, "discussions/forums/")
2909
+ )
2910
+ if error is not None:
2911
+ return _safe_error("discussions", error)
2912
+ if not complete:
2913
+ return _safe_error("discussions", A2LError("incomplete response"))
2914
+ if any(
2915
+ not isinstance(value, dict) or not _discussion_forum_is_valid(value) for value in values
2916
+ ):
2917
+ return _safe_error("discussions", A2LError("metadata endpoint returned an invalid forum"))
2918
+ forum_ids = [value["ForumId"] for value in values if isinstance(value, dict)]
2919
+ if len({str(value) for value in forum_ids}) != len(forum_ids):
2920
+ return _safe_error("discussions", A2LError("metadata endpoint returned duplicate forums"))
2921
+ try:
2922
+ existing_rows = _read_discussion_rows(course_dir / "_meta" / "discussions.json")
2923
+ except A2LError as exc:
2924
+ return _safe_error("discussions", exc)
2925
+ key = _discussion_key(vault)
2926
+ posts = [
2927
+ post for value in values if isinstance(value, dict) for post in _discussion_posts(value)
2928
+ ]
2929
+ identities = {
2930
+ identity for post in posts if (identity := _discussion_identity(post)) is not None
2931
+ }
2932
+ pseudonyms = _discussion_pseudonyms(key, identities)
2933
+ incoming_rows: list[dict[str, object]] = []
2934
+ for value in values:
2935
+ if not isinstance(value, dict) or not isinstance(value.get("ForumId"), int):
2936
+ continue
2937
+ description = value.get("Description")
2938
+ description_text = _discussion_body_text(description, school_base=client.school.base_url)
2939
+ forum_row: dict[str, object] = {
2940
+ "id": value["ForumId"],
2941
+ "name": _safe_text(value.get("Name")),
2942
+ "description": description_text,
2943
+ }
2944
+ rendered_posts: list[dict[str, object]] = []
2945
+ for post in _discussion_posts(value):
2946
+ identity = _discussion_identity(post)
2947
+ body = _discussion_body_text(post.get("Body", post), school_base=client.school.base_url)
2948
+ author = _discussion_author(post, identity, pseudonyms, include_authors)
2949
+ rendered_posts.append(
2950
+ {
2951
+ "id": _discussion_post_id(post),
2952
+ "author": author,
2953
+ "text": body,
2954
+ "date": _first_string(post, "Date", "PostingDate", "LastModifiedDate"),
2955
+ }
2956
+ )
2957
+ if rendered_posts:
2958
+ forum_row["posts"] = rendered_posts
2959
+ incoming_rows.append(forum_row)
2960
+
2961
+ rows = _merge_discussion_rows(existing_rows, incoming_rows)
2962
+ _write_list(course_dir / "_meta" / "discussions.json", rows, root=vault.root)
2963
+ markdown_lines = ["# Discussions", ""]
2964
+ for forum_row in rows:
2965
+ markdown_lines.extend([f"## {forum_row['name'] or 'Untitled forum'}", ""])
2966
+ description_text = str(forum_row.get("description") or "")
2967
+ if description_text:
2968
+ markdown_lines.extend([description_text, ""])
2969
+ if forum_row.get("withdrawn_at"):
2970
+ markdown_lines.extend(["> No longer posted in LEARN.", ""])
2971
+ raw_posts = forum_row.get("posts", [])
2972
+ for post in raw_posts if isinstance(raw_posts, list) else []:
2973
+ if not isinstance(post, dict):
2974
+ continue
2975
+ markdown_lines.extend(
2976
+ [
2977
+ f"### {post.get('author') or 'author-unknown'}",
2978
+ "",
2979
+ str(post.get("text") or ""),
2980
+ "",
2981
+ ]
2982
+ )
2983
+
2984
+ discussion_dir = course_dir / "discussions"
2985
+ paths.ensure_dir(discussion_dir, root=vault.root)
2986
+ paths.atomic_write_text(
2987
+ discussion_dir / "discussions.md", "\n".join(markdown_lines), root=vault.root
2988
+ )
2989
+ return None
2990
+
2991
+
2992
+ def _discussion_key(vault: Vault) -> bytes:
2993
+ private = vault.state() / "private"
2994
+ paths.ensure_dir(private, root=vault.root, mode=0o700)
2995
+ destination = private / "discussion-hmac.key"
2996
+ if paths.has_link_component(destination, root=vault.root):
2997
+ raise A2LError("discussion pseudonym key path contains a link component")
2998
+ try:
2999
+ with open(os.fspath(paths.long_path(destination)), "rb") as handle:
3000
+ key = handle.read()
3001
+ except FileNotFoundError:
3002
+ key = secrets.token_bytes(32)
3003
+ paths.atomic_write_bytes(destination, key, root=vault.root)
3004
+ if len(key) != 32:
3005
+ raise A2LError("discussion pseudonym key is invalid")
3006
+ return key
3007
+
3008
+
3009
+ def _discussion_posts(forum: Mapping[str, object]) -> list[dict[str, object]]:
3010
+ raw_topics = forum.get("Topics", forum.get("topics", []))
3011
+ if isinstance(raw_topics, dict):
3012
+ raw_topics = raw_topics.get("Items", raw_topics.get("Objects", []))
3013
+ raw_posts: list[object] = []
3014
+ if isinstance(raw_topics, list):
3015
+ for topic in raw_topics:
3016
+ if isinstance(topic, dict):
3017
+ posts = topic.get("Posts", topic.get("posts", []))
3018
+ if isinstance(posts, list):
3019
+ raw_posts.extend(posts)
3020
+ direct_posts = forum.get("Posts", forum.get("posts", []))
3021
+ if isinstance(direct_posts, list):
3022
+ raw_posts.extend(direct_posts)
3023
+ return [post for post in raw_posts if isinstance(post, dict)]
3024
+
3025
+
3026
+ def _read_discussion_rows(destination: Path) -> list[dict[str, object]]:
3027
+ rows = _read_list(destination)
3028
+ validated: list[dict[str, object]] = []
3029
+ seen_forums: set[str] = set()
3030
+ for row in rows:
3031
+ identifier = row.get("id")
3032
+ if isinstance(identifier, bool) or not isinstance(identifier, int):
3033
+ raise A2LError("discussions.json contains an invalid forum ID")
3034
+ forum_key = str(identifier)
3035
+ if forum_key in seen_forums:
3036
+ raise A2LError("discussions.json contains duplicate forum IDs")
3037
+ seen_forums.add(forum_key)
3038
+ raw_posts = row.get("posts", [])
3039
+ if not isinstance(raw_posts, list) or any(
3040
+ not isinstance(post, dict) or not _discussion_post_is_valid(post) for post in raw_posts
3041
+ ):
3042
+ raise A2LError("discussions.json contains invalid posts")
3043
+ post_ids = [str(post["id"]) for post in raw_posts if isinstance(post, dict)]
3044
+ if len(set(post_ids)) != len(post_ids):
3045
+ raise A2LError("discussions.json contains duplicate post IDs")
3046
+ validated.append(dict(row))
3047
+ return validated
3048
+
3049
+
3050
+ def _merge_discussion_rows(
3051
+ existing: Sequence[Mapping[str, object]], incoming: Sequence[Mapping[str, object]]
3052
+ ) -> list[dict[str, object]]:
3053
+ """Union complete discussion captures by forum and post ID without deleting history."""
3054
+ by_forum: dict[str, dict[str, object]] = {
3055
+ str(row["id"]): dict(row) for row in existing if isinstance(row.get("id"), int)
3056
+ }
3057
+ incoming_forums: set[str] = set()
3058
+ for row in incoming:
3059
+ forum_id = row.get("id")
3060
+ if isinstance(forum_id, bool) or not isinstance(forum_id, int):
3061
+ raise A2LError("discussion capture contains an invalid forum ID")
3062
+ forum_key = str(forum_id)
3063
+ incoming_forums.add(forum_key)
3064
+ prior = by_forum.get(forum_key, {})
3065
+ merged = dict(prior)
3066
+ for field, value in row.items():
3067
+ if field == "posts":
3068
+ continue
3069
+ if field in {"name", "description"} and not value and prior.get(field):
3070
+ continue
3071
+ merged[field] = value
3072
+ old_posts = prior.get("posts", [])
3073
+ new_posts = row.get("posts", [])
3074
+ merged["posts"] = _merge_rows(
3075
+ old_posts if isinstance(old_posts, list) else [],
3076
+ new_posts if isinstance(new_posts, list) else [],
3077
+ id_field="id",
3078
+ complete=True,
3079
+ )
3080
+ merged["missing_since"] = None
3081
+ merged["withdrawn_at"] = None
3082
+ by_forum[forum_key] = merged
3083
+
3084
+ now = _now()
3085
+ for forum_key, row in by_forum.items():
3086
+ if forum_key in incoming_forums:
3087
+ continue
3088
+ if row.get("missing_since") is None:
3089
+ row["missing_since"] = now
3090
+ elif row.get("withdrawn_at") is None:
3091
+ row["withdrawn_at"] = now
3092
+ return sorted(by_forum.values(), key=lambda row: str(row.get("id")))
3093
+
3094
+
3095
+ def _discussion_forum_is_valid(forum: Mapping[str, object]) -> bool:
3096
+ """Validate nested discussion containers before replacing a prior capture."""
3097
+ raw_topics = forum.get("Topics", forum.get("topics", []))
3098
+ if "Topics" in forum or "topics" in forum:
3099
+ if isinstance(raw_topics, dict):
3100
+ raw_topics = raw_topics.get("Items", raw_topics.get("Objects"))
3101
+ if not isinstance(raw_topics, list) or any(
3102
+ not isinstance(topic, dict) for topic in raw_topics
3103
+ ):
3104
+ return False
3105
+ for topic in raw_topics:
3106
+ if "Posts" not in topic and "posts" not in topic:
3107
+ return False
3108
+ raw_posts = topic.get("Posts", topic.get("posts", []))
3109
+ if not isinstance(raw_posts, list) or any(
3110
+ not isinstance(post, dict) or not _discussion_post_is_valid(post)
3111
+ for post in raw_posts
3112
+ ):
3113
+ return False
3114
+
3115
+ direct_posts = forum.get("Posts", forum.get("posts", []))
3116
+ if ("Posts" in forum or "posts" in forum) and (
3117
+ not isinstance(direct_posts, list)
3118
+ or any(
3119
+ not isinstance(post, dict) or not _discussion_post_is_valid(post)
3120
+ for post in direct_posts
3121
+ )
3122
+ ):
3123
+ return False
3124
+ posts = _discussion_posts(forum)
3125
+ post_ids = [str(_discussion_post_id(post)) for post in posts]
3126
+ return len(set(post_ids)) == len(post_ids)
3127
+
3128
+
3129
+ def _discussion_post_is_valid(post: Mapping[str, object]) -> bool:
3130
+ identifier = _discussion_post_id(post)
3131
+ return identifier is not None and (not isinstance(identifier, str) or bool(identifier))
3132
+
3133
+
3134
+ def _discussion_identity(post: Mapping[str, object]) -> str | None:
3135
+ author = post.get("Author", post.get("author", post.get("User", post.get("user"))))
3136
+ if isinstance(author, dict):
3137
+ for key in ("Identifier", "UserId", "AuthorId", "Id", "id"):
3138
+ value = author.get(key)
3139
+ if isinstance(value, (str, int)) and not isinstance(value, bool) and str(value):
3140
+ return f"id:{value}"
3141
+ for key in ("Identifier", "UserId", "AuthorId"):
3142
+ value = post.get(key)
3143
+ if isinstance(value, (str, int)) and not isinstance(value, bool) and str(value):
3144
+ return f"id:{value}"
3145
+ candidates: list[object] = [author, post]
3146
+ for candidate in candidates:
3147
+ if not isinstance(candidate, dict):
3148
+ continue
3149
+ for key in ("DisplayName", "Name", "UserName", "name"):
3150
+ value = candidate.get(key)
3151
+ if isinstance(value, str) and value.strip():
3152
+ normalized = unicodedata.normalize("NFKC", value)
3153
+ normalized = re.sub(r"\s+", " ", normalized).strip().casefold()
3154
+ return f"name:{normalized}"
3155
+ return None
3156
+
3157
+
3158
+ def _discussion_pseudonyms(key: bytes, identities: Iterable[str]) -> dict[str, str]:
3159
+ grouped: dict[str, list[str]] = defaultdict(list)
3160
+ for identity in identities:
3161
+ digest = hmac.new(key, identity.encode("utf-8"), "sha256").hexdigest()
3162
+ grouped[digest[:20]].append(identity)
3163
+ result: dict[str, str] = {}
3164
+ for digest, values in grouped.items():
3165
+ for identity in sorted(values):
3166
+ suffix = ""
3167
+ if len(values) > 1:
3168
+ suffix = "-" + sha256(identity.encode("utf-8")).hexdigest()[:16]
3169
+ result[identity] = f"author-{digest}{suffix}"
3170
+ return result
3171
+
3172
+
3173
+ def _discussion_author(
3174
+ post: Mapping[str, object],
3175
+ identity: str | None,
3176
+ pseudonyms: Mapping[str, str],
3177
+ include_authors: bool,
3178
+ ) -> str:
3179
+ if include_authors:
3180
+ author = post.get("Author", post.get("author", post.get("User", post.get("user"))))
3181
+ if isinstance(author, dict):
3182
+ return (
3183
+ _first_string(author, "DisplayName", "Name", "UserName")
3184
+ or _first_string(post, "DisplayName", "Name")
3185
+ or "unknown-author"
3186
+ )
3187
+ return pseudonyms.get(identity or "", "author-unknown")
3188
+
3189
+
3190
+ def _discussion_post_id(post: Mapping[str, object]) -> str | int | None:
3191
+ for key in ("PostId", "Id", "id"):
3192
+ value = post.get(key)
3193
+ if isinstance(value, (str, int)) and not isinstance(value, bool):
3194
+ return value
3195
+ return None
3196
+
3197
+
3198
+ def _discussion_body_text(value: object, *, school_base: str) -> str:
3199
+ if isinstance(value, dict):
3200
+ html_value = _first_string(value, "Html", "HTML", "html")
3201
+ text_value = _first_string(value, "Text", "text")
3202
+ value = html_value if html_value is not None else text_value
3203
+ if not isinstance(value, str):
3204
+ return ""
3205
+ sanitized = _sanitize_richtext(value, school_base)
3206
+ return _richtext_markdown("", sanitized).split("\n\n", 1)[-1].strip()
3207
+
3208
+
3209
+ def _update_course_index_from_map(course_dir: Path) -> None:
3210
+ del course_dir
3211
+
3212
+
3213
+ __all__ = [
3214
+ "CourseMetadata",
3215
+ "FetchReport",
3216
+ "FileReport",
3217
+ "MetadataReport",
3218
+ "OutlineReport",
3219
+ "PRIORITY_BUDGET_BYTES",
3220
+ "TopicRecord",
3221
+ "fetch_topic",
3222
+ "is_downloadable_topic",
3223
+ "is_media_topic",
3224
+ "ingest_files",
3225
+ "ingest_metadata",
3226
+ "load_metadata_report",
3227
+ "load_metadata_topics",
3228
+ "select_priority_topics",
3229
+ ]