agent2learn 0.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. agent2learn/__init__.py +3 -0
  2. agent2learn/_release.py +19 -0
  3. agent2learn/aipolicy.py +182 -0
  4. agent2learn/api.py +590 -0
  5. agent2learn/audit.py +358 -0
  6. agent2learn/auth/__init__.py +282 -0
  7. agent2learn/auth/cdp.py +1067 -0
  8. agent2learn/auth/paste.py +378 -0
  9. agent2learn/calendar.py +525 -0
  10. agent2learn/calibrate.py +347 -0
  11. agent2learn/check.py +1091 -0
  12. agent2learn/cli.py +2039 -0
  13. agent2learn/clock.py +39 -0
  14. agent2learn/config.py +205 -0
  15. agent2learn/console.py +229 -0
  16. agent2learn/convert.py +1223 -0
  17. agent2learn/doctor.py +1167 -0
  18. agent2learn/errors.py +32 -0
  19. agent2learn/ground.py +735 -0
  20. agent2learn/index.py +614 -0
  21. agent2learn/ingest.py +3229 -0
  22. agent2learn/locations.py +247 -0
  23. agent2learn/outlines.py +754 -0
  24. agent2learn/paths.py +683 -0
  25. agent2learn/pipeline.py +392 -0
  26. agent2learn/privacy.py +1123 -0
  27. agent2learn/schools/__init__.py +29 -0
  28. agent2learn/schools/_base.py +194 -0
  29. agent2learn/schools/generic.py +78 -0
  30. agent2learn/schools/uwaterloo.py +66 -0
  31. agent2learn/session.py +373 -0
  32. agent2learn/skills.py +1081 -0
  33. agent2learn/snapshot.py +399 -0
  34. agent2learn/submit.py +1047 -0
  35. agent2learn/transactions.py +157 -0
  36. agent2learn/upgrade.py +288 -0
  37. agent2learn/vault.py +1134 -0
  38. agent2learn-0.1.2.data/data/a2l-coursework/SKILL.md +52 -0
  39. agent2learn-0.1.2.data/data/a2l-setup/SKILL.md +27 -0
  40. agent2learn-0.1.2.data/data/a2l-study/SKILL.md +27 -0
  41. agent2learn-0.1.2.data/data/a2l-sync/SKILL.md +30 -0
  42. agent2learn-0.1.2.dist-info/METADATA +186 -0
  43. agent2learn-0.1.2.dist-info/RECORD +46 -0
  44. agent2learn-0.1.2.dist-info/WHEEL +4 -0
  45. agent2learn-0.1.2.dist-info/entry_points.txt +3 -0
  46. agent2learn-0.1.2.dist-info/licenses/LICENSE +202 -0
agent2learn/convert.py ADDED
@@ -0,0 +1,1223 @@
1
+ """Local, deterministic conversion of captured course sources into Markdown twins.
2
+
3
+ Conversion is deliberately downstream of ingestion. It receives only a local source path, never
4
+ an API client or session, and it writes nothing until the caller has accepted the complete result.
5
+ PDFs use the pinned PDF Oxide backend by default; notebooks and HTML archives are handled by small
6
+ auditable renderers rather than an execution-capable exporter stack.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import base64
12
+ import binascii
13
+ import importlib
14
+ import importlib.metadata
15
+ import io
16
+ import json
17
+ import os
18
+ import re
19
+ import shutil
20
+ import zipfile
21
+ from collections.abc import Callable, Collection, Mapping, Sequence
22
+ from dataclasses import dataclass, replace
23
+ from html import escape
24
+ from html.parser import HTMLParser
25
+ from pathlib import Path, PurePosixPath
26
+ from typing import Protocol, cast
27
+ from urllib.parse import urlsplit, urlunsplit
28
+
29
+ import pypdfium2 as pdfium # type: ignore[import-untyped]
30
+ import pytesseract # type: ignore[import-untyped]
31
+ from pdf_oxide import PdfDocument
32
+ from PIL import Image
33
+
34
+ from agent2learn import clock, paths
35
+ from agent2learn import index as course_index
36
+ from agent2learn.errors import A2LError
37
+ from agent2learn.vault import DerivedArtifact, ManifestEntry, Vault
38
+
39
+ DEFAULT_OCR_WORDS_PER_PAGE = 80
40
+ MIN_PDF_CHARS = 200
41
+ OCR_DPI = 300
42
+ CONVERTER_VERSION = "1"
43
+ RICH_TEXT_TOOL = "richtext-sanitizer"
44
+ RICH_TEXT_TOOL_VERSION = "1"
45
+ OCR_SETUP_ACTION = "install Tesseract with the 'eng' language pack, then rerun: a2l sync"
46
+ MAX_ZIP_MEMBERS = 1_000
47
+ MAX_ZIP_UNCOMPRESSED = 64 * 1024 * 1024
48
+ MAX_ZIP_MEMBER = 32 * 1024 * 1024
49
+ MAX_ZIP_COMPRESSION_RATIO = 1_000
50
+
51
+ _ANSI = re.compile(r"\x1b(?:\[[0-?]*[ -/]*[@-~]|\][^\x07]*(?:\x07|\x1b\\))")
52
+ _BACKTICKS = re.compile(r"`+")
53
+ _ATTACHMENT = re.compile(r"attachment:([^\s)\"'>]+)", re.IGNORECASE)
54
+ _WINDOWS_ABSOLUTE = re.compile(r"^[A-Za-z]:[\\/]")
55
+ _FORBIDDEN_ARCHIVE_PARTS = frozenset({"", ".", ".."})
56
+ _SAFE_IMAGE_MIME = frozenset({"image/gif", "image/jpeg", "image/png", "image/webp"})
57
+ _HTML_SKIP = frozenset({"script", "style", "form", "iframe", "object", "embed", "template"})
58
+ _BLOCK_TAGS = frozenset(
59
+ {
60
+ "address",
61
+ "article",
62
+ "aside",
63
+ "blockquote",
64
+ "div",
65
+ "dl",
66
+ "dt",
67
+ "dd",
68
+ "figure",
69
+ "footer",
70
+ "h1",
71
+ "h2",
72
+ "h3",
73
+ "h4",
74
+ "h5",
75
+ "h6",
76
+ "header",
77
+ "hr",
78
+ "li",
79
+ "main",
80
+ "nav",
81
+ "ol",
82
+ "p",
83
+ "pre",
84
+ "section",
85
+ "table",
86
+ "tr",
87
+ "ul",
88
+ }
89
+ )
90
+
91
+
92
+ class ConversionError(A2LError):
93
+ """A source could not be converted by the selected backend."""
94
+
95
+
96
+ @dataclass(frozen=True)
97
+ class PageCoverage:
98
+ """Deterministic coverage information for one one-based source page."""
99
+
100
+ page: int
101
+ mode: str
102
+ words: int
103
+ warning: str | None = None
104
+
105
+ @property
106
+ def word_count(self) -> int:
107
+ """Alias used by audit/report consumers that spell out the measurement."""
108
+
109
+ return self.words
110
+
111
+
112
+ @dataclass(frozen=True)
113
+ class ConversionResult:
114
+ """Pure conversion output; no filesystem installation occurs here."""
115
+
116
+ markdown: str
117
+ page_coverage: tuple[PageCoverage, ...] = ()
118
+ warnings: tuple[str, ...] = ()
119
+ backend: str = "agent2learn"
120
+ tool_version: str = CONVERTER_VERSION
121
+ gap: bool = False
122
+
123
+ @property
124
+ def coverage(self) -> tuple[PageCoverage, ...]:
125
+ """Short alias for callers rendering an audit summary."""
126
+
127
+ return self.page_coverage
128
+
129
+
130
+ @dataclass(frozen=True)
131
+ class ConversionReport:
132
+ """Summary of a vault conversion pass."""
133
+
134
+ converted: int = 0
135
+ skipped: int = 0
136
+ gaps: int = 0
137
+ warnings: tuple[str, ...] = ()
138
+ errors: tuple[str, ...] = ()
139
+
140
+
141
+ class ConverterBackend(Protocol):
142
+ """The narrow PDF backend boundary shared by the default and degraded implementations."""
143
+
144
+ name: str
145
+ version: str
146
+
147
+ def convert_pdf(self, source: Path, *, ocr_words_per_page: int) -> ConversionResult:
148
+ """Convert one local PDF without accessing network or session state."""
149
+
150
+
151
+ class PdfOxideBackend:
152
+ """The pinned PDF Oxide backend with explicit external-Tesseract OCR."""
153
+
154
+ name = "pdf-oxide"
155
+
156
+ def __init__(
157
+ self,
158
+ *,
159
+ document_factory: Callable[[Path], object] | None = None,
160
+ ocr_reader: Callable[[bytes], str] | None = None,
161
+ ocr_language: str = "eng",
162
+ dpi: int = OCR_DPI,
163
+ ) -> None:
164
+ if isinstance(dpi, bool) or not isinstance(dpi, int) or dpi <= 0:
165
+ raise ValueError("dpi must be a positive integer")
166
+ if not isinstance(ocr_language, str) or not ocr_language:
167
+ raise ValueError("ocr_language must be a non-empty string")
168
+ self.version = _package_version("pdf-oxide", "unknown")
169
+ self._document_factory = document_factory or self._open_document
170
+ self._ocr_reader = ocr_reader
171
+ self._ocr_language = ocr_language
172
+ self._dpi = dpi
173
+
174
+ @staticmethod
175
+ def _open_document(source: Path) -> object:
176
+ return PdfDocument(os.fspath(paths.long_path(source)))
177
+
178
+ def convert_pdf(self, source: Path, *, ocr_words_per_page: int) -> ConversionResult:
179
+ _validate_threshold(ocr_words_per_page)
180
+ document = self._document_factory(Path(source))
181
+ try:
182
+ page_count = _pdf_oxide_page_count(document)
183
+ if page_count < 1:
184
+ raise ConversionError("PDF has no pages")
185
+ extracted = [
186
+ _text_value(_call_method(document, "extract_text_auto", page))
187
+ for page in range(page_count)
188
+ ]
189
+ document_has_text = sum(len(text) for text in extracted) >= MIN_PDF_CHARS
190
+ healthy = [
191
+ document_has_text and len(text.split()) >= ocr_words_per_page for text in extracted
192
+ ]
193
+ page_markdown: list[str] = []
194
+ coverage: list[PageCoverage] = []
195
+ warnings: list[str] = []
196
+
197
+ all_markdown: list[str] | None = None
198
+ if all(healthy) and hasattr(document, "to_markdown_all"):
199
+ try:
200
+ all_value = _call_method(document, "to_markdown_all")
201
+ all_markdown = _split_all_markdown(_text_value(all_value), page_count)
202
+ except Exception as exc:
203
+ warnings.append(f"whole-document Markdown unavailable: {type(exc).__name__}")
204
+ else:
205
+ if all_markdown is None:
206
+ warnings.append("whole-document Markdown had no stable page split")
207
+
208
+ for page, text in enumerate(extracted):
209
+ if healthy[page]:
210
+ if all_markdown is not None:
211
+ markdown = all_markdown[page]
212
+ else:
213
+ markdown = _text_value(_call_method(document, "to_markdown", page))
214
+ mode = "markdown"
215
+ warning = None
216
+ words = len(text.split())
217
+ else:
218
+ try:
219
+ image_bytes = _text_image_bytes(
220
+ _call_method(document, "render_page", page, dpi=self._dpi)
221
+ )
222
+ ocr_text = self._read_ocr(image_bytes)
223
+ except (ConversionError, OSError, RuntimeError) as exc:
224
+ warning = f"OCR unavailable on page {page + 1}: {type(exc).__name__}"
225
+ warnings.append(warning)
226
+ markdown = f"[a2l conversion gap: {warning}]"
227
+ mode = "unresolved"
228
+ words = 0
229
+ else:
230
+ markdown = ocr_text
231
+ mode = "ocr"
232
+ warning = None
233
+ words = len(ocr_text.split())
234
+ page_markdown.append(_page_block(page + 1, markdown))
235
+ coverage.append(PageCoverage(page + 1, mode, words, warning))
236
+
237
+ return ConversionResult(
238
+ markdown=_normalise_markdown("\n\n".join(page_markdown)),
239
+ page_coverage=tuple(coverage),
240
+ warnings=tuple(warnings),
241
+ backend=self.name,
242
+ tool_version=self.version,
243
+ gap=any(page.mode == "unresolved" for page in coverage),
244
+ )
245
+ except ConversionError:
246
+ raise
247
+ except Exception as exc:
248
+ raise ConversionError(f"pdf-oxide could not convert {source.name}") from exc
249
+ finally:
250
+ _close_quietly(document)
251
+
252
+ def _read_ocr(self, image_bytes: bytes) -> str:
253
+ if self._ocr_reader is not None:
254
+ normalized = _normalise_markdown(self._ocr_reader(image_bytes))
255
+ if not normalized:
256
+ raise ConversionError("OCR returned no text")
257
+ return normalized
258
+ if not _configure_tesseract(self._ocr_language):
259
+ raise ConversionError("Tesseract is unavailable or lacks the requested language")
260
+ try:
261
+ with Image.open(io.BytesIO(image_bytes)) as image:
262
+ text = pytesseract.image_to_string(image, lang=self._ocr_language)
263
+ except pytesseract.TesseractError as exc:
264
+ raise ConversionError("Tesseract could not OCR the rendered page") from exc
265
+ normalized = _normalise_markdown(text)
266
+ if not normalized:
267
+ raise ConversionError("Tesseract returned no text")
268
+ return normalized
269
+
270
+
271
+ class PdfiumBackend:
272
+ """Named degraded PDFium fallback; it never silently replaces a successful default result."""
273
+
274
+ name = "pypdfium2"
275
+
276
+ def __init__(self) -> None:
277
+ self.version = _package_version("pypdfium2", "unknown")
278
+
279
+ def convert_pdf(self, source: Path, *, ocr_words_per_page: int) -> ConversionResult:
280
+ _validate_threshold(ocr_words_per_page)
281
+ try:
282
+ document = pdfium.PdfDocument(os.fspath(paths.long_path(source)))
283
+ except Exception as exc:
284
+ raise ConversionError(f"pypdfium2 could not open {source.name}") from exc
285
+
286
+ blocks: list[str] = []
287
+ coverage: list[PageCoverage] = []
288
+ try:
289
+ for index in range(len(document)):
290
+ page = document[index]
291
+ textpage = page.get_textpage()
292
+ try:
293
+ text = _normalise_markdown(textpage.get_text_bounded())
294
+ finally:
295
+ _close_quietly(textpage)
296
+ _close_quietly(page)
297
+ blocks.append(_page_block(index + 1, text))
298
+ coverage.append(PageCoverage(index + 1, "fallback", len(text.split())))
299
+ if not coverage:
300
+ raise ConversionError("PDF has no pages")
301
+ empty_page = any(page.words == 0 for page in coverage)
302
+ return ConversionResult(
303
+ markdown=_normalise_markdown("\n\n".join(blocks)),
304
+ page_coverage=tuple(coverage),
305
+ backend=self.name,
306
+ tool_version=self.version,
307
+ warnings=("fallback contains an empty page",) if empty_page else (),
308
+ gap=empty_page,
309
+ )
310
+ except ConversionError:
311
+ raise
312
+ except Exception as exc:
313
+ raise ConversionError(f"pypdfium2 could not convert {source.name}") from exc
314
+ finally:
315
+ _close_quietly(document)
316
+
317
+
318
+ def convert_pdf(
319
+ source: Path,
320
+ *,
321
+ backend: ConverterBackend | None = None,
322
+ fallback: ConverterBackend | None = None,
323
+ ocr_words_per_page: int = DEFAULT_OCR_WORDS_PER_PAGE,
324
+ ) -> ConversionResult:
325
+ """Use the default PDF backend, falling back only when it cannot convert at all."""
326
+
327
+ selected = backend or PdfOxideBackend()
328
+ degraded = fallback or PdfiumBackend()
329
+ try:
330
+ return selected.convert_pdf(source, ocr_words_per_page=ocr_words_per_page)
331
+ except Exception:
332
+ try:
333
+ result = degraded.convert_pdf(source, ocr_words_per_page=ocr_words_per_page)
334
+ except Exception as fallback_error:
335
+ raise ConversionError("both PDF backends failed") from fallback_error
336
+ warning = f"default PDF backend failed; accepted {degraded.name} fallback"
337
+ return replace(result, warnings=(warning, *result.warnings))
338
+
339
+
340
+ def convert_source(
341
+ source: Path,
342
+ *,
343
+ backend: ConverterBackend | None = None,
344
+ fallback: ConverterBackend | None = None,
345
+ ocr_words_per_page: int = DEFAULT_OCR_WORDS_PER_PAGE,
346
+ content_type: str | None = None,
347
+ ) -> ConversionResult:
348
+ """Convert one local source based on magic bytes plus extension, never executing it."""
349
+
350
+ del content_type # Server metadata is advisory; local bytes and the extension decide dispatch.
351
+ source = Path(source)
352
+ if not paths.long_path(source).is_file():
353
+ raise ConversionError(f"source is not a regular file: {source.name}")
354
+ kind = _classify_source(source)
355
+ if kind == "pdf":
356
+ return convert_pdf(
357
+ source,
358
+ backend=backend,
359
+ fallback=fallback,
360
+ ocr_words_per_page=ocr_words_per_page,
361
+ )
362
+ if kind == "notebook":
363
+ return render_notebook(source)
364
+ if kind == "html_zip":
365
+ return convert_html_zip(source)
366
+ if kind == "html":
367
+ return _convert_html_file(source)
368
+ if kind == "text":
369
+ try:
370
+ with open(os.fspath(paths.long_path(source)), encoding="utf-8", newline="") as handle:
371
+ text = handle.read()
372
+ except (OSError, UnicodeError) as exc:
373
+ raise ConversionError(f"text source could not be read: {source.name}") from exc
374
+ return ConversionResult(
375
+ markdown=_normalise_markdown(text),
376
+ page_coverage=(PageCoverage(1, "text", len(text.split())),),
377
+ backend="agent2learn-text",
378
+ tool_version=CONVERTER_VERSION,
379
+ )
380
+ if kind == "office":
381
+ return _convert_office(source)
382
+ return ConversionResult(
383
+ markdown="",
384
+ warnings=(f"unsupported or mismatched source format: {source.suffix or source.name}",),
385
+ backend="agent2learn",
386
+ tool_version=CONVERTER_VERSION,
387
+ gap=True,
388
+ )
389
+
390
+
391
+ def render_notebook(source: Path) -> ConversionResult:
392
+ """Render a v4 notebook without importing or executing any cell code."""
393
+
394
+ try:
395
+ nbformat = importlib.import_module("nbformat")
396
+ except ImportError:
397
+ return ConversionResult(
398
+ markdown="",
399
+ warnings=("optional notebook dependency nbformat is not installed",),
400
+ backend="nbformat",
401
+ tool_version="missing",
402
+ gap=True,
403
+ )
404
+ try:
405
+ notebook = nbformat.read(os.fspath(paths.long_path(source)), as_version=4)
406
+ except Exception as exc:
407
+ raise ConversionError(f"notebook could not be parsed: {source.name}") from exc
408
+
409
+ metadata = notebook.get("metadata", {})
410
+ language = "text"
411
+ if isinstance(metadata, Mapping):
412
+ language_info = metadata.get("language_info", {})
413
+ if isinstance(language_info, Mapping) and isinstance(language_info.get("name"), str):
414
+ language = cast(str, language_info["name"]).strip() or "text"
415
+
416
+ blocks: list[str] = []
417
+ warnings: list[str] = []
418
+ cells = notebook.get("cells", [])
419
+ if not isinstance(cells, Sequence) or isinstance(cells, (str, bytes)):
420
+ raise ConversionError("notebook cells must be an array")
421
+ for number, cell in enumerate(cells, start=1):
422
+ if not isinstance(cell, Mapping):
423
+ warnings.append(f"cell {number} is not an object")
424
+ continue
425
+ cell_type = cell.get("cell_type")
426
+ source_text = _cell_text(cell.get("source"))
427
+ if cell_type == "markdown":
428
+ attachments = cell.get("attachments", {})
429
+ blocks.append(_replace_attachments(source_text, attachments))
430
+ elif cell_type == "code":
431
+ blocks.append(_fenced(source_text, language))
432
+ outputs = cell.get("outputs", [])
433
+ if isinstance(outputs, Sequence) and not isinstance(outputs, (str, bytes)):
434
+ for output in outputs:
435
+ rendered, output_warning = _render_notebook_output(output)
436
+ if rendered:
437
+ blocks.append(rendered)
438
+ if output_warning is not None:
439
+ warnings.append(f"cell {number}: {output_warning}")
440
+ else:
441
+ warnings.append(f"cell {number}: unsupported cell type {cell_type!r}")
442
+
443
+ markdown = _normalise_markdown("\n\n".join(blocks))
444
+ return ConversionResult(
445
+ markdown=markdown,
446
+ page_coverage=(PageCoverage(1, "notebook", len(markdown.split())),),
447
+ warnings=tuple(warnings),
448
+ backend="nbformat",
449
+ tool_version=_package_version("nbformat", "unknown"),
450
+ gap=False,
451
+ )
452
+
453
+
454
+ def convert_html_zip(source: Path) -> ConversionResult:
455
+ """Extract only a safe main HTML member from a bounded archive in memory."""
456
+
457
+ try:
458
+ with zipfile.ZipFile(os.fspath(paths.long_path(source))) as archive:
459
+ infos = archive.infolist()
460
+ html_info = _validate_archive(infos)
461
+ if html_info is None:
462
+ return ConversionResult(
463
+ markdown="",
464
+ warnings=("HTML archive has no main HTML member",),
465
+ backend="html-archive",
466
+ tool_version=CONVERTER_VERSION,
467
+ gap=True,
468
+ )
469
+ html_bytes = archive.read(html_info)
470
+ except (OSError, ValueError, zipfile.BadZipFile, zipfile.LargeZipFile) as exc:
471
+ raise ConversionError(f"HTML archive rejected: {source.name}") from exc
472
+ try:
473
+ text = html_bytes.decode("utf-8")
474
+ except UnicodeDecodeError as exc:
475
+ raise ConversionError("HTML archive main member is not UTF-8") from exc
476
+ return _html_result(text)
477
+
478
+
479
+ def convert_vault(
480
+ vault: Vault,
481
+ *,
482
+ backend: ConverterBackend | None = None,
483
+ fallback: ConverterBackend | None = None,
484
+ ocr_words_per_page: int = DEFAULT_OCR_WORDS_PER_PAGE,
485
+ source_keys: Collection[str] | None = None,
486
+ ) -> ConversionReport:
487
+ """Install current, hash-linked twins for manifest sources without losing revisions."""
488
+
489
+ _validate_threshold(ocr_words_per_page)
490
+ entries = vault.manifest()
491
+ if source_keys is not None:
492
+ selected = set(source_keys)
493
+ if selected - entries.keys():
494
+ raise A2LError("conversion source is not recorded in the manifest")
495
+ entries = {key: entry for key, entry in entries.items() if key in selected}
496
+ selected_backend = backend or PdfOxideBackend()
497
+ selected_fallback = fallback or PdfiumBackend()
498
+ converted = skipped = gaps = 0
499
+ warnings: list[str] = []
500
+ errors: list[str] = []
501
+ for key, entry in sorted(entries.items()):
502
+ source = vault.materialized(entry)
503
+ if not paths.long_path(source).is_file():
504
+ gaps += 1
505
+ message = f"{key}: source is missing"
506
+ errors.append(message)
507
+ _update_content_map(vault, key, availability="integrity_gap", next_action=message)
508
+ continue
509
+ source_hash, source_size = _hash_file(source)
510
+ if source_hash != entry.sha256 or source_size != entry.size:
511
+ gaps += 1
512
+ message = f"{key}: source hash does not match manifest"
513
+ errors.append(message)
514
+ _update_content_map(
515
+ vault,
516
+ key,
517
+ availability="integrity_gap",
518
+ source_path=entry.path,
519
+ path=None,
520
+ sha256=entry.sha256,
521
+ source_sha256=entry.sha256,
522
+ size=entry.size,
523
+ next_action=message,
524
+ )
525
+ continue
526
+
527
+ artifact = entry.derived.get("markdown")
528
+ try:
529
+ source_kind = _classify_source(source)
530
+ expected_tool, expected_version = _expected_tool_for_kind(source_kind, selected_backend)
531
+ except Exception as exc:
532
+ gaps += 1
533
+ message = f"{key}: {type(exc).__name__}"
534
+ errors.append(message)
535
+ _update_content_map(
536
+ vault,
537
+ key,
538
+ availability="conversion_gap",
539
+ source_path=entry.path,
540
+ path=None,
541
+ sha256=entry.sha256,
542
+ source_sha256=entry.sha256,
543
+ size=entry.size,
544
+ next_action=message,
545
+ )
546
+ continue
547
+ expected_threshold = ocr_words_per_page if source_kind == "pdf" else None
548
+ if (
549
+ artifact is not None
550
+ and vault.owns_derived_path(key, artifact.path)
551
+ and _artifact_is_current(
552
+ vault,
553
+ artifact,
554
+ entry,
555
+ expected_tool,
556
+ expected_version,
557
+ expected_threshold,
558
+ )
559
+ ):
560
+ skipped += 1
561
+ continue
562
+
563
+ try:
564
+ result = convert_source(
565
+ source,
566
+ backend=selected_backend,
567
+ fallback=selected_fallback,
568
+ ocr_words_per_page=ocr_words_per_page,
569
+ )
570
+ except Exception as exc:
571
+ gaps += 1
572
+ message = f"{key}: {type(exc).__name__}"
573
+ errors.append(message)
574
+ _update_content_map(
575
+ vault,
576
+ key,
577
+ availability="conversion_gap",
578
+ source_path=entry.path,
579
+ path=None,
580
+ sha256=entry.sha256,
581
+ source_sha256=entry.sha256,
582
+ size=entry.size,
583
+ next_action=message,
584
+ )
585
+ continue
586
+ warnings.extend(f"{key}: {warning}" for warning in result.warnings)
587
+ if result.gap or not result.markdown:
588
+ gaps += 1
589
+ lowered = tuple(warning.casefold() for warning in result.warnings)
590
+ availability = (
591
+ "unsupported_format"
592
+ if any("unsupported" in warning or "optional" in warning for warning in lowered)
593
+ else "conversion_gap"
594
+ )
595
+ if any("ocr unavailable" in warning for warning in lowered):
596
+ next_action = OCR_SETUP_ACTION
597
+ else:
598
+ next_action = result.warnings[0] if result.warnings else "conversion gap"
599
+ _update_content_map(
600
+ vault,
601
+ key,
602
+ availability=availability,
603
+ source_path=entry.path,
604
+ path=None,
605
+ sha256=entry.sha256,
606
+ source_sha256=entry.sha256,
607
+ size=entry.size,
608
+ next_action=next_action,
609
+ )
610
+ continue
611
+
612
+ destination = vault.derived_destination(key, source.with_suffix(".md"))
613
+ prior_artifact = entry.derived.get("markdown")
614
+ local_modification = False
615
+ if prior_artifact is not None:
616
+ prior_path = _artifact_path(vault, prior_artifact)
617
+ if paths.long_path(prior_path).is_file() and _hash_file(prior_path)[0] != (
618
+ prior_artifact.sha256
619
+ ):
620
+ vault.preserve_revision(key, changed_at=clock.now())
621
+ local_modification = True
622
+ paths.ensure_dir(destination.parent, root=vault.root)
623
+ paths.atomic_write_text(destination, result.markdown, root=vault.root)
624
+ derived = DerivedArtifact(
625
+ path=paths.rel_posix(destination, vault.root),
626
+ sha256=_hash_file(destination)[0],
627
+ source_sha256=entry.sha256,
628
+ tool=result.backend,
629
+ tool_version=result.tool_version,
630
+ created_at=_now(),
631
+ ocr_words_per_page=(ocr_words_per_page if source_kind == "pdf" else None),
632
+ page_coverage=tuple(
633
+ {
634
+ "page": page.page,
635
+ "mode": page.mode,
636
+ "words": page.words,
637
+ "warning": page.warning,
638
+ }
639
+ for page in result.page_coverage
640
+ ),
641
+ )
642
+ updated = replace(entry, derived={"markdown": derived})
643
+ vault.mark(key, updated)
644
+ vault.save_manifest()
645
+ _update_content_map(
646
+ vault,
647
+ key,
648
+ availability="markdown_ready",
649
+ source_path=updated.path,
650
+ path=derived.path,
651
+ sha256=updated.sha256,
652
+ source_sha256=updated.sha256,
653
+ size=updated.size,
654
+ next_action="ready for citation",
655
+ )
656
+ if local_modification:
657
+ warnings.append(f"{key}: preserved locally modified markdown twin")
658
+ converted += 1
659
+ return ConversionReport(converted, skipped, gaps, tuple(warnings), tuple(errors))
660
+
661
+
662
+ def _convert_html_file(source: Path) -> ConversionResult:
663
+ try:
664
+ with open(os.fspath(paths.long_path(source)), encoding="utf-8", newline="") as handle:
665
+ text = handle.read()
666
+ except (OSError, UnicodeError) as exc:
667
+ raise ConversionError(f"HTML source could not be read: {source.name}") from exc
668
+ return _html_result(text)
669
+
670
+
671
+ def _html_result(text: str) -> ConversionResult:
672
+ parser = _SafeHtmlMarkdown()
673
+ parser.feed(text)
674
+ parser.close()
675
+ markdown = _normalise_markdown(parser.markdown())
676
+ return ConversionResult(
677
+ markdown=markdown,
678
+ page_coverage=(PageCoverage(1, "html", len(markdown.split())),),
679
+ warnings=(),
680
+ backend="html-sanitizer",
681
+ tool_version=CONVERTER_VERSION,
682
+ )
683
+
684
+
685
+ def _convert_office(source: Path) -> ConversionResult:
686
+ try:
687
+ module = importlib.import_module("markitdown")
688
+ converter_type = module.MarkItDown
689
+ converted = converter_type().convert(os.fspath(paths.long_path(source)))
690
+ text = getattr(converted, "text_content", None)
691
+ if not isinstance(text, str):
692
+ raise ConversionError("MarkItDown returned no text content")
693
+ except ImportError:
694
+ return ConversionResult(
695
+ markdown="",
696
+ warnings=("optional office dependency markitdown is not installed",),
697
+ backend="markitdown",
698
+ tool_version="missing",
699
+ gap=True,
700
+ )
701
+ except Exception as exc:
702
+ return ConversionResult(
703
+ markdown="",
704
+ warnings=(f"office conversion failed: {type(exc).__name__}",),
705
+ backend="markitdown",
706
+ tool_version=_package_version("markitdown", "unknown"),
707
+ gap=True,
708
+ )
709
+ markdown = _normalise_markdown(text)
710
+ return ConversionResult(
711
+ markdown=markdown,
712
+ page_coverage=(PageCoverage(1, "office", len(markdown.split())),),
713
+ backend="markitdown",
714
+ tool_version=_package_version("markitdown", "unknown"),
715
+ )
716
+
717
+
718
+ def _classify_source(source: Path) -> str:
719
+ suffixes = [suffix.casefold() for suffix in source.suffixes]
720
+ suffix = source.suffix.casefold()
721
+ try:
722
+ with open(os.fspath(paths.long_path(source)), "rb") as handle:
723
+ magic = handle.read(16)
724
+ except OSError as exc:
725
+ raise ConversionError(f"source could not be inspected: {source.name}") from exc
726
+ if magic.startswith(b"%PDF"):
727
+ return "pdf"
728
+ if suffix == ".pdf":
729
+ return "unsupported"
730
+ if suffix == ".ipynb" or (magic.lstrip().startswith(b"{") and _looks_like_notebook(source)):
731
+ return "notebook"
732
+ if suffix in {".docx", ".pptx", ".xlsx"}:
733
+ return "office" if _looks_like_office(source, suffix) else "unsupported"
734
+ if suffix in {".doc", ".ppt", ".xls"}:
735
+ return "office" if magic.startswith(b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1") else "unsupported"
736
+ archive_path = os.fspath(paths.long_path(source))
737
+ if zipfile.is_zipfile(archive_path):
738
+ if suffix == ".zip" and ".html" in suffixes:
739
+ return "html_zip"
740
+ try:
741
+ with zipfile.ZipFile(archive_path) as archive:
742
+ if any(_is_html_name(info.filename) for info in archive.infolist()):
743
+ return "html_zip"
744
+ except (OSError, zipfile.BadZipFile):
745
+ return "html_zip"
746
+ if suffix in {".html", ".htm"} or magic.lstrip().lower().startswith(
747
+ (b"<!doctype html", b"<html")
748
+ ):
749
+ return "html"
750
+ if suffix in {".md", ".markdown", ".rmd", ".txt", ".csv", ".tsv"}:
751
+ return "text"
752
+ return "unsupported"
753
+
754
+
755
+ def _looks_like_notebook(source: Path) -> bool:
756
+ try:
757
+ with open(os.fspath(paths.long_path(source)), encoding="utf-8", newline="") as handle:
758
+ raw = json.load(handle)
759
+ except (OSError, UnicodeError, json.JSONDecodeError):
760
+ return False
761
+ return isinstance(raw, dict) and isinstance(raw.get("cells"), list)
762
+
763
+
764
+ def _looks_like_office(source: Path, suffix: str) -> bool:
765
+ required_member = {
766
+ ".docx": "word/document.xml",
767
+ ".pptx": "ppt/presentation.xml",
768
+ ".xlsx": "xl/workbook.xml",
769
+ }[suffix]
770
+ try:
771
+ with zipfile.ZipFile(os.fspath(paths.long_path(source))) as archive:
772
+ names = set(archive.namelist())
773
+ except (OSError, zipfile.BadZipFile):
774
+ return False
775
+ return "[Content_Types].xml" in names and required_member in names
776
+
777
+
778
+ def _validate_archive(infos: Sequence[zipfile.ZipInfo]) -> zipfile.ZipInfo | None:
779
+ if len(infos) > MAX_ZIP_MEMBERS:
780
+ raise ValueError("archive member-count limit exceeded")
781
+ total = 0
782
+ html_members: list[zipfile.ZipInfo] = []
783
+ for info in infos:
784
+ name = info.filename
785
+ if not name or "\\" in name or _WINDOWS_ABSOLUTE.match(name):
786
+ raise ValueError("archive member path is unsafe")
787
+ pure = PurePosixPath(name)
788
+ if pure.is_absolute() or any(part in _FORBIDDEN_ARCHIVE_PARTS for part in pure.parts):
789
+ raise ValueError("archive member path escapes extraction root")
790
+ if any(paths.safe_name(part) != part for part in pure.parts):
791
+ raise ValueError("archive member contains an unsafe filename component")
792
+ if info.flag_bits & 0x1:
793
+ raise ValueError("encrypted archive members are unsupported")
794
+ mode = (info.external_attr >> 16) & 0o170000
795
+ if mode == 0o120000:
796
+ raise ValueError("archive symlinks are unsupported")
797
+ if info.file_size > MAX_ZIP_MEMBER:
798
+ raise ValueError("archive member-size limit exceeded")
799
+ if info.file_size and info.compress_size == 0:
800
+ raise ValueError("archive compression ratio is unsafe")
801
+ if info.file_size / max(info.compress_size, 1) > MAX_ZIP_COMPRESSION_RATIO:
802
+ raise ValueError("archive compression ratio is unsafe")
803
+ total += info.file_size
804
+ if total > MAX_ZIP_UNCOMPRESSED:
805
+ raise ValueError("archive uncompressed-size limit exceeded")
806
+ if _is_html_name(name):
807
+ html_members.append(info)
808
+ if not html_members:
809
+ return None
810
+ return sorted(
811
+ html_members,
812
+ key=lambda item: (
813
+ 0 if PurePosixPath(item.filename).name.casefold() == "index.html" else 1,
814
+ len(PurePosixPath(item.filename).parts),
815
+ item.filename.casefold(),
816
+ ),
817
+ )[0]
818
+
819
+
820
+ def _is_html_name(name: str) -> bool:
821
+ return PurePosixPath(name).suffix.casefold() in {".html", ".htm"}
822
+
823
+
824
+ class _SafeHtmlMarkdown(HTMLParser):
825
+ """Small inert HTML-to-Markdown parser with no URL fetching."""
826
+
827
+ def __init__(self) -> None:
828
+ super().__init__(convert_charrefs=True)
829
+ self._parts: list[str] = []
830
+ self._skip_depth = 0
831
+ self._pre_depth = 0
832
+ self._links: list[str | None] = []
833
+
834
+ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
835
+ tag = tag.casefold()
836
+ if self._skip_depth:
837
+ if tag in _HTML_SKIP:
838
+ self._skip_depth += 1
839
+ return
840
+ if tag in _HTML_SKIP:
841
+ self._skip_depth = 1
842
+ return
843
+ if tag in {f"h{level}" for level in range(1, 7)}:
844
+ self._block()
845
+ self._parts.append("#" * int(tag[1]) + " ")
846
+ elif tag in {"p", "div", "section", "article", "main", "header", "footer", "figure"}:
847
+ self._block()
848
+ elif tag == "li":
849
+ self._block()
850
+ self._parts.append("- ")
851
+ elif tag == "pre":
852
+ self._block()
853
+ self._pre_depth += 1
854
+ elif tag == "code" and not self._pre_depth:
855
+ self._parts.append("`")
856
+ elif tag == "br":
857
+ self._parts.append("\n")
858
+ elif tag == "a":
859
+ href = next((value for name, value in attrs if name.casefold() == "href"), None)
860
+ self._parts.append("[")
861
+ self._links.append(_safe_href(href))
862
+ elif tag == "img":
863
+ src = next((value for name, value in attrs if name.casefold() == "src"), None)
864
+ if isinstance(src, str):
865
+ image_uri = _safe_image_uri(src)
866
+ if image_uri is not None:
867
+ self._parts.append(f"![image]({image_uri})")
868
+
869
+ def handle_endtag(self, tag: str) -> None:
870
+ tag = tag.casefold()
871
+ if self._skip_depth:
872
+ if tag in _HTML_SKIP:
873
+ self._skip_depth -= 1
874
+ return
875
+ if tag == "pre" and self._pre_depth:
876
+ self._pre_depth -= 1
877
+ self._block()
878
+ elif tag == "code" and not self._pre_depth:
879
+ self._parts.append("`")
880
+ elif tag == "a" and self._links:
881
+ href = self._links.pop()
882
+ self._parts.append(f"]({href})" if href else "]")
883
+ elif tag in _BLOCK_TAGS:
884
+ self._block()
885
+
886
+ def handle_data(self, data: str) -> None:
887
+ if self._skip_depth:
888
+ return
889
+ safe_data = escape(data, quote=False)
890
+ if self._pre_depth:
891
+ self._parts.append(safe_data)
892
+ else:
893
+ self._parts.append(re.sub(r"\s+", " ", safe_data))
894
+
895
+ def markdown(self) -> str:
896
+ return "".join(self._parts)
897
+
898
+ def _block(self) -> None:
899
+ if self._parts and not self._parts[-1].endswith("\n\n"):
900
+ self._parts.append("\n\n")
901
+
902
+
903
+ def _safe_href(value: str | None) -> str | None:
904
+ if not isinstance(value, str) or not value:
905
+ return None
906
+ parsed = urlsplit(value.strip())
907
+ if parsed.username is not None or parsed.password is not None:
908
+ return None
909
+ if parsed.scheme.casefold() not in {"", "http", "https"}:
910
+ return None
911
+ if parsed.scheme and not parsed.netloc:
912
+ return None
913
+ return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, "", ""))
914
+
915
+
916
+ def _safe_image_uri(value: str) -> str | None:
917
+ header, separator, payload = value.partition(",")
918
+ if not separator or not header.casefold().startswith("data:"):
919
+ return None
920
+ mime, separator, encoding = header[5:].partition(";")
921
+ if not separator or encoding.casefold() != "base64":
922
+ return None
923
+ mime = mime.casefold()
924
+ if mime not in _SAFE_IMAGE_MIME:
925
+ return None
926
+ encoded = "".join(payload.split())
927
+ if not encoded:
928
+ return None
929
+ try:
930
+ base64.b64decode(encoded, validate=True)
931
+ except (ValueError, binascii.Error):
932
+ return None
933
+ return f"data:{mime};base64,{encoded}"
934
+
935
+
936
+ def _render_notebook_output(output: object) -> tuple[str, str | None]:
937
+ if not isinstance(output, Mapping):
938
+ return "", "output is not an object"
939
+ output_type = output.get("output_type")
940
+ if output_type == "stream":
941
+ name = output.get("name") if isinstance(output.get("name"), str) else "output"
942
+ return f"### {name}\n\n{_fenced(_cell_text(output.get('text')), 'text')}", None
943
+ if output_type == "error":
944
+ traceback = _cell_text(output.get("traceback"))
945
+ if not traceback:
946
+ traceback = f"{output.get('ename', 'Error')}: {output.get('evalue', '')}"
947
+ return f"### Error\n\n{_fenced(_ANSI.sub('', traceback), 'text')}", None
948
+ if output_type in {"display_data", "execute_result"}:
949
+ data = output.get("data")
950
+ if not isinstance(data, Mapping):
951
+ return "", "output data is not an object"
952
+ markdown = data.get("text/markdown")
953
+ if markdown is not None:
954
+ return f"### Output\n\n{_cell_text(markdown)}", None
955
+ plain = data.get("text/plain")
956
+ if plain is not None:
957
+ return f"### Output\n\n{_fenced(_cell_text(plain), 'text')}", None
958
+ for mime, value in sorted(data.items(), key=lambda item: str(item[0])):
959
+ if isinstance(mime, str) and mime.startswith("image/") and isinstance(value, str):
960
+ image_uri = _safe_image_uri(f"data:{mime};base64,{value}")
961
+ if image_uri is not None:
962
+ return f"### Output\n\n![output]({image_uri})", None
963
+ mime_names = ", ".join(sorted(str(name) for name in data))
964
+ marker = f"[a2l unsupported notebook output MIME: {mime_names}]"
965
+ return marker, f"unsupported output MIME: {mime_names}"
966
+ return f"[a2l unsupported notebook output type: {output_type!r}]", "unsupported output type"
967
+
968
+
969
+ def _replace_attachments(text: str, attachments: object) -> str:
970
+ if not isinstance(attachments, Mapping):
971
+ return text
972
+
973
+ def replacement(match: re.Match[str]) -> str:
974
+ name = match.group(1)
975
+ payloads = attachments.get(name)
976
+ if not isinstance(payloads, Mapping):
977
+ return match.group(0)
978
+ for mime, value in sorted(payloads.items(), key=lambda item: str(item[0])):
979
+ if isinstance(mime, str) and isinstance(value, str):
980
+ image_uri = _safe_image_uri(f"data:{mime};base64,{value}")
981
+ if image_uri is not None:
982
+ return image_uri
983
+ return match.group(0)
984
+
985
+ return _ATTACHMENT.sub(replacement, text)
986
+
987
+
988
+ def _fenced(text: str, language: str) -> str:
989
+ longest = max((len(match.group(0)) for match in _BACKTICKS.finditer(text)), default=0)
990
+ fence = "`" * max(3, longest + 1)
991
+ body = text.rstrip("\n")
992
+ return f"{fence}{language}\n{body}\n{fence}"
993
+
994
+
995
+ def _cell_text(value: object) -> str:
996
+ if isinstance(value, str):
997
+ return value
998
+ if isinstance(value, list) and all(isinstance(item, str) for item in value):
999
+ return "".join(cast(list[str], value))
1000
+ return ""
1001
+
1002
+
1003
+ def _split_all_markdown(text: str, pages: int) -> list[str] | None:
1004
+ if pages == 1:
1005
+ return [text]
1006
+ pieces = re.split(r"\n\s*---\s*\n", text.strip())
1007
+ return pieces if len(pieces) == pages else None
1008
+
1009
+
1010
+ def _page_block(page: int, text: str) -> str:
1011
+ return f"<!-- a2l:page {page} -->\n{_normalise_markdown(text).rstrip()}"
1012
+
1013
+
1014
+ def _normalise_markdown(text: str) -> str:
1015
+ text = str(text).replace("\r\n", "\n").replace("\r", "\n")
1016
+ return text.rstrip() + "\n" if text.strip() else ""
1017
+
1018
+
1019
+ def _text_value(value: object) -> str:
1020
+ if isinstance(value, str):
1021
+ return value
1022
+ if isinstance(value, bytes):
1023
+ return value.decode("utf-8", "replace")
1024
+ return str(value)
1025
+
1026
+
1027
+ def _text_image_bytes(value: object) -> bytes:
1028
+ if isinstance(value, bytes):
1029
+ return value
1030
+ if isinstance(value, bytearray):
1031
+ return bytes(value)
1032
+ raise ConversionError("PDF renderer returned a non-byte image")
1033
+
1034
+
1035
+ def _pdf_oxide_page_count(document: object) -> int:
1036
+ value = getattr(document, "page_count", None)
1037
+ if callable(value):
1038
+ value = value()
1039
+ if isinstance(value, bool) or not isinstance(value, int) or value < 0:
1040
+ raise ConversionError("PDF backend returned an invalid page count")
1041
+ return value
1042
+
1043
+
1044
+ def _call_method(value: object, name: str, *args: object, **kwargs: object) -> object:
1045
+ method = getattr(value, name, None)
1046
+ if not callable(method):
1047
+ raise ConversionError(f"PDF backend does not provide {name}")
1048
+ return method(*args, **kwargs)
1049
+
1050
+
1051
+ def _close_quietly(value: object) -> None:
1052
+ close = getattr(value, "close", None)
1053
+ if callable(close):
1054
+ try:
1055
+ close()
1056
+ except Exception:
1057
+ return
1058
+
1059
+
1060
+ def _configure_tesseract(language: str) -> bool:
1061
+ candidates: list[Path] = []
1062
+ found = shutil.which("tesseract")
1063
+ if found:
1064
+ candidates.append(Path(found))
1065
+ if os.name == "nt":
1066
+ program_files = os.environ.get("PROGRAMFILES")
1067
+ if program_files:
1068
+ candidates.append(Path(program_files) / "Tesseract-OCR" / "tesseract.exe")
1069
+ candidates.append(Path.home() / "AppData" / "Local" / "Tesseract-OCR" / "tesseract.exe")
1070
+ executable = next(
1071
+ (candidate for candidate in candidates if paths.long_path(candidate).is_file()), None
1072
+ )
1073
+ if executable is None:
1074
+ return False
1075
+ pytesseract.pytesseract.tesseract_cmd = os.fspath(executable)
1076
+ try:
1077
+ languages = pytesseract.get_languages(config="")
1078
+ except (OSError, pytesseract.TesseractError):
1079
+ return False
1080
+ return language in languages
1081
+
1082
+
1083
+ def _validate_threshold(value: int) -> None:
1084
+ if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
1085
+ raise ValueError("ocr_words_per_page must be a positive integer")
1086
+
1087
+
1088
+ def _package_version(distribution: str, default: str) -> str:
1089
+ try:
1090
+ return importlib.metadata.version(distribution)
1091
+ except importlib.metadata.PackageNotFoundError:
1092
+ return default
1093
+
1094
+
1095
+ def _expected_tool(source: Path, backend: ConverterBackend) -> tuple[str, str]:
1096
+ return _expected_tool_for_kind(_classify_source(source), backend)
1097
+
1098
+
1099
+ def _expected_tool_for_kind(kind: str, backend: ConverterBackend) -> tuple[str, str]:
1100
+ if kind == "pdf":
1101
+ return backend.name, backend.version
1102
+ if kind == "notebook":
1103
+ return "nbformat", _package_version("nbformat", "missing")
1104
+ if kind == "html_zip":
1105
+ return "html-archive", CONVERTER_VERSION
1106
+ if kind == "html":
1107
+ return "html-sanitizer", CONVERTER_VERSION
1108
+ if kind == "office":
1109
+ return "markitdown", _package_version("markitdown", "missing")
1110
+ return "agent2learn-text", CONVERTER_VERSION
1111
+
1112
+
1113
+ def _artifact_is_current(
1114
+ vault: Vault,
1115
+ artifact: DerivedArtifact,
1116
+ entry: ManifestEntry,
1117
+ expected_tool: str,
1118
+ expected_version: str,
1119
+ expected_threshold: int | None,
1120
+ ) -> bool:
1121
+ # Ingest renders assignment prompts with title context that generic HTML conversion does not
1122
+ # have. Preserve that specialized twin, but only after the path, tool, version, and hashes are
1123
+ # all checked; every other HTML artifact still has to match the generic converter exactly.
1124
+ tool_is_current = artifact.tool == expected_tool and artifact.tool_version == expected_version
1125
+ if not tool_is_current and not (
1126
+ expected_tool == "html-sanitizer"
1127
+ and artifact.tool == RICH_TEXT_TOOL
1128
+ and artifact.tool_version == RICH_TEXT_TOOL_VERSION
1129
+ and _is_assignment_prompt_artifact(entry, artifact)
1130
+ ):
1131
+ return False
1132
+ if artifact.source_sha256 != entry.sha256 or artifact.ocr_words_per_page != expected_threshold:
1133
+ return False
1134
+ path = _artifact_path(vault, artifact)
1135
+ return paths.long_path(path).is_file() and _hash_file(path)[0] == artifact.sha256
1136
+
1137
+
1138
+ def _is_assignment_prompt_artifact(entry: ManifestEntry, artifact: DerivedArtifact) -> bool:
1139
+ """Recognize only the ingest-owned HTML/Markdown pair used for an assignment prompt."""
1140
+
1141
+ source = PurePosixPath(entry.path)
1142
+ derived = PurePosixPath(artifact.path)
1143
+ return (
1144
+ re.fullmatch(r"instructions(?:_\d+)?\.html", source.name.casefold()) is not None
1145
+ and any(part.casefold() == "assignments" for part in source.parts)
1146
+ and derived.parent == source.parent
1147
+ and derived.suffix.casefold() == ".md"
1148
+ )
1149
+
1150
+
1151
+ def _artifact_path(vault: Vault, artifact: DerivedArtifact) -> Path:
1152
+ return vault.materialized(
1153
+ ManifestEntry(
1154
+ path=artifact.path,
1155
+ sha256=artifact.sha256,
1156
+ source_id="derived",
1157
+ etag=None,
1158
+ last_modified=None,
1159
+ size=0,
1160
+ fetched_at="2026-01-01T00:00:00Z",
1161
+ )
1162
+ )
1163
+
1164
+
1165
+ def _hash_file(source: Path) -> tuple[str, int]:
1166
+ from hashlib import sha256
1167
+
1168
+ digest = sha256()
1169
+ size = 0
1170
+ try:
1171
+ with open(os.fspath(paths.long_path(source)), "rb") as handle:
1172
+ while chunk := handle.read(1024 * 1024):
1173
+ digest.update(chunk)
1174
+ size += len(chunk)
1175
+ except (FileNotFoundError, IsADirectoryError, OSError):
1176
+ return "", -1
1177
+ return digest.hexdigest(), size
1178
+
1179
+
1180
+ def _update_content_map(vault: Vault, key: str, **updates: object) -> None:
1181
+ for destination in sorted(
1182
+ path for path in paths.walk(vault.root) if path.name == "content_map.json"
1183
+ ):
1184
+ try:
1185
+ raw = course_index.read_content_map(destination.parent.parent)
1186
+ except (A2LError, UnicodeError):
1187
+ continue
1188
+ changed = False
1189
+ raw_rows = raw.get("topics")
1190
+ if not isinstance(raw_rows, list):
1191
+ continue
1192
+ rows: list[object] = raw_rows
1193
+ for row in rows:
1194
+ if isinstance(row, dict) and row.get("source_key") == key:
1195
+ row.update(updates)
1196
+ changed = True
1197
+ if changed:
1198
+ checked = course_index.reconcile_content_map(vault, rows)
1199
+ course_index.write_content_map(destination.parent.parent, checked, root=vault.root)
1200
+
1201
+
1202
+ def _now() -> str:
1203
+ return clock.stamp()
1204
+
1205
+
1206
+ __all__ = [
1207
+ "CONVERTER_VERSION",
1208
+ "DEFAULT_OCR_WORDS_PER_PAGE",
1209
+ "MIN_PDF_CHARS",
1210
+ "OCR_SETUP_ACTION",
1211
+ "ConversionError",
1212
+ "ConversionReport",
1213
+ "ConversionResult",
1214
+ "ConverterBackend",
1215
+ "PageCoverage",
1216
+ "PdfOxideBackend",
1217
+ "PdfiumBackend",
1218
+ "convert_html_zip",
1219
+ "convert_pdf",
1220
+ "convert_source",
1221
+ "convert_vault",
1222
+ "render_notebook",
1223
+ ]