agent2learn 0.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent2learn/__init__.py +3 -0
- agent2learn/_release.py +19 -0
- agent2learn/aipolicy.py +182 -0
- agent2learn/api.py +590 -0
- agent2learn/audit.py +358 -0
- agent2learn/auth/__init__.py +282 -0
- agent2learn/auth/cdp.py +1067 -0
- agent2learn/auth/paste.py +378 -0
- agent2learn/calendar.py +525 -0
- agent2learn/calibrate.py +347 -0
- agent2learn/check.py +1091 -0
- agent2learn/cli.py +2039 -0
- agent2learn/clock.py +39 -0
- agent2learn/config.py +205 -0
- agent2learn/console.py +229 -0
- agent2learn/convert.py +1223 -0
- agent2learn/doctor.py +1167 -0
- agent2learn/errors.py +32 -0
- agent2learn/ground.py +735 -0
- agent2learn/index.py +614 -0
- agent2learn/ingest.py +3229 -0
- agent2learn/locations.py +247 -0
- agent2learn/outlines.py +754 -0
- agent2learn/paths.py +683 -0
- agent2learn/pipeline.py +392 -0
- agent2learn/privacy.py +1123 -0
- agent2learn/schools/__init__.py +29 -0
- agent2learn/schools/_base.py +194 -0
- agent2learn/schools/generic.py +78 -0
- agent2learn/schools/uwaterloo.py +66 -0
- agent2learn/session.py +373 -0
- agent2learn/skills.py +1081 -0
- agent2learn/snapshot.py +399 -0
- agent2learn/submit.py +1047 -0
- agent2learn/transactions.py +157 -0
- agent2learn/upgrade.py +288 -0
- agent2learn/vault.py +1134 -0
- agent2learn-0.1.2.data/data/a2l-coursework/SKILL.md +52 -0
- agent2learn-0.1.2.data/data/a2l-setup/SKILL.md +27 -0
- agent2learn-0.1.2.data/data/a2l-study/SKILL.md +27 -0
- agent2learn-0.1.2.data/data/a2l-sync/SKILL.md +30 -0
- agent2learn-0.1.2.dist-info/METADATA +186 -0
- agent2learn-0.1.2.dist-info/RECORD +46 -0
- agent2learn-0.1.2.dist-info/WHEEL +4 -0
- agent2learn-0.1.2.dist-info/entry_points.txt +3 -0
- agent2learn-0.1.2.dist-info/licenses/LICENSE +202 -0
agent2learn/convert.py
ADDED
|
@@ -0,0 +1,1223 @@
|
|
|
1
|
+
"""Local, deterministic conversion of captured course sources into Markdown twins.
|
|
2
|
+
|
|
3
|
+
Conversion is deliberately downstream of ingestion. It receives only a local source path, never
|
|
4
|
+
an API client or session, and it writes nothing until the caller has accepted the complete result.
|
|
5
|
+
PDFs use the pinned PDF Oxide backend by default; notebooks and HTML archives are handled by small
|
|
6
|
+
auditable renderers rather than an execution-capable exporter stack.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import base64
|
|
12
|
+
import binascii
|
|
13
|
+
import importlib
|
|
14
|
+
import importlib.metadata
|
|
15
|
+
import io
|
|
16
|
+
import json
|
|
17
|
+
import os
|
|
18
|
+
import re
|
|
19
|
+
import shutil
|
|
20
|
+
import zipfile
|
|
21
|
+
from collections.abc import Callable, Collection, Mapping, Sequence
|
|
22
|
+
from dataclasses import dataclass, replace
|
|
23
|
+
from html import escape
|
|
24
|
+
from html.parser import HTMLParser
|
|
25
|
+
from pathlib import Path, PurePosixPath
|
|
26
|
+
from typing import Protocol, cast
|
|
27
|
+
from urllib.parse import urlsplit, urlunsplit
|
|
28
|
+
|
|
29
|
+
import pypdfium2 as pdfium # type: ignore[import-untyped]
|
|
30
|
+
import pytesseract # type: ignore[import-untyped]
|
|
31
|
+
from pdf_oxide import PdfDocument
|
|
32
|
+
from PIL import Image
|
|
33
|
+
|
|
34
|
+
from agent2learn import clock, paths
|
|
35
|
+
from agent2learn import index as course_index
|
|
36
|
+
from agent2learn.errors import A2LError
|
|
37
|
+
from agent2learn.vault import DerivedArtifact, ManifestEntry, Vault
|
|
38
|
+
|
|
39
|
+
DEFAULT_OCR_WORDS_PER_PAGE = 80
|
|
40
|
+
MIN_PDF_CHARS = 200
|
|
41
|
+
OCR_DPI = 300
|
|
42
|
+
CONVERTER_VERSION = "1"
|
|
43
|
+
RICH_TEXT_TOOL = "richtext-sanitizer"
|
|
44
|
+
RICH_TEXT_TOOL_VERSION = "1"
|
|
45
|
+
OCR_SETUP_ACTION = "install Tesseract with the 'eng' language pack, then rerun: a2l sync"
|
|
46
|
+
MAX_ZIP_MEMBERS = 1_000
|
|
47
|
+
MAX_ZIP_UNCOMPRESSED = 64 * 1024 * 1024
|
|
48
|
+
MAX_ZIP_MEMBER = 32 * 1024 * 1024
|
|
49
|
+
MAX_ZIP_COMPRESSION_RATIO = 1_000
|
|
50
|
+
|
|
51
|
+
_ANSI = re.compile(r"\x1b(?:\[[0-?]*[ -/]*[@-~]|\][^\x07]*(?:\x07|\x1b\\))")
|
|
52
|
+
_BACKTICKS = re.compile(r"`+")
|
|
53
|
+
_ATTACHMENT = re.compile(r"attachment:([^\s)\"'>]+)", re.IGNORECASE)
|
|
54
|
+
_WINDOWS_ABSOLUTE = re.compile(r"^[A-Za-z]:[\\/]")
|
|
55
|
+
_FORBIDDEN_ARCHIVE_PARTS = frozenset({"", ".", ".."})
|
|
56
|
+
_SAFE_IMAGE_MIME = frozenset({"image/gif", "image/jpeg", "image/png", "image/webp"})
|
|
57
|
+
_HTML_SKIP = frozenset({"script", "style", "form", "iframe", "object", "embed", "template"})
|
|
58
|
+
_BLOCK_TAGS = frozenset(
|
|
59
|
+
{
|
|
60
|
+
"address",
|
|
61
|
+
"article",
|
|
62
|
+
"aside",
|
|
63
|
+
"blockquote",
|
|
64
|
+
"div",
|
|
65
|
+
"dl",
|
|
66
|
+
"dt",
|
|
67
|
+
"dd",
|
|
68
|
+
"figure",
|
|
69
|
+
"footer",
|
|
70
|
+
"h1",
|
|
71
|
+
"h2",
|
|
72
|
+
"h3",
|
|
73
|
+
"h4",
|
|
74
|
+
"h5",
|
|
75
|
+
"h6",
|
|
76
|
+
"header",
|
|
77
|
+
"hr",
|
|
78
|
+
"li",
|
|
79
|
+
"main",
|
|
80
|
+
"nav",
|
|
81
|
+
"ol",
|
|
82
|
+
"p",
|
|
83
|
+
"pre",
|
|
84
|
+
"section",
|
|
85
|
+
"table",
|
|
86
|
+
"tr",
|
|
87
|
+
"ul",
|
|
88
|
+
}
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class ConversionError(A2LError):
|
|
93
|
+
"""A source could not be converted by the selected backend."""
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@dataclass(frozen=True)
|
|
97
|
+
class PageCoverage:
|
|
98
|
+
"""Deterministic coverage information for one one-based source page."""
|
|
99
|
+
|
|
100
|
+
page: int
|
|
101
|
+
mode: str
|
|
102
|
+
words: int
|
|
103
|
+
warning: str | None = None
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
def word_count(self) -> int:
|
|
107
|
+
"""Alias used by audit/report consumers that spell out the measurement."""
|
|
108
|
+
|
|
109
|
+
return self.words
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
@dataclass(frozen=True)
|
|
113
|
+
class ConversionResult:
|
|
114
|
+
"""Pure conversion output; no filesystem installation occurs here."""
|
|
115
|
+
|
|
116
|
+
markdown: str
|
|
117
|
+
page_coverage: tuple[PageCoverage, ...] = ()
|
|
118
|
+
warnings: tuple[str, ...] = ()
|
|
119
|
+
backend: str = "agent2learn"
|
|
120
|
+
tool_version: str = CONVERTER_VERSION
|
|
121
|
+
gap: bool = False
|
|
122
|
+
|
|
123
|
+
@property
|
|
124
|
+
def coverage(self) -> tuple[PageCoverage, ...]:
|
|
125
|
+
"""Short alias for callers rendering an audit summary."""
|
|
126
|
+
|
|
127
|
+
return self.page_coverage
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@dataclass(frozen=True)
|
|
131
|
+
class ConversionReport:
|
|
132
|
+
"""Summary of a vault conversion pass."""
|
|
133
|
+
|
|
134
|
+
converted: int = 0
|
|
135
|
+
skipped: int = 0
|
|
136
|
+
gaps: int = 0
|
|
137
|
+
warnings: tuple[str, ...] = ()
|
|
138
|
+
errors: tuple[str, ...] = ()
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class ConverterBackend(Protocol):
|
|
142
|
+
"""The narrow PDF backend boundary shared by the default and degraded implementations."""
|
|
143
|
+
|
|
144
|
+
name: str
|
|
145
|
+
version: str
|
|
146
|
+
|
|
147
|
+
def convert_pdf(self, source: Path, *, ocr_words_per_page: int) -> ConversionResult:
|
|
148
|
+
"""Convert one local PDF without accessing network or session state."""
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
class PdfOxideBackend:
|
|
152
|
+
"""The pinned PDF Oxide backend with explicit external-Tesseract OCR."""
|
|
153
|
+
|
|
154
|
+
name = "pdf-oxide"
|
|
155
|
+
|
|
156
|
+
def __init__(
|
|
157
|
+
self,
|
|
158
|
+
*,
|
|
159
|
+
document_factory: Callable[[Path], object] | None = None,
|
|
160
|
+
ocr_reader: Callable[[bytes], str] | None = None,
|
|
161
|
+
ocr_language: str = "eng",
|
|
162
|
+
dpi: int = OCR_DPI,
|
|
163
|
+
) -> None:
|
|
164
|
+
if isinstance(dpi, bool) or not isinstance(dpi, int) or dpi <= 0:
|
|
165
|
+
raise ValueError("dpi must be a positive integer")
|
|
166
|
+
if not isinstance(ocr_language, str) or not ocr_language:
|
|
167
|
+
raise ValueError("ocr_language must be a non-empty string")
|
|
168
|
+
self.version = _package_version("pdf-oxide", "unknown")
|
|
169
|
+
self._document_factory = document_factory or self._open_document
|
|
170
|
+
self._ocr_reader = ocr_reader
|
|
171
|
+
self._ocr_language = ocr_language
|
|
172
|
+
self._dpi = dpi
|
|
173
|
+
|
|
174
|
+
@staticmethod
|
|
175
|
+
def _open_document(source: Path) -> object:
|
|
176
|
+
return PdfDocument(os.fspath(paths.long_path(source)))
|
|
177
|
+
|
|
178
|
+
def convert_pdf(self, source: Path, *, ocr_words_per_page: int) -> ConversionResult:
|
|
179
|
+
_validate_threshold(ocr_words_per_page)
|
|
180
|
+
document = self._document_factory(Path(source))
|
|
181
|
+
try:
|
|
182
|
+
page_count = _pdf_oxide_page_count(document)
|
|
183
|
+
if page_count < 1:
|
|
184
|
+
raise ConversionError("PDF has no pages")
|
|
185
|
+
extracted = [
|
|
186
|
+
_text_value(_call_method(document, "extract_text_auto", page))
|
|
187
|
+
for page in range(page_count)
|
|
188
|
+
]
|
|
189
|
+
document_has_text = sum(len(text) for text in extracted) >= MIN_PDF_CHARS
|
|
190
|
+
healthy = [
|
|
191
|
+
document_has_text and len(text.split()) >= ocr_words_per_page for text in extracted
|
|
192
|
+
]
|
|
193
|
+
page_markdown: list[str] = []
|
|
194
|
+
coverage: list[PageCoverage] = []
|
|
195
|
+
warnings: list[str] = []
|
|
196
|
+
|
|
197
|
+
all_markdown: list[str] | None = None
|
|
198
|
+
if all(healthy) and hasattr(document, "to_markdown_all"):
|
|
199
|
+
try:
|
|
200
|
+
all_value = _call_method(document, "to_markdown_all")
|
|
201
|
+
all_markdown = _split_all_markdown(_text_value(all_value), page_count)
|
|
202
|
+
except Exception as exc:
|
|
203
|
+
warnings.append(f"whole-document Markdown unavailable: {type(exc).__name__}")
|
|
204
|
+
else:
|
|
205
|
+
if all_markdown is None:
|
|
206
|
+
warnings.append("whole-document Markdown had no stable page split")
|
|
207
|
+
|
|
208
|
+
for page, text in enumerate(extracted):
|
|
209
|
+
if healthy[page]:
|
|
210
|
+
if all_markdown is not None:
|
|
211
|
+
markdown = all_markdown[page]
|
|
212
|
+
else:
|
|
213
|
+
markdown = _text_value(_call_method(document, "to_markdown", page))
|
|
214
|
+
mode = "markdown"
|
|
215
|
+
warning = None
|
|
216
|
+
words = len(text.split())
|
|
217
|
+
else:
|
|
218
|
+
try:
|
|
219
|
+
image_bytes = _text_image_bytes(
|
|
220
|
+
_call_method(document, "render_page", page, dpi=self._dpi)
|
|
221
|
+
)
|
|
222
|
+
ocr_text = self._read_ocr(image_bytes)
|
|
223
|
+
except (ConversionError, OSError, RuntimeError) as exc:
|
|
224
|
+
warning = f"OCR unavailable on page {page + 1}: {type(exc).__name__}"
|
|
225
|
+
warnings.append(warning)
|
|
226
|
+
markdown = f"[a2l conversion gap: {warning}]"
|
|
227
|
+
mode = "unresolved"
|
|
228
|
+
words = 0
|
|
229
|
+
else:
|
|
230
|
+
markdown = ocr_text
|
|
231
|
+
mode = "ocr"
|
|
232
|
+
warning = None
|
|
233
|
+
words = len(ocr_text.split())
|
|
234
|
+
page_markdown.append(_page_block(page + 1, markdown))
|
|
235
|
+
coverage.append(PageCoverage(page + 1, mode, words, warning))
|
|
236
|
+
|
|
237
|
+
return ConversionResult(
|
|
238
|
+
markdown=_normalise_markdown("\n\n".join(page_markdown)),
|
|
239
|
+
page_coverage=tuple(coverage),
|
|
240
|
+
warnings=tuple(warnings),
|
|
241
|
+
backend=self.name,
|
|
242
|
+
tool_version=self.version,
|
|
243
|
+
gap=any(page.mode == "unresolved" for page in coverage),
|
|
244
|
+
)
|
|
245
|
+
except ConversionError:
|
|
246
|
+
raise
|
|
247
|
+
except Exception as exc:
|
|
248
|
+
raise ConversionError(f"pdf-oxide could not convert {source.name}") from exc
|
|
249
|
+
finally:
|
|
250
|
+
_close_quietly(document)
|
|
251
|
+
|
|
252
|
+
def _read_ocr(self, image_bytes: bytes) -> str:
|
|
253
|
+
if self._ocr_reader is not None:
|
|
254
|
+
normalized = _normalise_markdown(self._ocr_reader(image_bytes))
|
|
255
|
+
if not normalized:
|
|
256
|
+
raise ConversionError("OCR returned no text")
|
|
257
|
+
return normalized
|
|
258
|
+
if not _configure_tesseract(self._ocr_language):
|
|
259
|
+
raise ConversionError("Tesseract is unavailable or lacks the requested language")
|
|
260
|
+
try:
|
|
261
|
+
with Image.open(io.BytesIO(image_bytes)) as image:
|
|
262
|
+
text = pytesseract.image_to_string(image, lang=self._ocr_language)
|
|
263
|
+
except pytesseract.TesseractError as exc:
|
|
264
|
+
raise ConversionError("Tesseract could not OCR the rendered page") from exc
|
|
265
|
+
normalized = _normalise_markdown(text)
|
|
266
|
+
if not normalized:
|
|
267
|
+
raise ConversionError("Tesseract returned no text")
|
|
268
|
+
return normalized
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
class PdfiumBackend:
|
|
272
|
+
"""Named degraded PDFium fallback; it never silently replaces a successful default result."""
|
|
273
|
+
|
|
274
|
+
name = "pypdfium2"
|
|
275
|
+
|
|
276
|
+
def __init__(self) -> None:
|
|
277
|
+
self.version = _package_version("pypdfium2", "unknown")
|
|
278
|
+
|
|
279
|
+
def convert_pdf(self, source: Path, *, ocr_words_per_page: int) -> ConversionResult:
|
|
280
|
+
_validate_threshold(ocr_words_per_page)
|
|
281
|
+
try:
|
|
282
|
+
document = pdfium.PdfDocument(os.fspath(paths.long_path(source)))
|
|
283
|
+
except Exception as exc:
|
|
284
|
+
raise ConversionError(f"pypdfium2 could not open {source.name}") from exc
|
|
285
|
+
|
|
286
|
+
blocks: list[str] = []
|
|
287
|
+
coverage: list[PageCoverage] = []
|
|
288
|
+
try:
|
|
289
|
+
for index in range(len(document)):
|
|
290
|
+
page = document[index]
|
|
291
|
+
textpage = page.get_textpage()
|
|
292
|
+
try:
|
|
293
|
+
text = _normalise_markdown(textpage.get_text_bounded())
|
|
294
|
+
finally:
|
|
295
|
+
_close_quietly(textpage)
|
|
296
|
+
_close_quietly(page)
|
|
297
|
+
blocks.append(_page_block(index + 1, text))
|
|
298
|
+
coverage.append(PageCoverage(index + 1, "fallback", len(text.split())))
|
|
299
|
+
if not coverage:
|
|
300
|
+
raise ConversionError("PDF has no pages")
|
|
301
|
+
empty_page = any(page.words == 0 for page in coverage)
|
|
302
|
+
return ConversionResult(
|
|
303
|
+
markdown=_normalise_markdown("\n\n".join(blocks)),
|
|
304
|
+
page_coverage=tuple(coverage),
|
|
305
|
+
backend=self.name,
|
|
306
|
+
tool_version=self.version,
|
|
307
|
+
warnings=("fallback contains an empty page",) if empty_page else (),
|
|
308
|
+
gap=empty_page,
|
|
309
|
+
)
|
|
310
|
+
except ConversionError:
|
|
311
|
+
raise
|
|
312
|
+
except Exception as exc:
|
|
313
|
+
raise ConversionError(f"pypdfium2 could not convert {source.name}") from exc
|
|
314
|
+
finally:
|
|
315
|
+
_close_quietly(document)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def convert_pdf(
|
|
319
|
+
source: Path,
|
|
320
|
+
*,
|
|
321
|
+
backend: ConverterBackend | None = None,
|
|
322
|
+
fallback: ConverterBackend | None = None,
|
|
323
|
+
ocr_words_per_page: int = DEFAULT_OCR_WORDS_PER_PAGE,
|
|
324
|
+
) -> ConversionResult:
|
|
325
|
+
"""Use the default PDF backend, falling back only when it cannot convert at all."""
|
|
326
|
+
|
|
327
|
+
selected = backend or PdfOxideBackend()
|
|
328
|
+
degraded = fallback or PdfiumBackend()
|
|
329
|
+
try:
|
|
330
|
+
return selected.convert_pdf(source, ocr_words_per_page=ocr_words_per_page)
|
|
331
|
+
except Exception:
|
|
332
|
+
try:
|
|
333
|
+
result = degraded.convert_pdf(source, ocr_words_per_page=ocr_words_per_page)
|
|
334
|
+
except Exception as fallback_error:
|
|
335
|
+
raise ConversionError("both PDF backends failed") from fallback_error
|
|
336
|
+
warning = f"default PDF backend failed; accepted {degraded.name} fallback"
|
|
337
|
+
return replace(result, warnings=(warning, *result.warnings))
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def convert_source(
|
|
341
|
+
source: Path,
|
|
342
|
+
*,
|
|
343
|
+
backend: ConverterBackend | None = None,
|
|
344
|
+
fallback: ConverterBackend | None = None,
|
|
345
|
+
ocr_words_per_page: int = DEFAULT_OCR_WORDS_PER_PAGE,
|
|
346
|
+
content_type: str | None = None,
|
|
347
|
+
) -> ConversionResult:
|
|
348
|
+
"""Convert one local source based on magic bytes plus extension, never executing it."""
|
|
349
|
+
|
|
350
|
+
del content_type # Server metadata is advisory; local bytes and the extension decide dispatch.
|
|
351
|
+
source = Path(source)
|
|
352
|
+
if not paths.long_path(source).is_file():
|
|
353
|
+
raise ConversionError(f"source is not a regular file: {source.name}")
|
|
354
|
+
kind = _classify_source(source)
|
|
355
|
+
if kind == "pdf":
|
|
356
|
+
return convert_pdf(
|
|
357
|
+
source,
|
|
358
|
+
backend=backend,
|
|
359
|
+
fallback=fallback,
|
|
360
|
+
ocr_words_per_page=ocr_words_per_page,
|
|
361
|
+
)
|
|
362
|
+
if kind == "notebook":
|
|
363
|
+
return render_notebook(source)
|
|
364
|
+
if kind == "html_zip":
|
|
365
|
+
return convert_html_zip(source)
|
|
366
|
+
if kind == "html":
|
|
367
|
+
return _convert_html_file(source)
|
|
368
|
+
if kind == "text":
|
|
369
|
+
try:
|
|
370
|
+
with open(os.fspath(paths.long_path(source)), encoding="utf-8", newline="") as handle:
|
|
371
|
+
text = handle.read()
|
|
372
|
+
except (OSError, UnicodeError) as exc:
|
|
373
|
+
raise ConversionError(f"text source could not be read: {source.name}") from exc
|
|
374
|
+
return ConversionResult(
|
|
375
|
+
markdown=_normalise_markdown(text),
|
|
376
|
+
page_coverage=(PageCoverage(1, "text", len(text.split())),),
|
|
377
|
+
backend="agent2learn-text",
|
|
378
|
+
tool_version=CONVERTER_VERSION,
|
|
379
|
+
)
|
|
380
|
+
if kind == "office":
|
|
381
|
+
return _convert_office(source)
|
|
382
|
+
return ConversionResult(
|
|
383
|
+
markdown="",
|
|
384
|
+
warnings=(f"unsupported or mismatched source format: {source.suffix or source.name}",),
|
|
385
|
+
backend="agent2learn",
|
|
386
|
+
tool_version=CONVERTER_VERSION,
|
|
387
|
+
gap=True,
|
|
388
|
+
)
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def render_notebook(source: Path) -> ConversionResult:
|
|
392
|
+
"""Render a v4 notebook without importing or executing any cell code."""
|
|
393
|
+
|
|
394
|
+
try:
|
|
395
|
+
nbformat = importlib.import_module("nbformat")
|
|
396
|
+
except ImportError:
|
|
397
|
+
return ConversionResult(
|
|
398
|
+
markdown="",
|
|
399
|
+
warnings=("optional notebook dependency nbformat is not installed",),
|
|
400
|
+
backend="nbformat",
|
|
401
|
+
tool_version="missing",
|
|
402
|
+
gap=True,
|
|
403
|
+
)
|
|
404
|
+
try:
|
|
405
|
+
notebook = nbformat.read(os.fspath(paths.long_path(source)), as_version=4)
|
|
406
|
+
except Exception as exc:
|
|
407
|
+
raise ConversionError(f"notebook could not be parsed: {source.name}") from exc
|
|
408
|
+
|
|
409
|
+
metadata = notebook.get("metadata", {})
|
|
410
|
+
language = "text"
|
|
411
|
+
if isinstance(metadata, Mapping):
|
|
412
|
+
language_info = metadata.get("language_info", {})
|
|
413
|
+
if isinstance(language_info, Mapping) and isinstance(language_info.get("name"), str):
|
|
414
|
+
language = cast(str, language_info["name"]).strip() or "text"
|
|
415
|
+
|
|
416
|
+
blocks: list[str] = []
|
|
417
|
+
warnings: list[str] = []
|
|
418
|
+
cells = notebook.get("cells", [])
|
|
419
|
+
if not isinstance(cells, Sequence) or isinstance(cells, (str, bytes)):
|
|
420
|
+
raise ConversionError("notebook cells must be an array")
|
|
421
|
+
for number, cell in enumerate(cells, start=1):
|
|
422
|
+
if not isinstance(cell, Mapping):
|
|
423
|
+
warnings.append(f"cell {number} is not an object")
|
|
424
|
+
continue
|
|
425
|
+
cell_type = cell.get("cell_type")
|
|
426
|
+
source_text = _cell_text(cell.get("source"))
|
|
427
|
+
if cell_type == "markdown":
|
|
428
|
+
attachments = cell.get("attachments", {})
|
|
429
|
+
blocks.append(_replace_attachments(source_text, attachments))
|
|
430
|
+
elif cell_type == "code":
|
|
431
|
+
blocks.append(_fenced(source_text, language))
|
|
432
|
+
outputs = cell.get("outputs", [])
|
|
433
|
+
if isinstance(outputs, Sequence) and not isinstance(outputs, (str, bytes)):
|
|
434
|
+
for output in outputs:
|
|
435
|
+
rendered, output_warning = _render_notebook_output(output)
|
|
436
|
+
if rendered:
|
|
437
|
+
blocks.append(rendered)
|
|
438
|
+
if output_warning is not None:
|
|
439
|
+
warnings.append(f"cell {number}: {output_warning}")
|
|
440
|
+
else:
|
|
441
|
+
warnings.append(f"cell {number}: unsupported cell type {cell_type!r}")
|
|
442
|
+
|
|
443
|
+
markdown = _normalise_markdown("\n\n".join(blocks))
|
|
444
|
+
return ConversionResult(
|
|
445
|
+
markdown=markdown,
|
|
446
|
+
page_coverage=(PageCoverage(1, "notebook", len(markdown.split())),),
|
|
447
|
+
warnings=tuple(warnings),
|
|
448
|
+
backend="nbformat",
|
|
449
|
+
tool_version=_package_version("nbformat", "unknown"),
|
|
450
|
+
gap=False,
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def convert_html_zip(source: Path) -> ConversionResult:
|
|
455
|
+
"""Extract only a safe main HTML member from a bounded archive in memory."""
|
|
456
|
+
|
|
457
|
+
try:
|
|
458
|
+
with zipfile.ZipFile(os.fspath(paths.long_path(source))) as archive:
|
|
459
|
+
infos = archive.infolist()
|
|
460
|
+
html_info = _validate_archive(infos)
|
|
461
|
+
if html_info is None:
|
|
462
|
+
return ConversionResult(
|
|
463
|
+
markdown="",
|
|
464
|
+
warnings=("HTML archive has no main HTML member",),
|
|
465
|
+
backend="html-archive",
|
|
466
|
+
tool_version=CONVERTER_VERSION,
|
|
467
|
+
gap=True,
|
|
468
|
+
)
|
|
469
|
+
html_bytes = archive.read(html_info)
|
|
470
|
+
except (OSError, ValueError, zipfile.BadZipFile, zipfile.LargeZipFile) as exc:
|
|
471
|
+
raise ConversionError(f"HTML archive rejected: {source.name}") from exc
|
|
472
|
+
try:
|
|
473
|
+
text = html_bytes.decode("utf-8")
|
|
474
|
+
except UnicodeDecodeError as exc:
|
|
475
|
+
raise ConversionError("HTML archive main member is not UTF-8") from exc
|
|
476
|
+
return _html_result(text)
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def convert_vault(
|
|
480
|
+
vault: Vault,
|
|
481
|
+
*,
|
|
482
|
+
backend: ConverterBackend | None = None,
|
|
483
|
+
fallback: ConverterBackend | None = None,
|
|
484
|
+
ocr_words_per_page: int = DEFAULT_OCR_WORDS_PER_PAGE,
|
|
485
|
+
source_keys: Collection[str] | None = None,
|
|
486
|
+
) -> ConversionReport:
|
|
487
|
+
"""Install current, hash-linked twins for manifest sources without losing revisions."""
|
|
488
|
+
|
|
489
|
+
_validate_threshold(ocr_words_per_page)
|
|
490
|
+
entries = vault.manifest()
|
|
491
|
+
if source_keys is not None:
|
|
492
|
+
selected = set(source_keys)
|
|
493
|
+
if selected - entries.keys():
|
|
494
|
+
raise A2LError("conversion source is not recorded in the manifest")
|
|
495
|
+
entries = {key: entry for key, entry in entries.items() if key in selected}
|
|
496
|
+
selected_backend = backend or PdfOxideBackend()
|
|
497
|
+
selected_fallback = fallback or PdfiumBackend()
|
|
498
|
+
converted = skipped = gaps = 0
|
|
499
|
+
warnings: list[str] = []
|
|
500
|
+
errors: list[str] = []
|
|
501
|
+
for key, entry in sorted(entries.items()):
|
|
502
|
+
source = vault.materialized(entry)
|
|
503
|
+
if not paths.long_path(source).is_file():
|
|
504
|
+
gaps += 1
|
|
505
|
+
message = f"{key}: source is missing"
|
|
506
|
+
errors.append(message)
|
|
507
|
+
_update_content_map(vault, key, availability="integrity_gap", next_action=message)
|
|
508
|
+
continue
|
|
509
|
+
source_hash, source_size = _hash_file(source)
|
|
510
|
+
if source_hash != entry.sha256 or source_size != entry.size:
|
|
511
|
+
gaps += 1
|
|
512
|
+
message = f"{key}: source hash does not match manifest"
|
|
513
|
+
errors.append(message)
|
|
514
|
+
_update_content_map(
|
|
515
|
+
vault,
|
|
516
|
+
key,
|
|
517
|
+
availability="integrity_gap",
|
|
518
|
+
source_path=entry.path,
|
|
519
|
+
path=None,
|
|
520
|
+
sha256=entry.sha256,
|
|
521
|
+
source_sha256=entry.sha256,
|
|
522
|
+
size=entry.size,
|
|
523
|
+
next_action=message,
|
|
524
|
+
)
|
|
525
|
+
continue
|
|
526
|
+
|
|
527
|
+
artifact = entry.derived.get("markdown")
|
|
528
|
+
try:
|
|
529
|
+
source_kind = _classify_source(source)
|
|
530
|
+
expected_tool, expected_version = _expected_tool_for_kind(source_kind, selected_backend)
|
|
531
|
+
except Exception as exc:
|
|
532
|
+
gaps += 1
|
|
533
|
+
message = f"{key}: {type(exc).__name__}"
|
|
534
|
+
errors.append(message)
|
|
535
|
+
_update_content_map(
|
|
536
|
+
vault,
|
|
537
|
+
key,
|
|
538
|
+
availability="conversion_gap",
|
|
539
|
+
source_path=entry.path,
|
|
540
|
+
path=None,
|
|
541
|
+
sha256=entry.sha256,
|
|
542
|
+
source_sha256=entry.sha256,
|
|
543
|
+
size=entry.size,
|
|
544
|
+
next_action=message,
|
|
545
|
+
)
|
|
546
|
+
continue
|
|
547
|
+
expected_threshold = ocr_words_per_page if source_kind == "pdf" else None
|
|
548
|
+
if (
|
|
549
|
+
artifact is not None
|
|
550
|
+
and vault.owns_derived_path(key, artifact.path)
|
|
551
|
+
and _artifact_is_current(
|
|
552
|
+
vault,
|
|
553
|
+
artifact,
|
|
554
|
+
entry,
|
|
555
|
+
expected_tool,
|
|
556
|
+
expected_version,
|
|
557
|
+
expected_threshold,
|
|
558
|
+
)
|
|
559
|
+
):
|
|
560
|
+
skipped += 1
|
|
561
|
+
continue
|
|
562
|
+
|
|
563
|
+
try:
|
|
564
|
+
result = convert_source(
|
|
565
|
+
source,
|
|
566
|
+
backend=selected_backend,
|
|
567
|
+
fallback=selected_fallback,
|
|
568
|
+
ocr_words_per_page=ocr_words_per_page,
|
|
569
|
+
)
|
|
570
|
+
except Exception as exc:
|
|
571
|
+
gaps += 1
|
|
572
|
+
message = f"{key}: {type(exc).__name__}"
|
|
573
|
+
errors.append(message)
|
|
574
|
+
_update_content_map(
|
|
575
|
+
vault,
|
|
576
|
+
key,
|
|
577
|
+
availability="conversion_gap",
|
|
578
|
+
source_path=entry.path,
|
|
579
|
+
path=None,
|
|
580
|
+
sha256=entry.sha256,
|
|
581
|
+
source_sha256=entry.sha256,
|
|
582
|
+
size=entry.size,
|
|
583
|
+
next_action=message,
|
|
584
|
+
)
|
|
585
|
+
continue
|
|
586
|
+
warnings.extend(f"{key}: {warning}" for warning in result.warnings)
|
|
587
|
+
if result.gap or not result.markdown:
|
|
588
|
+
gaps += 1
|
|
589
|
+
lowered = tuple(warning.casefold() for warning in result.warnings)
|
|
590
|
+
availability = (
|
|
591
|
+
"unsupported_format"
|
|
592
|
+
if any("unsupported" in warning or "optional" in warning for warning in lowered)
|
|
593
|
+
else "conversion_gap"
|
|
594
|
+
)
|
|
595
|
+
if any("ocr unavailable" in warning for warning in lowered):
|
|
596
|
+
next_action = OCR_SETUP_ACTION
|
|
597
|
+
else:
|
|
598
|
+
next_action = result.warnings[0] if result.warnings else "conversion gap"
|
|
599
|
+
_update_content_map(
|
|
600
|
+
vault,
|
|
601
|
+
key,
|
|
602
|
+
availability=availability,
|
|
603
|
+
source_path=entry.path,
|
|
604
|
+
path=None,
|
|
605
|
+
sha256=entry.sha256,
|
|
606
|
+
source_sha256=entry.sha256,
|
|
607
|
+
size=entry.size,
|
|
608
|
+
next_action=next_action,
|
|
609
|
+
)
|
|
610
|
+
continue
|
|
611
|
+
|
|
612
|
+
destination = vault.derived_destination(key, source.with_suffix(".md"))
|
|
613
|
+
prior_artifact = entry.derived.get("markdown")
|
|
614
|
+
local_modification = False
|
|
615
|
+
if prior_artifact is not None:
|
|
616
|
+
prior_path = _artifact_path(vault, prior_artifact)
|
|
617
|
+
if paths.long_path(prior_path).is_file() and _hash_file(prior_path)[0] != (
|
|
618
|
+
prior_artifact.sha256
|
|
619
|
+
):
|
|
620
|
+
vault.preserve_revision(key, changed_at=clock.now())
|
|
621
|
+
local_modification = True
|
|
622
|
+
paths.ensure_dir(destination.parent, root=vault.root)
|
|
623
|
+
paths.atomic_write_text(destination, result.markdown, root=vault.root)
|
|
624
|
+
derived = DerivedArtifact(
|
|
625
|
+
path=paths.rel_posix(destination, vault.root),
|
|
626
|
+
sha256=_hash_file(destination)[0],
|
|
627
|
+
source_sha256=entry.sha256,
|
|
628
|
+
tool=result.backend,
|
|
629
|
+
tool_version=result.tool_version,
|
|
630
|
+
created_at=_now(),
|
|
631
|
+
ocr_words_per_page=(ocr_words_per_page if source_kind == "pdf" else None),
|
|
632
|
+
page_coverage=tuple(
|
|
633
|
+
{
|
|
634
|
+
"page": page.page,
|
|
635
|
+
"mode": page.mode,
|
|
636
|
+
"words": page.words,
|
|
637
|
+
"warning": page.warning,
|
|
638
|
+
}
|
|
639
|
+
for page in result.page_coverage
|
|
640
|
+
),
|
|
641
|
+
)
|
|
642
|
+
updated = replace(entry, derived={"markdown": derived})
|
|
643
|
+
vault.mark(key, updated)
|
|
644
|
+
vault.save_manifest()
|
|
645
|
+
_update_content_map(
|
|
646
|
+
vault,
|
|
647
|
+
key,
|
|
648
|
+
availability="markdown_ready",
|
|
649
|
+
source_path=updated.path,
|
|
650
|
+
path=derived.path,
|
|
651
|
+
sha256=updated.sha256,
|
|
652
|
+
source_sha256=updated.sha256,
|
|
653
|
+
size=updated.size,
|
|
654
|
+
next_action="ready for citation",
|
|
655
|
+
)
|
|
656
|
+
if local_modification:
|
|
657
|
+
warnings.append(f"{key}: preserved locally modified markdown twin")
|
|
658
|
+
converted += 1
|
|
659
|
+
return ConversionReport(converted, skipped, gaps, tuple(warnings), tuple(errors))
|
|
660
|
+
|
|
661
|
+
|
|
662
|
+
def _convert_html_file(source: Path) -> ConversionResult:
|
|
663
|
+
try:
|
|
664
|
+
with open(os.fspath(paths.long_path(source)), encoding="utf-8", newline="") as handle:
|
|
665
|
+
text = handle.read()
|
|
666
|
+
except (OSError, UnicodeError) as exc:
|
|
667
|
+
raise ConversionError(f"HTML source could not be read: {source.name}") from exc
|
|
668
|
+
return _html_result(text)
|
|
669
|
+
|
|
670
|
+
|
|
671
|
+
def _html_result(text: str) -> ConversionResult:
|
|
672
|
+
parser = _SafeHtmlMarkdown()
|
|
673
|
+
parser.feed(text)
|
|
674
|
+
parser.close()
|
|
675
|
+
markdown = _normalise_markdown(parser.markdown())
|
|
676
|
+
return ConversionResult(
|
|
677
|
+
markdown=markdown,
|
|
678
|
+
page_coverage=(PageCoverage(1, "html", len(markdown.split())),),
|
|
679
|
+
warnings=(),
|
|
680
|
+
backend="html-sanitizer",
|
|
681
|
+
tool_version=CONVERTER_VERSION,
|
|
682
|
+
)
|
|
683
|
+
|
|
684
|
+
|
|
685
|
+
def _convert_office(source: Path) -> ConversionResult:
|
|
686
|
+
try:
|
|
687
|
+
module = importlib.import_module("markitdown")
|
|
688
|
+
converter_type = module.MarkItDown
|
|
689
|
+
converted = converter_type().convert(os.fspath(paths.long_path(source)))
|
|
690
|
+
text = getattr(converted, "text_content", None)
|
|
691
|
+
if not isinstance(text, str):
|
|
692
|
+
raise ConversionError("MarkItDown returned no text content")
|
|
693
|
+
except ImportError:
|
|
694
|
+
return ConversionResult(
|
|
695
|
+
markdown="",
|
|
696
|
+
warnings=("optional office dependency markitdown is not installed",),
|
|
697
|
+
backend="markitdown",
|
|
698
|
+
tool_version="missing",
|
|
699
|
+
gap=True,
|
|
700
|
+
)
|
|
701
|
+
except Exception as exc:
|
|
702
|
+
return ConversionResult(
|
|
703
|
+
markdown="",
|
|
704
|
+
warnings=(f"office conversion failed: {type(exc).__name__}",),
|
|
705
|
+
backend="markitdown",
|
|
706
|
+
tool_version=_package_version("markitdown", "unknown"),
|
|
707
|
+
gap=True,
|
|
708
|
+
)
|
|
709
|
+
markdown = _normalise_markdown(text)
|
|
710
|
+
return ConversionResult(
|
|
711
|
+
markdown=markdown,
|
|
712
|
+
page_coverage=(PageCoverage(1, "office", len(markdown.split())),),
|
|
713
|
+
backend="markitdown",
|
|
714
|
+
tool_version=_package_version("markitdown", "unknown"),
|
|
715
|
+
)
|
|
716
|
+
|
|
717
|
+
|
|
718
|
+
def _classify_source(source: Path) -> str:
|
|
719
|
+
suffixes = [suffix.casefold() for suffix in source.suffixes]
|
|
720
|
+
suffix = source.suffix.casefold()
|
|
721
|
+
try:
|
|
722
|
+
with open(os.fspath(paths.long_path(source)), "rb") as handle:
|
|
723
|
+
magic = handle.read(16)
|
|
724
|
+
except OSError as exc:
|
|
725
|
+
raise ConversionError(f"source could not be inspected: {source.name}") from exc
|
|
726
|
+
if magic.startswith(b"%PDF"):
|
|
727
|
+
return "pdf"
|
|
728
|
+
if suffix == ".pdf":
|
|
729
|
+
return "unsupported"
|
|
730
|
+
if suffix == ".ipynb" or (magic.lstrip().startswith(b"{") and _looks_like_notebook(source)):
|
|
731
|
+
return "notebook"
|
|
732
|
+
if suffix in {".docx", ".pptx", ".xlsx"}:
|
|
733
|
+
return "office" if _looks_like_office(source, suffix) else "unsupported"
|
|
734
|
+
if suffix in {".doc", ".ppt", ".xls"}:
|
|
735
|
+
return "office" if magic.startswith(b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1") else "unsupported"
|
|
736
|
+
archive_path = os.fspath(paths.long_path(source))
|
|
737
|
+
if zipfile.is_zipfile(archive_path):
|
|
738
|
+
if suffix == ".zip" and ".html" in suffixes:
|
|
739
|
+
return "html_zip"
|
|
740
|
+
try:
|
|
741
|
+
with zipfile.ZipFile(archive_path) as archive:
|
|
742
|
+
if any(_is_html_name(info.filename) for info in archive.infolist()):
|
|
743
|
+
return "html_zip"
|
|
744
|
+
except (OSError, zipfile.BadZipFile):
|
|
745
|
+
return "html_zip"
|
|
746
|
+
if suffix in {".html", ".htm"} or magic.lstrip().lower().startswith(
|
|
747
|
+
(b"<!doctype html", b"<html")
|
|
748
|
+
):
|
|
749
|
+
return "html"
|
|
750
|
+
if suffix in {".md", ".markdown", ".rmd", ".txt", ".csv", ".tsv"}:
|
|
751
|
+
return "text"
|
|
752
|
+
return "unsupported"
|
|
753
|
+
|
|
754
|
+
|
|
755
|
+
def _looks_like_notebook(source: Path) -> bool:
|
|
756
|
+
try:
|
|
757
|
+
with open(os.fspath(paths.long_path(source)), encoding="utf-8", newline="") as handle:
|
|
758
|
+
raw = json.load(handle)
|
|
759
|
+
except (OSError, UnicodeError, json.JSONDecodeError):
|
|
760
|
+
return False
|
|
761
|
+
return isinstance(raw, dict) and isinstance(raw.get("cells"), list)
|
|
762
|
+
|
|
763
|
+
|
|
764
|
+
def _looks_like_office(source: Path, suffix: str) -> bool:
|
|
765
|
+
required_member = {
|
|
766
|
+
".docx": "word/document.xml",
|
|
767
|
+
".pptx": "ppt/presentation.xml",
|
|
768
|
+
".xlsx": "xl/workbook.xml",
|
|
769
|
+
}[suffix]
|
|
770
|
+
try:
|
|
771
|
+
with zipfile.ZipFile(os.fspath(paths.long_path(source))) as archive:
|
|
772
|
+
names = set(archive.namelist())
|
|
773
|
+
except (OSError, zipfile.BadZipFile):
|
|
774
|
+
return False
|
|
775
|
+
return "[Content_Types].xml" in names and required_member in names
|
|
776
|
+
|
|
777
|
+
|
|
778
|
+
def _validate_archive(infos: Sequence[zipfile.ZipInfo]) -> zipfile.ZipInfo | None:
|
|
779
|
+
if len(infos) > MAX_ZIP_MEMBERS:
|
|
780
|
+
raise ValueError("archive member-count limit exceeded")
|
|
781
|
+
total = 0
|
|
782
|
+
html_members: list[zipfile.ZipInfo] = []
|
|
783
|
+
for info in infos:
|
|
784
|
+
name = info.filename
|
|
785
|
+
if not name or "\\" in name or _WINDOWS_ABSOLUTE.match(name):
|
|
786
|
+
raise ValueError("archive member path is unsafe")
|
|
787
|
+
pure = PurePosixPath(name)
|
|
788
|
+
if pure.is_absolute() or any(part in _FORBIDDEN_ARCHIVE_PARTS for part in pure.parts):
|
|
789
|
+
raise ValueError("archive member path escapes extraction root")
|
|
790
|
+
if any(paths.safe_name(part) != part for part in pure.parts):
|
|
791
|
+
raise ValueError("archive member contains an unsafe filename component")
|
|
792
|
+
if info.flag_bits & 0x1:
|
|
793
|
+
raise ValueError("encrypted archive members are unsupported")
|
|
794
|
+
mode = (info.external_attr >> 16) & 0o170000
|
|
795
|
+
if mode == 0o120000:
|
|
796
|
+
raise ValueError("archive symlinks are unsupported")
|
|
797
|
+
if info.file_size > MAX_ZIP_MEMBER:
|
|
798
|
+
raise ValueError("archive member-size limit exceeded")
|
|
799
|
+
if info.file_size and info.compress_size == 0:
|
|
800
|
+
raise ValueError("archive compression ratio is unsafe")
|
|
801
|
+
if info.file_size / max(info.compress_size, 1) > MAX_ZIP_COMPRESSION_RATIO:
|
|
802
|
+
raise ValueError("archive compression ratio is unsafe")
|
|
803
|
+
total += info.file_size
|
|
804
|
+
if total > MAX_ZIP_UNCOMPRESSED:
|
|
805
|
+
raise ValueError("archive uncompressed-size limit exceeded")
|
|
806
|
+
if _is_html_name(name):
|
|
807
|
+
html_members.append(info)
|
|
808
|
+
if not html_members:
|
|
809
|
+
return None
|
|
810
|
+
return sorted(
|
|
811
|
+
html_members,
|
|
812
|
+
key=lambda item: (
|
|
813
|
+
0 if PurePosixPath(item.filename).name.casefold() == "index.html" else 1,
|
|
814
|
+
len(PurePosixPath(item.filename).parts),
|
|
815
|
+
item.filename.casefold(),
|
|
816
|
+
),
|
|
817
|
+
)[0]
|
|
818
|
+
|
|
819
|
+
|
|
820
|
+
def _is_html_name(name: str) -> bool:
|
|
821
|
+
return PurePosixPath(name).suffix.casefold() in {".html", ".htm"}
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
class _SafeHtmlMarkdown(HTMLParser):
|
|
825
|
+
"""Small inert HTML-to-Markdown parser with no URL fetching."""
|
|
826
|
+
|
|
827
|
+
def __init__(self) -> None:
|
|
828
|
+
super().__init__(convert_charrefs=True)
|
|
829
|
+
self._parts: list[str] = []
|
|
830
|
+
self._skip_depth = 0
|
|
831
|
+
self._pre_depth = 0
|
|
832
|
+
self._links: list[str | None] = []
|
|
833
|
+
|
|
834
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
835
|
+
tag = tag.casefold()
|
|
836
|
+
if self._skip_depth:
|
|
837
|
+
if tag in _HTML_SKIP:
|
|
838
|
+
self._skip_depth += 1
|
|
839
|
+
return
|
|
840
|
+
if tag in _HTML_SKIP:
|
|
841
|
+
self._skip_depth = 1
|
|
842
|
+
return
|
|
843
|
+
if tag in {f"h{level}" for level in range(1, 7)}:
|
|
844
|
+
self._block()
|
|
845
|
+
self._parts.append("#" * int(tag[1]) + " ")
|
|
846
|
+
elif tag in {"p", "div", "section", "article", "main", "header", "footer", "figure"}:
|
|
847
|
+
self._block()
|
|
848
|
+
elif tag == "li":
|
|
849
|
+
self._block()
|
|
850
|
+
self._parts.append("- ")
|
|
851
|
+
elif tag == "pre":
|
|
852
|
+
self._block()
|
|
853
|
+
self._pre_depth += 1
|
|
854
|
+
elif tag == "code" and not self._pre_depth:
|
|
855
|
+
self._parts.append("`")
|
|
856
|
+
elif tag == "br":
|
|
857
|
+
self._parts.append("\n")
|
|
858
|
+
elif tag == "a":
|
|
859
|
+
href = next((value for name, value in attrs if name.casefold() == "href"), None)
|
|
860
|
+
self._parts.append("[")
|
|
861
|
+
self._links.append(_safe_href(href))
|
|
862
|
+
elif tag == "img":
|
|
863
|
+
src = next((value for name, value in attrs if name.casefold() == "src"), None)
|
|
864
|
+
if isinstance(src, str):
|
|
865
|
+
image_uri = _safe_image_uri(src)
|
|
866
|
+
if image_uri is not None:
|
|
867
|
+
self._parts.append(f"")
|
|
868
|
+
|
|
869
|
+
def handle_endtag(self, tag: str) -> None:
|
|
870
|
+
tag = tag.casefold()
|
|
871
|
+
if self._skip_depth:
|
|
872
|
+
if tag in _HTML_SKIP:
|
|
873
|
+
self._skip_depth -= 1
|
|
874
|
+
return
|
|
875
|
+
if tag == "pre" and self._pre_depth:
|
|
876
|
+
self._pre_depth -= 1
|
|
877
|
+
self._block()
|
|
878
|
+
elif tag == "code" and not self._pre_depth:
|
|
879
|
+
self._parts.append("`")
|
|
880
|
+
elif tag == "a" and self._links:
|
|
881
|
+
href = self._links.pop()
|
|
882
|
+
self._parts.append(f"]({href})" if href else "]")
|
|
883
|
+
elif tag in _BLOCK_TAGS:
|
|
884
|
+
self._block()
|
|
885
|
+
|
|
886
|
+
def handle_data(self, data: str) -> None:
|
|
887
|
+
if self._skip_depth:
|
|
888
|
+
return
|
|
889
|
+
safe_data = escape(data, quote=False)
|
|
890
|
+
if self._pre_depth:
|
|
891
|
+
self._parts.append(safe_data)
|
|
892
|
+
else:
|
|
893
|
+
self._parts.append(re.sub(r"\s+", " ", safe_data))
|
|
894
|
+
|
|
895
|
+
def markdown(self) -> str:
|
|
896
|
+
return "".join(self._parts)
|
|
897
|
+
|
|
898
|
+
def _block(self) -> None:
|
|
899
|
+
if self._parts and not self._parts[-1].endswith("\n\n"):
|
|
900
|
+
self._parts.append("\n\n")
|
|
901
|
+
|
|
902
|
+
|
|
903
|
+
def _safe_href(value: str | None) -> str | None:
|
|
904
|
+
if not isinstance(value, str) or not value:
|
|
905
|
+
return None
|
|
906
|
+
parsed = urlsplit(value.strip())
|
|
907
|
+
if parsed.username is not None or parsed.password is not None:
|
|
908
|
+
return None
|
|
909
|
+
if parsed.scheme.casefold() not in {"", "http", "https"}:
|
|
910
|
+
return None
|
|
911
|
+
if parsed.scheme and not parsed.netloc:
|
|
912
|
+
return None
|
|
913
|
+
return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, "", ""))
|
|
914
|
+
|
|
915
|
+
|
|
916
|
+
def _safe_image_uri(value: str) -> str | None:
|
|
917
|
+
header, separator, payload = value.partition(",")
|
|
918
|
+
if not separator or not header.casefold().startswith("data:"):
|
|
919
|
+
return None
|
|
920
|
+
mime, separator, encoding = header[5:].partition(";")
|
|
921
|
+
if not separator or encoding.casefold() != "base64":
|
|
922
|
+
return None
|
|
923
|
+
mime = mime.casefold()
|
|
924
|
+
if mime not in _SAFE_IMAGE_MIME:
|
|
925
|
+
return None
|
|
926
|
+
encoded = "".join(payload.split())
|
|
927
|
+
if not encoded:
|
|
928
|
+
return None
|
|
929
|
+
try:
|
|
930
|
+
base64.b64decode(encoded, validate=True)
|
|
931
|
+
except (ValueError, binascii.Error):
|
|
932
|
+
return None
|
|
933
|
+
return f"data:{mime};base64,{encoded}"
|
|
934
|
+
|
|
935
|
+
|
|
936
|
+
def _render_notebook_output(output: object) -> tuple[str, str | None]:
|
|
937
|
+
if not isinstance(output, Mapping):
|
|
938
|
+
return "", "output is not an object"
|
|
939
|
+
output_type = output.get("output_type")
|
|
940
|
+
if output_type == "stream":
|
|
941
|
+
name = output.get("name") if isinstance(output.get("name"), str) else "output"
|
|
942
|
+
return f"### {name}\n\n{_fenced(_cell_text(output.get('text')), 'text')}", None
|
|
943
|
+
if output_type == "error":
|
|
944
|
+
traceback = _cell_text(output.get("traceback"))
|
|
945
|
+
if not traceback:
|
|
946
|
+
traceback = f"{output.get('ename', 'Error')}: {output.get('evalue', '')}"
|
|
947
|
+
return f"### Error\n\n{_fenced(_ANSI.sub('', traceback), 'text')}", None
|
|
948
|
+
if output_type in {"display_data", "execute_result"}:
|
|
949
|
+
data = output.get("data")
|
|
950
|
+
if not isinstance(data, Mapping):
|
|
951
|
+
return "", "output data is not an object"
|
|
952
|
+
markdown = data.get("text/markdown")
|
|
953
|
+
if markdown is not None:
|
|
954
|
+
return f"### Output\n\n{_cell_text(markdown)}", None
|
|
955
|
+
plain = data.get("text/plain")
|
|
956
|
+
if plain is not None:
|
|
957
|
+
return f"### Output\n\n{_fenced(_cell_text(plain), 'text')}", None
|
|
958
|
+
for mime, value in sorted(data.items(), key=lambda item: str(item[0])):
|
|
959
|
+
if isinstance(mime, str) and mime.startswith("image/") and isinstance(value, str):
|
|
960
|
+
image_uri = _safe_image_uri(f"data:{mime};base64,{value}")
|
|
961
|
+
if image_uri is not None:
|
|
962
|
+
return f"### Output\n\n", None
|
|
963
|
+
mime_names = ", ".join(sorted(str(name) for name in data))
|
|
964
|
+
marker = f"[a2l unsupported notebook output MIME: {mime_names}]"
|
|
965
|
+
return marker, f"unsupported output MIME: {mime_names}"
|
|
966
|
+
return f"[a2l unsupported notebook output type: {output_type!r}]", "unsupported output type"
|
|
967
|
+
|
|
968
|
+
|
|
969
|
+
def _replace_attachments(text: str, attachments: object) -> str:
|
|
970
|
+
if not isinstance(attachments, Mapping):
|
|
971
|
+
return text
|
|
972
|
+
|
|
973
|
+
def replacement(match: re.Match[str]) -> str:
|
|
974
|
+
name = match.group(1)
|
|
975
|
+
payloads = attachments.get(name)
|
|
976
|
+
if not isinstance(payloads, Mapping):
|
|
977
|
+
return match.group(0)
|
|
978
|
+
for mime, value in sorted(payloads.items(), key=lambda item: str(item[0])):
|
|
979
|
+
if isinstance(mime, str) and isinstance(value, str):
|
|
980
|
+
image_uri = _safe_image_uri(f"data:{mime};base64,{value}")
|
|
981
|
+
if image_uri is not None:
|
|
982
|
+
return image_uri
|
|
983
|
+
return match.group(0)
|
|
984
|
+
|
|
985
|
+
return _ATTACHMENT.sub(replacement, text)
|
|
986
|
+
|
|
987
|
+
|
|
988
|
+
def _fenced(text: str, language: str) -> str:
|
|
989
|
+
longest = max((len(match.group(0)) for match in _BACKTICKS.finditer(text)), default=0)
|
|
990
|
+
fence = "`" * max(3, longest + 1)
|
|
991
|
+
body = text.rstrip("\n")
|
|
992
|
+
return f"{fence}{language}\n{body}\n{fence}"
|
|
993
|
+
|
|
994
|
+
|
|
995
|
+
def _cell_text(value: object) -> str:
|
|
996
|
+
if isinstance(value, str):
|
|
997
|
+
return value
|
|
998
|
+
if isinstance(value, list) and all(isinstance(item, str) for item in value):
|
|
999
|
+
return "".join(cast(list[str], value))
|
|
1000
|
+
return ""
|
|
1001
|
+
|
|
1002
|
+
|
|
1003
|
+
def _split_all_markdown(text: str, pages: int) -> list[str] | None:
|
|
1004
|
+
if pages == 1:
|
|
1005
|
+
return [text]
|
|
1006
|
+
pieces = re.split(r"\n\s*---\s*\n", text.strip())
|
|
1007
|
+
return pieces if len(pieces) == pages else None
|
|
1008
|
+
|
|
1009
|
+
|
|
1010
|
+
def _page_block(page: int, text: str) -> str:
|
|
1011
|
+
return f"<!-- a2l:page {page} -->\n{_normalise_markdown(text).rstrip()}"
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
def _normalise_markdown(text: str) -> str:
|
|
1015
|
+
text = str(text).replace("\r\n", "\n").replace("\r", "\n")
|
|
1016
|
+
return text.rstrip() + "\n" if text.strip() else ""
|
|
1017
|
+
|
|
1018
|
+
|
|
1019
|
+
def _text_value(value: object) -> str:
|
|
1020
|
+
if isinstance(value, str):
|
|
1021
|
+
return value
|
|
1022
|
+
if isinstance(value, bytes):
|
|
1023
|
+
return value.decode("utf-8", "replace")
|
|
1024
|
+
return str(value)
|
|
1025
|
+
|
|
1026
|
+
|
|
1027
|
+
def _text_image_bytes(value: object) -> bytes:
|
|
1028
|
+
if isinstance(value, bytes):
|
|
1029
|
+
return value
|
|
1030
|
+
if isinstance(value, bytearray):
|
|
1031
|
+
return bytes(value)
|
|
1032
|
+
raise ConversionError("PDF renderer returned a non-byte image")
|
|
1033
|
+
|
|
1034
|
+
|
|
1035
|
+
def _pdf_oxide_page_count(document: object) -> int:
|
|
1036
|
+
value = getattr(document, "page_count", None)
|
|
1037
|
+
if callable(value):
|
|
1038
|
+
value = value()
|
|
1039
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
|
1040
|
+
raise ConversionError("PDF backend returned an invalid page count")
|
|
1041
|
+
return value
|
|
1042
|
+
|
|
1043
|
+
|
|
1044
|
+
def _call_method(value: object, name: str, *args: object, **kwargs: object) -> object:
|
|
1045
|
+
method = getattr(value, name, None)
|
|
1046
|
+
if not callable(method):
|
|
1047
|
+
raise ConversionError(f"PDF backend does not provide {name}")
|
|
1048
|
+
return method(*args, **kwargs)
|
|
1049
|
+
|
|
1050
|
+
|
|
1051
|
+
def _close_quietly(value: object) -> None:
|
|
1052
|
+
close = getattr(value, "close", None)
|
|
1053
|
+
if callable(close):
|
|
1054
|
+
try:
|
|
1055
|
+
close()
|
|
1056
|
+
except Exception:
|
|
1057
|
+
return
|
|
1058
|
+
|
|
1059
|
+
|
|
1060
|
+
def _configure_tesseract(language: str) -> bool:
|
|
1061
|
+
candidates: list[Path] = []
|
|
1062
|
+
found = shutil.which("tesseract")
|
|
1063
|
+
if found:
|
|
1064
|
+
candidates.append(Path(found))
|
|
1065
|
+
if os.name == "nt":
|
|
1066
|
+
program_files = os.environ.get("PROGRAMFILES")
|
|
1067
|
+
if program_files:
|
|
1068
|
+
candidates.append(Path(program_files) / "Tesseract-OCR" / "tesseract.exe")
|
|
1069
|
+
candidates.append(Path.home() / "AppData" / "Local" / "Tesseract-OCR" / "tesseract.exe")
|
|
1070
|
+
executable = next(
|
|
1071
|
+
(candidate for candidate in candidates if paths.long_path(candidate).is_file()), None
|
|
1072
|
+
)
|
|
1073
|
+
if executable is None:
|
|
1074
|
+
return False
|
|
1075
|
+
pytesseract.pytesseract.tesseract_cmd = os.fspath(executable)
|
|
1076
|
+
try:
|
|
1077
|
+
languages = pytesseract.get_languages(config="")
|
|
1078
|
+
except (OSError, pytesseract.TesseractError):
|
|
1079
|
+
return False
|
|
1080
|
+
return language in languages
|
|
1081
|
+
|
|
1082
|
+
|
|
1083
|
+
def _validate_threshold(value: int) -> None:
|
|
1084
|
+
if isinstance(value, bool) or not isinstance(value, int) or value <= 0:
|
|
1085
|
+
raise ValueError("ocr_words_per_page must be a positive integer")
|
|
1086
|
+
|
|
1087
|
+
|
|
1088
|
+
def _package_version(distribution: str, default: str) -> str:
|
|
1089
|
+
try:
|
|
1090
|
+
return importlib.metadata.version(distribution)
|
|
1091
|
+
except importlib.metadata.PackageNotFoundError:
|
|
1092
|
+
return default
|
|
1093
|
+
|
|
1094
|
+
|
|
1095
|
+
def _expected_tool(source: Path, backend: ConverterBackend) -> tuple[str, str]:
|
|
1096
|
+
return _expected_tool_for_kind(_classify_source(source), backend)
|
|
1097
|
+
|
|
1098
|
+
|
|
1099
|
+
def _expected_tool_for_kind(kind: str, backend: ConverterBackend) -> tuple[str, str]:
|
|
1100
|
+
if kind == "pdf":
|
|
1101
|
+
return backend.name, backend.version
|
|
1102
|
+
if kind == "notebook":
|
|
1103
|
+
return "nbformat", _package_version("nbformat", "missing")
|
|
1104
|
+
if kind == "html_zip":
|
|
1105
|
+
return "html-archive", CONVERTER_VERSION
|
|
1106
|
+
if kind == "html":
|
|
1107
|
+
return "html-sanitizer", CONVERTER_VERSION
|
|
1108
|
+
if kind == "office":
|
|
1109
|
+
return "markitdown", _package_version("markitdown", "missing")
|
|
1110
|
+
return "agent2learn-text", CONVERTER_VERSION
|
|
1111
|
+
|
|
1112
|
+
|
|
1113
|
+
def _artifact_is_current(
|
|
1114
|
+
vault: Vault,
|
|
1115
|
+
artifact: DerivedArtifact,
|
|
1116
|
+
entry: ManifestEntry,
|
|
1117
|
+
expected_tool: str,
|
|
1118
|
+
expected_version: str,
|
|
1119
|
+
expected_threshold: int | None,
|
|
1120
|
+
) -> bool:
|
|
1121
|
+
# Ingest renders assignment prompts with title context that generic HTML conversion does not
|
|
1122
|
+
# have. Preserve that specialized twin, but only after the path, tool, version, and hashes are
|
|
1123
|
+
# all checked; every other HTML artifact still has to match the generic converter exactly.
|
|
1124
|
+
tool_is_current = artifact.tool == expected_tool and artifact.tool_version == expected_version
|
|
1125
|
+
if not tool_is_current and not (
|
|
1126
|
+
expected_tool == "html-sanitizer"
|
|
1127
|
+
and artifact.tool == RICH_TEXT_TOOL
|
|
1128
|
+
and artifact.tool_version == RICH_TEXT_TOOL_VERSION
|
|
1129
|
+
and _is_assignment_prompt_artifact(entry, artifact)
|
|
1130
|
+
):
|
|
1131
|
+
return False
|
|
1132
|
+
if artifact.source_sha256 != entry.sha256 or artifact.ocr_words_per_page != expected_threshold:
|
|
1133
|
+
return False
|
|
1134
|
+
path = _artifact_path(vault, artifact)
|
|
1135
|
+
return paths.long_path(path).is_file() and _hash_file(path)[0] == artifact.sha256
|
|
1136
|
+
|
|
1137
|
+
|
|
1138
|
+
def _is_assignment_prompt_artifact(entry: ManifestEntry, artifact: DerivedArtifact) -> bool:
|
|
1139
|
+
"""Recognize only the ingest-owned HTML/Markdown pair used for an assignment prompt."""
|
|
1140
|
+
|
|
1141
|
+
source = PurePosixPath(entry.path)
|
|
1142
|
+
derived = PurePosixPath(artifact.path)
|
|
1143
|
+
return (
|
|
1144
|
+
re.fullmatch(r"instructions(?:_\d+)?\.html", source.name.casefold()) is not None
|
|
1145
|
+
and any(part.casefold() == "assignments" for part in source.parts)
|
|
1146
|
+
and derived.parent == source.parent
|
|
1147
|
+
and derived.suffix.casefold() == ".md"
|
|
1148
|
+
)
|
|
1149
|
+
|
|
1150
|
+
|
|
1151
|
+
def _artifact_path(vault: Vault, artifact: DerivedArtifact) -> Path:
|
|
1152
|
+
return vault.materialized(
|
|
1153
|
+
ManifestEntry(
|
|
1154
|
+
path=artifact.path,
|
|
1155
|
+
sha256=artifact.sha256,
|
|
1156
|
+
source_id="derived",
|
|
1157
|
+
etag=None,
|
|
1158
|
+
last_modified=None,
|
|
1159
|
+
size=0,
|
|
1160
|
+
fetched_at="2026-01-01T00:00:00Z",
|
|
1161
|
+
)
|
|
1162
|
+
)
|
|
1163
|
+
|
|
1164
|
+
|
|
1165
|
+
def _hash_file(source: Path) -> tuple[str, int]:
|
|
1166
|
+
from hashlib import sha256
|
|
1167
|
+
|
|
1168
|
+
digest = sha256()
|
|
1169
|
+
size = 0
|
|
1170
|
+
try:
|
|
1171
|
+
with open(os.fspath(paths.long_path(source)), "rb") as handle:
|
|
1172
|
+
while chunk := handle.read(1024 * 1024):
|
|
1173
|
+
digest.update(chunk)
|
|
1174
|
+
size += len(chunk)
|
|
1175
|
+
except (FileNotFoundError, IsADirectoryError, OSError):
|
|
1176
|
+
return "", -1
|
|
1177
|
+
return digest.hexdigest(), size
|
|
1178
|
+
|
|
1179
|
+
|
|
1180
|
+
def _update_content_map(vault: Vault, key: str, **updates: object) -> None:
|
|
1181
|
+
for destination in sorted(
|
|
1182
|
+
path for path in paths.walk(vault.root) if path.name == "content_map.json"
|
|
1183
|
+
):
|
|
1184
|
+
try:
|
|
1185
|
+
raw = course_index.read_content_map(destination.parent.parent)
|
|
1186
|
+
except (A2LError, UnicodeError):
|
|
1187
|
+
continue
|
|
1188
|
+
changed = False
|
|
1189
|
+
raw_rows = raw.get("topics")
|
|
1190
|
+
if not isinstance(raw_rows, list):
|
|
1191
|
+
continue
|
|
1192
|
+
rows: list[object] = raw_rows
|
|
1193
|
+
for row in rows:
|
|
1194
|
+
if isinstance(row, dict) and row.get("source_key") == key:
|
|
1195
|
+
row.update(updates)
|
|
1196
|
+
changed = True
|
|
1197
|
+
if changed:
|
|
1198
|
+
checked = course_index.reconcile_content_map(vault, rows)
|
|
1199
|
+
course_index.write_content_map(destination.parent.parent, checked, root=vault.root)
|
|
1200
|
+
|
|
1201
|
+
|
|
1202
|
+
def _now() -> str:
|
|
1203
|
+
return clock.stamp()
|
|
1204
|
+
|
|
1205
|
+
|
|
1206
|
+
__all__ = [
|
|
1207
|
+
"CONVERTER_VERSION",
|
|
1208
|
+
"DEFAULT_OCR_WORDS_PER_PAGE",
|
|
1209
|
+
"MIN_PDF_CHARS",
|
|
1210
|
+
"OCR_SETUP_ACTION",
|
|
1211
|
+
"ConversionError",
|
|
1212
|
+
"ConversionReport",
|
|
1213
|
+
"ConversionResult",
|
|
1214
|
+
"ConverterBackend",
|
|
1215
|
+
"PageCoverage",
|
|
1216
|
+
"PdfOxideBackend",
|
|
1217
|
+
"PdfiumBackend",
|
|
1218
|
+
"convert_html_zip",
|
|
1219
|
+
"convert_pdf",
|
|
1220
|
+
"convert_source",
|
|
1221
|
+
"convert_vault",
|
|
1222
|
+
"render_notebook",
|
|
1223
|
+
]
|