corpora-py 0.1.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. admin/__init__.py +22 -0
  2. admin/converters/CLAUDE.md +62 -0
  3. admin/converters/__init__.py +44 -0
  4. admin/converters/_epub_to_tf.py +51 -0
  5. admin/converters/_html_to_tf.py +43 -0
  6. admin/converters/_pdf_to_tf.py +37 -0
  7. admin/converters/_tei_to_tf.py +53 -0
  8. admin/converters/_tei_zip_to_tf.py +94 -0
  9. admin/converters/_text_to_tf.py +36 -0
  10. admin/converters/_tf_zip_to_tf.py +87 -0
  11. admin/converters/_walker.py +182 -0
  12. admin/converters/convert_to_cfm.py +36 -0
  13. admin/converters/convert_to_corpus.py +219 -0
  14. admin/ingest/__init__.py +32 -0
  15. admin/ingest/_ids.py +25 -0
  16. admin/ingest/docling_graph.py +775 -0
  17. admin/ingest/validation.py +87 -0
  18. admin/parsers/__init__.py +34 -0
  19. admin/parsers/_epub.py +68 -0
  20. admin/parsers/_html.py +107 -0
  21. admin/parsers/_pdf.py +51 -0
  22. admin/parsers/_plain.py +40 -0
  23. admin/parsers/_tei.py +73 -0
  24. admin/parsers/_xml.py +49 -0
  25. admin/parsers/schema.py +164 -0
  26. admin/services/CLAUDE.md +127 -0
  27. admin/services/__init__.py +50 -0
  28. admin/services/api.py +281 -0
  29. admin/services/corpus_detail.py +637 -0
  30. admin/services/corpus_detail_api.py +153 -0
  31. admin/services/corpus_detail_mcp.py +228 -0
  32. admin/services/ingest_api.py +192 -0
  33. admin/services/jobs.py +332 -0
  34. admin/services/storage.py +309 -0
  35. admin/services/storage_api.py +170 -0
  36. admin/services/storage_mcp.py +110 -0
  37. admin/services/validation_api.py +157 -0
  38. admin/services/websocket.py +76 -0
  39. common/__init__.py +0 -0
  40. common/schemas/context_fabric/v1/annotation.schema.json +64 -0
  41. common/schemas/context_fabric/v1/api-payloads.schema.json +155 -0
  42. common/schemas/context_fabric/v1/common.defs.schema.json +163 -0
  43. common/schemas/context_fabric/v1/content-node.schema.json +69 -0
  44. common/schemas/context_fabric/v1/corpus.schema.json +31 -0
  45. common/schemas/context_fabric/v1/edition.schema.json +74 -0
  46. common/schemas/context_fabric/v1/examples/academic-paper.json +154 -0
  47. common/schemas/context_fabric/v1/examples/book-chapter.json +91 -0
  48. common/schemas/context_fabric/v1/examples/index.json +11 -0
  49. common/schemas/context_fabric/v1/examples/letter-page.json +97 -0
  50. common/schemas/context_fabric/v1/examples/scripture-node.json +89 -0
  51. common/schemas/context_fabric/v1/examples/speech.json +73 -0
  52. common/schemas/context_fabric/v1/examples/transcript.json +129 -0
  53. common/schemas/context_fabric/v1/physical-locator.schema.json +51 -0
  54. common/schemas/context_fabric/v1/reference.schema.json +56 -0
  55. common/schemas/context_fabric/v1/relationship.schema.json +28 -0
  56. common/schemas/context_fabric/v1/source-asset.schema.json +32 -0
  57. common/schemas/context_fabric/v1/text-fragment.schema.json +44 -0
  58. common/schemas/context_fabric/v1/work.schema.json +37 -0
  59. common/utils/__init__.py +9 -0
  60. common/utils/config.py +95 -0
  61. common/utils/console.py +34 -0
  62. common/utils/constant.py +36 -0
  63. common/utils/helpers.py +79 -0
  64. common/utils/jwt_auth.py +64 -0
  65. corpora_mcp/__init__.py +9 -0
  66. corpora_mcp/__main__.py +4 -0
  67. corpora_mcp/corpus.py +90 -0
  68. corpora_mcp/server.py +780 -0
  69. corpora_mcp/validate.py +559 -0
  70. corpora_py/__init__.py +12 -0
  71. corpora_py/app.py +191 -0
  72. corpora_py/auth.py +108 -0
  73. corpora_py/bridge.py +39 -0
  74. corpora_py/validation_api.py +124 -0
  75. corpora_py-0.1.3.dist-info/METADATA +332 -0
  76. corpora_py-0.1.3.dist-info/RECORD +79 -0
  77. corpora_py-0.1.3.dist-info/WHEEL +4 -0
  78. corpora_py-0.1.3.dist-info/entry_points.txt +3 -0
  79. corpora_py-0.1.3.dist-info/licenses/LICENSE +21 -0
admin/__init__.py ADDED
@@ -0,0 +1,22 @@
1
+ """admin — admin-only / full-featured tooling.
2
+
3
+ Houses conversion pipelines (EPUB/HTML → Text-Fabric, TF → .exg packaging)
4
+ and other heavy or privileged operations. These typically require the
5
+ [full] extra (text-fabric) and are not needed in the slim client runtime.
6
+ """
7
+
8
+ from . import converters, parsers, services
9
+
10
+ __all__ = ["converters", "parsers", "services"]
11
+
12
+ try:
13
+ # admin.ingest imports docling-core (pandas, pillow) at module import
14
+ # time. Slim serverless bundles (Vercel — see vercel.json) exclude that
15
+ # chain to stay under the function size limit, so its absence must not
16
+ # break `import admin`; /ingest degrades to 503 there (see
17
+ # services/ingest_api.py).
18
+ from . import ingest # noqa: F401
19
+
20
+ __all__.append("ingest")
21
+ except ModuleNotFoundError: # pragma: no cover - exercised only in slim deploys
22
+ pass
@@ -0,0 +1,62 @@
1
+ # CLAUDE.md — `admin.converters`
2
+
3
+ `Document`/`Unit` tree → Text-Fabric dataset → `.cfm` cache → `.corpus` archive. See
4
+ `packages/admin/CLAUDE.md` for the shared-schema/shared-walker architecture that makes this package
5
+ possible; this file covers the Text-Fabric, Context-Fabric, and `.corpus` details specific to
6
+ `converters/`.
7
+
8
+ Read `_walker.py` before touching any feature/node-creation logic — it's the one TF walk reused by
9
+ every `_{format}_to_tf.py`.
10
+
11
+ ## Text-Fabric walker gotchas (all handled in `_walker.py`)
12
+
13
+ - **Every feature name must have metadata**, or `cv.walk()` fails validation with `"node feature has no metadata"`.
14
+ Feature names here vary per document (HTML attributes, TEI `@type`, ...) so they can't be declared upfront in
15
+ `featureMeta=`; `set_features()` registers each one dynamically via `cv.meta(name, valueType="str")` right before
16
+ setting it. If you add a
17
+ `cv.feature()` call anywhere, route it through `set_features()`, not
18
+ `cv.feature()` directly, or you'll reintroduce this failure.
19
+ - **Dynamically-registered features need an explicit `valueType`** or the exporter warns `"Missing @valueType"`
20
+ (non-fatal, but avoidable — that's why `set_features()` always passes `valueType="str"`).
21
+ - **A node covering zero slots gets silently deleted** by Text-Fabric's
22
+ "remove unlinked nodes" pass. A leaf `Unit` with no tokens and no children (a blank PDF page, an `<img>`, an `<hr>`)
23
+ would otherwise vanish along with its attributes — `_walk_unit()` gives genuinely empty leaves one placeholder
24
+ empty-text slot so they survive.
25
+ - **`otext.sectionTypes`/`sectionFeatures` can't be empty**, but also don't need to be elaborate: every converter uses a
26
+ single section level (the root `book`/`document`/`text` node, with `title` as its section feature). Finer structure
27
+ (chapters, pages, divs) is still expressed as ordinary node types via `otype_for` — it just isn't declared as TF
28
+ "sections", which would require strict, consistent nesting we can't guarantee across arbitrary source documents.
29
+ - **`SKIP_TAGS` in `parsers/_html.py` is scoped to tags that only make sense to drop when nested inside `<body>`**
30
+ (script/style/noscript/svg/math). It used to include `"head"` for HTML's metadata tag, which silently ate TEI's
31
+ `<head>` (a heading element, reused by the shared walker) — don't add HTML-specific tag names back to that set without
32
+ checking what they mean in TEI/XML first.
33
+
34
+ ## Context-Fabric (`cfabric`) notes
35
+
36
+ - There is **no separate compile API**. `.cfm` compilation happens automatically the first time a dataset is loaded via
37
+ `cfabric.Fabric(locations=...).loadAll()`. `convert_to_cfm()` exists only to trigger that load on purpose and hand
38
+ back the resulting `.cfm` path.
39
+ - `Fabric(...).loadAll()` returns `Api | bool` (`False` on failure) — always narrow with `isinstance(result, bool)`
40
+ before touching `.F`/`.T`/`.Fall()`; those are dynamically populated at load time so mypy can't see their attributes
41
+ either (hence the `type: ignore[attr-defined]` in
42
+ `convert_to_corpus.py`).
43
+
44
+ ## `.corpus` archive
45
+
46
+ The archive format (`manifest.yml`, `toc.yml`, `assets/`, `.git/`,
47
+ `corpora/{*.tf, .cfm/}`) is the contract both the Corpora and Exegia apps parse. The canonical spec is maintained in an
48
+ external vault by the Corpora team. Before changing manifest/toc shape in `convert_to_corpus.py`, consult the current
49
+ schema definition with the team or check the app's schema loader to understand the expected format.
50
+
51
+ ## Known gaps (converter-side)
52
+
53
+ Service-side gaps (progress reporting, job registry, archive cleanup) live in
54
+ `src/admin/services/CLAUDE.md`.
55
+
56
+ - **No `_xml_to_tf.py`.** `XmlParser` exists (`admin.parsers`) but there's no matching Text-Fabric converter — generic
57
+ XML has no fixed node-type vocabulary to map onto, unlike TEI's `<div>`/`<p>` convention. Add one the same way as the
58
+ others (pick an `otype_for`, wire it into
59
+ `converters/__init__.py`'s `CONVERTERS`) if a concrete need shows up; don't add it speculatively.
60
+ - `dataset_id`/`project_id`/`publisher_id`/`author_ids` in
61
+ `convert_to_corpus()` are caller-supplied and default to `""` — this package has no way to know them; they're assigned
62
+ by whatever backend calls it (the Corpora/Exegia app, not this converter).
@@ -0,0 +1,44 @@
1
+ """admin.converters — conversion and packaging tools (admin / full runtime).
2
+
3
+ These modules depend on optional heavy deps (text-fabric, context-fabric,
4
+ under the `[full]` extra) and are intended for dataset ingestion / admin
5
+ tooling only.
6
+
7
+ Pipeline: a `_{format}_to_tf.py` converter turns a source document into a
8
+ Text-Fabric dataset; `convert_to_cfm` compiles that into Context-Fabric's
9
+ `.cfm` cache; `convert_to_corpus` packages both into a `.corpus` archive.
10
+ """
11
+
12
+ from ..parsers import SourceFormat
13
+ from ._epub_to_tf import convert_epub_to_tf
14
+ from ._html_to_tf import convert_html_to_tf
15
+ from ._pdf_to_tf import convert_pdf_to_tf
16
+ from ._tei_to_tf import convert_tei_to_tf
17
+ from ._tei_zip_to_tf import convert_tei_zip_to_tf
18
+ from ._text_to_tf import convert_text_to_tf
19
+ from ._tf_zip_to_tf import convert_tf_zip_to_tf
20
+ from .convert_to_cfm import convert_to_cfm
21
+ from .convert_to_corpus import convert_to_corpus
22
+
23
+ CONVERTERS = {
24
+ SourceFormat.EPUB: convert_epub_to_tf,
25
+ SourceFormat.HTML: convert_html_to_tf,
26
+ SourceFormat.PDF: convert_pdf_to_tf,
27
+ SourceFormat.TEI: convert_tei_to_tf,
28
+ SourceFormat.TEI_ZIP: convert_tei_zip_to_tf,
29
+ SourceFormat.PLAIN: convert_text_to_tf,
30
+ SourceFormat.TF_ZIP: convert_tf_zip_to_tf,
31
+ }
32
+
33
+ __all__ = [
34
+ "CONVERTERS",
35
+ "convert_epub_to_tf",
36
+ "convert_html_to_tf",
37
+ "convert_pdf_to_tf",
38
+ "convert_tei_to_tf",
39
+ "convert_tei_zip_to_tf",
40
+ "convert_text_to_tf",
41
+ "convert_tf_zip_to_tf",
42
+ "convert_to_cfm",
43
+ "convert_to_corpus",
44
+ ]
@@ -0,0 +1,51 @@
1
+ """
2
+ EPUB to Text-Fabric Converter
3
+
4
+ Converts EPUB ebook files into Text-Fabric datasets using the epub service
5
+ for parsing and the tf.convert.walker library for TF generation.
6
+
7
+ Features:
8
+ - Extracts EPUB metadata (title, author, publisher, etc.)
9
+ - Preserves book structure (chapters/pages)
10
+ - Converts HTML content to queryable nodes
11
+ - Creates semantic nodes from clean HTML
12
+ - Tracks conversion progress
13
+
14
+ Node Types:
15
+ - book: Root node for the entire EPUB
16
+ - chapter: Individual pages/chapters from the EPUB
17
+ - element: HTML elements from page content
18
+ - paragraph: Paragraph-like elements
19
+ - link: Link elements with href
20
+ - word: Individual words (slots)
21
+ """
22
+
23
+ from pathlib import Path
24
+
25
+ from ..parsers import EpubParser
26
+ from ..parsers.schema import Unit
27
+ from ._walker import convert_document
28
+
29
+ _PARAGRAPH_TAGS = {"p", "blockquote"}
30
+ _LINK_TAGS = {"a"}
31
+
32
+
33
+ def _otype_for(unit: Unit) -> str:
34
+ if unit.type == "chapter":
35
+ return "chapter"
36
+ if unit.type in _PARAGRAPH_TAGS:
37
+ return "paragraph"
38
+ if unit.type in _LINK_TAGS:
39
+ return "link"
40
+ return "element"
41
+
42
+
43
+ def convert_epub_to_tf(source: str, output_dir: str | Path) -> Path:
44
+ """Convert an EPUB at `source` (path or URL) into a Text-Fabric dataset."""
45
+ return convert_document(
46
+ EpubParser(),
47
+ source,
48
+ output_dir,
49
+ root_type="book",
50
+ otype_for=_otype_for,
51
+ )
@@ -0,0 +1,43 @@
1
+ """
2
+ HTML to Text-Fabric Converter
3
+
4
+ Converts HTML documents into Text-Fabric (TF) datasets using the tf.convert.walker library.
5
+ This allows HTML content to be queried using Context-Fabric's powerful graph query API.
6
+
7
+ Features:
8
+ - Preserves HTML structure as a node hierarchy
9
+ - Creates slots for text content (words/tokens)
10
+ - Stores HTML attributes as node features
11
+ - Supports nested HTML elements
12
+ - Generates valid Text-Fabric datasets
13
+
14
+ Node Types:
15
+ - document: Root node for each HTML document
16
+ - element: HTML tags (div, p, span, etc.)
17
+ - word: Individual words (slots)
18
+
19
+ Features:
20
+ - tag: HTML tag name (div, p, span, etc.)
21
+ - class: CSS class names
22
+ - id: HTML id attribute
23
+ - href: Link URLs (for <a> tags)
24
+ - src: Source URLs (for <img>, <script> tags)
25
+ - text: Raw text content
26
+ - * (any HTML attribute preserved as feature)
27
+ """
28
+
29
+ from pathlib import Path
30
+
31
+ from ..parsers import HtmlParser
32
+ from ._walker import convert_document
33
+
34
+
35
+ def convert_html_to_tf(source: str, output_dir: str | Path) -> Path:
36
+ """Convert an HTML document at `source` (path or URL) into a Text-Fabric dataset."""
37
+ return convert_document(
38
+ HtmlParser(),
39
+ source,
40
+ output_dir,
41
+ root_type="document",
42
+ otype_for=lambda unit: "element",
43
+ )
@@ -0,0 +1,37 @@
1
+ """
2
+ PDF to Text-Fabric Converter
3
+
4
+ Converts PDF files into Text-Fabric datasets using the pdf parser for
5
+ extraction and the tf.convert.walker library for TF generation.
6
+
7
+ Features:
8
+ - Extracts PDF metadata (title, author, producer, etc.)
9
+ - One node per page, so a large PDF converts one page at a time
10
+ - Tracks conversion progress
11
+
12
+ Node Types:
13
+ - book: Root node for the entire PDF
14
+ - page: Individual pages
15
+ - word: Individual words (slots)
16
+
17
+ Features:
18
+ - title, creators, date, subjects, ...: document metadata
19
+ - page_number: 1-based page number
20
+ - text, after: word text and its trailing whitespace
21
+ """
22
+
23
+ from pathlib import Path
24
+
25
+ from ..parsers import PdfParser
26
+ from ._walker import convert_document
27
+
28
+
29
+ def convert_pdf_to_tf(source: str, output_dir: str | Path) -> Path:
30
+ """Convert a PDF at `source` (path or URL) into a Text-Fabric dataset."""
31
+ return convert_document(
32
+ PdfParser(),
33
+ source,
34
+ output_dir,
35
+ root_type="book",
36
+ otype_for=lambda unit: "page",
37
+ )
@@ -0,0 +1,53 @@
1
+ """
2
+ TEI to Text-Fabric Converter
3
+
4
+ Converts TEI (Text Encoding Initiative) XML documents into Text-Fabric
5
+ datasets using the tei parser for extraction and the tf.convert.walker
6
+ library for TF generation.
7
+
8
+ Features:
9
+ - Extracts teiHeader metadata (title, author, publisher, date, ...)
10
+ - Preserves the TEI division hierarchy (<div> nesting)
11
+ - Lifts each division's <head> into a `label` feature
12
+ - One node per top-level division, so a large edition (a multi-book bible,
13
+ a multi-volume critical edition) converts one division at a time
14
+
15
+ Node Types:
16
+ - text: Root node for the entire TEI document
17
+ - div: A TEI division (chapter, book, ...); its `type` attribute survives
18
+ as an ordinary feature (e.g. `type="chapter"`)
19
+ - paragraph: <p> elements
20
+ - element: Any other TEI element (l, seg, note, ...)
21
+ - word: Individual words (slots)
22
+
23
+ Features:
24
+ - title, creators, language, publisher, date, identifier, rights: document metadata
25
+ - label: division heading, lifted from its <head> child
26
+ - type: TEI `@type` attribute (e.g. div type="chapter")
27
+ - text, after: word text and its trailing whitespace
28
+ """
29
+
30
+ from pathlib import Path
31
+
32
+ from ..parsers import TeiParser
33
+ from ..parsers.schema import Unit
34
+ from ._walker import convert_document
35
+
36
+
37
+ def _otype_for(unit: Unit) -> str:
38
+ if unit.type == "div":
39
+ return "div"
40
+ if unit.type == "p":
41
+ return "paragraph"
42
+ return "element"
43
+
44
+
45
+ def convert_tei_to_tf(source: str, output_dir: str | Path) -> Path:
46
+ """Convert a TEI document at `source` (path or URL) into a Text-Fabric dataset."""
47
+ return convert_document(
48
+ TeiParser(),
49
+ source,
50
+ output_dir,
51
+ root_type="text",
52
+ otype_for=_otype_for,
53
+ )
@@ -0,0 +1,94 @@
1
+ """
2
+ TEI ZIP to Text-Fabric Converter
3
+
4
+ Converts a ZIP archive of TEI (or TEI-in-plain-`.xml`) documents into a
5
+ single Text-Fabric dataset: each member document is parsed with the same
6
+ `TeiParser` the single-file `tei` converter uses, and all of them are walked
7
+ into one dataset via `convert_documents` -- one "text" root node per member
8
+ document, carrying that document's `<teiHeader>` metadata as features.
9
+ Members convert in archive-path order (sorted), so a corpus zipped as
10
+ `01-genesis.xml`, `02-exodus.xml`, ... keeps its intended sequence.
11
+
12
+ Node Types / Features: identical to `_tei_to_tf.py` (same parser, same
13
+ `otype_for`) -- the only difference is that the dataset can contain several
14
+ "text" roots instead of exactly one.
15
+
16
+ Archive safety mirrors `_tf_zip_to_tf.py`: path traversal, symlinks,
17
+ encrypted members, member count, and expanded size are all validated before
18
+ any bytes are written. macOS `__MACOSX/` resource forks and hidden dotfiles
19
+ are ignored.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import shutil
25
+ import tempfile
26
+ from pathlib import Path, PurePosixPath
27
+ from zipfile import BadZipFile, ZipFile, ZipInfo
28
+
29
+ from ..parsers import TeiParser
30
+ from ..parsers.schema import Document
31
+ from ._tei_to_tf import _otype_for
32
+ from ._tf_zip_to_tf import _MAX_FILES, _MAX_UNCOMPRESSED_BYTES, _safe_path
33
+ from ._walker import convert_documents
34
+
35
+ _TEI_SUFFIXES = frozenset({".tei", ".xml"})
36
+
37
+
38
+ def _is_noise(path: PurePosixPath) -> bool:
39
+ """Housekeeping entries that say nothing about the corpus itself."""
40
+ return path.parts[0] == "__MACOSX" or path.name.startswith(".")
41
+
42
+
43
+ def _tei_members(archive: ZipFile) -> list[ZipInfo]:
44
+ infos = [info for info in archive.infolist() if not info.is_dir()]
45
+ if len(infos) > _MAX_FILES:
46
+ raise ValueError(f"TEI ZIP contains more than {_MAX_FILES:,} files")
47
+ if sum(info.file_size for info in infos) > _MAX_UNCOMPRESSED_BYTES:
48
+ raise ValueError("Expanded TEI ZIP exceeds the 2 GiB limit")
49
+
50
+ members = [
51
+ info
52
+ for info in infos
53
+ for path in (_safe_path(info),)
54
+ if not _is_noise(path) and path.suffix.lower() in _TEI_SUFFIXES
55
+ ]
56
+ if not members:
57
+ raise ValueError(
58
+ "ZIP does not contain any TEI documents (.tei or .xml files are required)"
59
+ )
60
+ return sorted(members, key=lambda info: info.filename)
61
+
62
+
63
+ def convert_tei_zip_to_tf(source: str, output_dir: str | Path) -> Path:
64
+ """Convert every TEI/XML document inside ``source`` into one Text-Fabric
65
+ dataset at ``output_dir``."""
66
+ parser = TeiParser()
67
+ documents: list[Document] = []
68
+
69
+ try:
70
+ with (
71
+ ZipFile(source) as archive,
72
+ tempfile.TemporaryDirectory(prefix="tei-zip-") as scratch,
73
+ ):
74
+ for info in _tei_members(archive):
75
+ # Flattened to the basename: members were validated against
76
+ # traversal above, and the parser only needs a readable file.
77
+ extracted = Path(scratch) / f"{len(documents)}-{PurePosixPath(info.filename).name}"
78
+ with (
79
+ archive.open(info) as member,
80
+ extracted.open("wb") as out,
81
+ ):
82
+ shutil.copyfileobj(member, out)
83
+ documents.append(parser.parse(str(extracted)))
84
+ except BadZipFile as exc:
85
+ raise ValueError("Uploaded file is not a valid ZIP archive") from exc
86
+
87
+ return convert_documents(
88
+ documents,
89
+ output_dir,
90
+ root_type="text",
91
+ otype_for=_otype_for,
92
+ format_value=parser.format.value,
93
+ source_label=source,
94
+ )
@@ -0,0 +1,36 @@
1
+ """
2
+ Plain Text to Text-Fabric Converter
3
+
4
+ Converts raw text files into Text-Fabric datasets using the plain-text
5
+ parser for extraction and the tf.convert.walker library for TF generation.
6
+
7
+ Features:
8
+ - One node per paragraph (blank-line-separated), so a large plain-text
9
+ corpus converts one paragraph at a time
10
+ - Minimal metadata (a title derived from the file name)
11
+
12
+ Node Types:
13
+ - book: Root node for the entire text file
14
+ - paragraph: A blank-line-separated block of text
15
+ - word: Individual words (slots)
16
+
17
+ Features:
18
+ - title: derived from the source file name
19
+ - text, after: word text and its trailing whitespace
20
+ """
21
+
22
+ from pathlib import Path
23
+
24
+ from ..parsers import PlainTextParser
25
+ from ._walker import convert_document
26
+
27
+
28
+ def convert_text_to_tf(source: str, output_dir: str | Path) -> Path:
29
+ """Convert a plain-text file at `source` (path or URL) into a Text-Fabric dataset."""
30
+ return convert_document(
31
+ PlainTextParser(),
32
+ source,
33
+ output_dir,
34
+ root_type="book",
35
+ otype_for=lambda unit: "paragraph",
36
+ )
@@ -0,0 +1,87 @@
1
+ """Import an existing Text-Fabric dataset from a ZIP archive."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import shutil
6
+ import stat
7
+ from pathlib import Path, PurePosixPath
8
+ from zipfile import BadZipFile, ZipFile, ZipInfo
9
+
10
+ _MAX_FILES = 10_000
11
+ _MAX_UNCOMPRESSED_BYTES = 2 * 1024 * 1024 * 1024
12
+ _REQUIRED_FILES = frozenset({"otype.tf", "oslots.tf"})
13
+
14
+
15
+ def _safe_path(info: ZipInfo) -> PurePosixPath:
16
+ path = PurePosixPath(info.filename)
17
+ mode = info.external_attr >> 16
18
+ if path.is_absolute() or ".." in path.parts:
19
+ raise ValueError(f"Unsafe path in Text-Fabric ZIP: {info.filename!r}")
20
+ if stat.S_ISLNK(mode):
21
+ raise ValueError(
22
+ f"Symbolic links are not allowed in Text-Fabric ZIPs: {info.filename!r}"
23
+ )
24
+ if info.flag_bits & 0x1:
25
+ raise ValueError("Encrypted Text-Fabric ZIPs are not supported")
26
+ return path
27
+
28
+
29
+ def _find_dataset_root(files: dict[PurePosixPath, ZipInfo]) -> PurePosixPath:
30
+ candidates: list[PurePosixPath] = []
31
+ for path in files:
32
+ if path.name != "otype.tf":
33
+ continue
34
+ if all(path.parent / required in files for required in _REQUIRED_FILES):
35
+ candidates.append(path.parent)
36
+
37
+ if not candidates:
38
+ raise ValueError(
39
+ "ZIP does not contain a Text-Fabric dataset (otype.tf and oslots.tf are required)"
40
+ )
41
+ if len(candidates) > 1:
42
+ roots = ", ".join(str(path) for path in sorted(candidates, key=str))
43
+ raise ValueError(f"ZIP contains multiple Text-Fabric datasets: {roots}")
44
+ return candidates[0]
45
+
46
+
47
+ def convert_tf_zip_to_tf(source: str, output_dir: str | Path) -> Path:
48
+ """Extract the single Text-Fabric dataset in ``source`` to ``output_dir``.
49
+
50
+ Archive paths, symlinks, member count, and expanded size are validated
51
+ before any bytes are written so a malformed upload cannot escape or fill
52
+ the conversion work directory.
53
+ """
54
+ output_dir = Path(output_dir)
55
+
56
+ try:
57
+ with ZipFile(source) as archive:
58
+ infos = [info for info in archive.infolist() if not info.is_dir()]
59
+ if len(infos) > _MAX_FILES:
60
+ raise ValueError(
61
+ f"Text-Fabric ZIP contains more than {_MAX_FILES:,} files"
62
+ )
63
+ if sum(info.file_size for info in infos) > _MAX_UNCOMPRESSED_BYTES:
64
+ raise ValueError("Expanded Text-Fabric ZIP exceeds the 2 GiB limit")
65
+
66
+ files = {_safe_path(info): info for info in infos}
67
+ dataset_root = _find_dataset_root(files)
68
+ dataset_files = {
69
+ path: info
70
+ for path, info in files.items()
71
+ if path.is_relative_to(dataset_root) and path.suffix == ".tf"
72
+ }
73
+
74
+ output_dir.mkdir(parents=True, exist_ok=False)
75
+ for path, info in dataset_files.items():
76
+ relative = path.relative_to(dataset_root)
77
+ destination = output_dir.joinpath(*relative.parts)
78
+ destination.parent.mkdir(parents=True, exist_ok=True)
79
+ with (
80
+ archive.open(info) as source_file,
81
+ destination.open("wb") as output_file,
82
+ ):
83
+ shutil.copyfileobj(source_file, output_file)
84
+ except BadZipFile as exc:
85
+ raise ValueError("Uploaded file is not a valid ZIP archive") from exc
86
+
87
+ return output_dir