corpora-py 0.1.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- admin/__init__.py +22 -0
- admin/converters/CLAUDE.md +62 -0
- admin/converters/__init__.py +44 -0
- admin/converters/_epub_to_tf.py +51 -0
- admin/converters/_html_to_tf.py +43 -0
- admin/converters/_pdf_to_tf.py +37 -0
- admin/converters/_tei_to_tf.py +53 -0
- admin/converters/_tei_zip_to_tf.py +94 -0
- admin/converters/_text_to_tf.py +36 -0
- admin/converters/_tf_zip_to_tf.py +87 -0
- admin/converters/_walker.py +182 -0
- admin/converters/convert_to_cfm.py +36 -0
- admin/converters/convert_to_corpus.py +219 -0
- admin/ingest/__init__.py +32 -0
- admin/ingest/_ids.py +25 -0
- admin/ingest/docling_graph.py +775 -0
- admin/ingest/validation.py +87 -0
- admin/parsers/__init__.py +34 -0
- admin/parsers/_epub.py +68 -0
- admin/parsers/_html.py +107 -0
- admin/parsers/_pdf.py +51 -0
- admin/parsers/_plain.py +40 -0
- admin/parsers/_tei.py +73 -0
- admin/parsers/_xml.py +49 -0
- admin/parsers/schema.py +164 -0
- admin/services/CLAUDE.md +127 -0
- admin/services/__init__.py +50 -0
- admin/services/api.py +281 -0
- admin/services/corpus_detail.py +637 -0
- admin/services/corpus_detail_api.py +153 -0
- admin/services/corpus_detail_mcp.py +228 -0
- admin/services/ingest_api.py +192 -0
- admin/services/jobs.py +332 -0
- admin/services/storage.py +309 -0
- admin/services/storage_api.py +170 -0
- admin/services/storage_mcp.py +110 -0
- admin/services/validation_api.py +157 -0
- admin/services/websocket.py +76 -0
- common/__init__.py +0 -0
- common/schemas/context_fabric/v1/annotation.schema.json +64 -0
- common/schemas/context_fabric/v1/api-payloads.schema.json +155 -0
- common/schemas/context_fabric/v1/common.defs.schema.json +163 -0
- common/schemas/context_fabric/v1/content-node.schema.json +69 -0
- common/schemas/context_fabric/v1/corpus.schema.json +31 -0
- common/schemas/context_fabric/v1/edition.schema.json +74 -0
- common/schemas/context_fabric/v1/examples/academic-paper.json +154 -0
- common/schemas/context_fabric/v1/examples/book-chapter.json +91 -0
- common/schemas/context_fabric/v1/examples/index.json +11 -0
- common/schemas/context_fabric/v1/examples/letter-page.json +97 -0
- common/schemas/context_fabric/v1/examples/scripture-node.json +89 -0
- common/schemas/context_fabric/v1/examples/speech.json +73 -0
- common/schemas/context_fabric/v1/examples/transcript.json +129 -0
- common/schemas/context_fabric/v1/physical-locator.schema.json +51 -0
- common/schemas/context_fabric/v1/reference.schema.json +56 -0
- common/schemas/context_fabric/v1/relationship.schema.json +28 -0
- common/schemas/context_fabric/v1/source-asset.schema.json +32 -0
- common/schemas/context_fabric/v1/text-fragment.schema.json +44 -0
- common/schemas/context_fabric/v1/work.schema.json +37 -0
- common/utils/__init__.py +9 -0
- common/utils/config.py +95 -0
- common/utils/console.py +34 -0
- common/utils/constant.py +36 -0
- common/utils/helpers.py +79 -0
- common/utils/jwt_auth.py +64 -0
- corpora_mcp/__init__.py +9 -0
- corpora_mcp/__main__.py +4 -0
- corpora_mcp/corpus.py +90 -0
- corpora_mcp/server.py +780 -0
- corpora_mcp/validate.py +559 -0
- corpora_py/__init__.py +12 -0
- corpora_py/app.py +191 -0
- corpora_py/auth.py +108 -0
- corpora_py/bridge.py +39 -0
- corpora_py/validation_api.py +124 -0
- corpora_py-0.1.3.dist-info/METADATA +332 -0
- corpora_py-0.1.3.dist-info/RECORD +79 -0
- corpora_py-0.1.3.dist-info/WHEEL +4 -0
- corpora_py-0.1.3.dist-info/entry_points.txt +3 -0
- corpora_py-0.1.3.dist-info/licenses/LICENSE +21 -0
admin/__init__.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""admin — admin-only / full-featured tooling.
|
|
2
|
+
|
|
3
|
+
Houses conversion pipelines (EPUB/HTML → Text-Fabric, TF → .exg packaging)
|
|
4
|
+
and other heavy or privileged operations. These typically require the
|
|
5
|
+
[full] extra (text-fabric) and are not needed in the slim client runtime.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from . import converters, parsers, services
|
|
9
|
+
|
|
10
|
+
__all__ = ["converters", "parsers", "services"]
|
|
11
|
+
|
|
12
|
+
try:
|
|
13
|
+
# admin.ingest imports docling-core (pandas, pillow) at module import
|
|
14
|
+
# time. Slim serverless bundles (Vercel — see vercel.json) exclude that
|
|
15
|
+
# chain to stay under the function size limit, so its absence must not
|
|
16
|
+
# break `import admin`; /ingest degrades to 503 there (see
|
|
17
|
+
# services/ingest_api.py).
|
|
18
|
+
from . import ingest # noqa: F401
|
|
19
|
+
|
|
20
|
+
__all__.append("ingest")
|
|
21
|
+
except ModuleNotFoundError: # pragma: no cover - exercised only in slim deploys
|
|
22
|
+
pass
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
# CLAUDE.md — `admin.converters`
|
|
2
|
+
|
|
3
|
+
`Document`/`Unit` tree → Text-Fabric dataset → `.cfm` cache → `.corpus` archive. See
|
|
4
|
+
`packages/admin/CLAUDE.md` for the shared-schema/shared-walker architecture that makes this package
|
|
5
|
+
possible; this file covers the Text-Fabric, Context-Fabric, and `.corpus` details specific to
|
|
6
|
+
`converters/`.
|
|
7
|
+
|
|
8
|
+
Read `_walker.py` before touching any feature/node-creation logic — it's the one TF walk reused by
|
|
9
|
+
every `_{format}_to_tf.py`.
|
|
10
|
+
|
|
11
|
+
## Text-Fabric walker gotchas (all handled in `_walker.py`)
|
|
12
|
+
|
|
13
|
+
- **Every feature name must have metadata**, or `cv.walk()` fails validation with `"node feature has no metadata"`.
|
|
14
|
+
Feature names here vary per document (HTML attributes, TEI `@type`, ...) so they can't be declared upfront in
|
|
15
|
+
`featureMeta=`; `set_features()` registers each one dynamically via `cv.meta(name, valueType="str")` right before
|
|
16
|
+
setting it. If you add a
|
|
17
|
+
`cv.feature()` call anywhere, route it through `set_features()`, not
|
|
18
|
+
`cv.feature()` directly, or you'll reintroduce this failure.
|
|
19
|
+
- **Dynamically-registered features need an explicit `valueType`** or the exporter warns `"Missing @valueType"`
|
|
20
|
+
(non-fatal, but avoidable — that's why `set_features()` always passes `valueType="str"`).
|
|
21
|
+
- **A node covering zero slots gets silently deleted** by Text-Fabric's
|
|
22
|
+
"remove unlinked nodes" pass. A leaf `Unit` with no tokens and no children (a blank PDF page, an `<img>`, an `<hr>`)
|
|
23
|
+
would otherwise vanish along with its attributes — `_walk_unit()` gives genuinely empty leaves one placeholder
|
|
24
|
+
empty-text slot so they survive.
|
|
25
|
+
- **`otext.sectionTypes`/`sectionFeatures` can't be empty**, but also don't need to be elaborate: every converter uses a
|
|
26
|
+
single section level (the root `book`/`document`/`text` node, with `title` as its section feature). Finer structure
|
|
27
|
+
(chapters, pages, divs) is still expressed as ordinary node types via `otype_for` — it just isn't declared as TF
|
|
28
|
+
"sections", which would require strict, consistent nesting we can't guarantee across arbitrary source documents.
|
|
29
|
+
- **`SKIP_TAGS` in `parsers/_html.py` is scoped to tags that only make sense to drop when nested inside `<body>`**
|
|
30
|
+
(script/style/noscript/svg/math). It used to include `"head"` for HTML's metadata tag, which silently ate TEI's
|
|
31
|
+
`<head>` (a heading element, reused by the shared walker) — don't add HTML-specific tag names back to that set without
|
|
32
|
+
checking what they mean in TEI/XML first.
|
|
33
|
+
|
|
34
|
+
## Context-Fabric (`cfabric`) notes
|
|
35
|
+
|
|
36
|
+
- There is **no separate compile API**. `.cfm` compilation happens automatically the first time a dataset is loaded via
|
|
37
|
+
`cfabric.Fabric(locations=...).loadAll()`. `convert_to_cfm()` exists only to trigger that load on purpose and hand
|
|
38
|
+
back the resulting `.cfm` path.
|
|
39
|
+
- `Fabric(...).loadAll()` returns `Api | bool` (`False` on failure) — always narrow with `isinstance(result, bool)`
|
|
40
|
+
before touching `.F`/`.T`/`.Fall()`; those are dynamically populated at load time so mypy can't see their attributes
|
|
41
|
+
either (hence the `type: ignore[attr-defined]` in
|
|
42
|
+
`convert_to_corpus.py`).
|
|
43
|
+
|
|
44
|
+
## `.corpus` archive
|
|
45
|
+
|
|
46
|
+
The archive format (`manifest.yml`, `toc.yml`, `assets/`, `.git/`,
|
|
47
|
+
`corpora/{*.tf, .cfm/}`) is the contract both the Corpora and Exegia apps parse. The canonical spec is maintained in an
|
|
48
|
+
external vault by the Corpora team. Before changing manifest/toc shape in `convert_to_corpus.py`, consult the current
|
|
49
|
+
schema definition with the team or check the app's schema loader to understand the expected format.
|
|
50
|
+
|
|
51
|
+
## Known gaps (converter-side)
|
|
52
|
+
|
|
53
|
+
Service-side gaps (progress reporting, job registry, archive cleanup) live in
|
|
54
|
+
`src/admin/services/CLAUDE.md`.
|
|
55
|
+
|
|
56
|
+
- **No `_xml_to_tf.py`.** `XmlParser` exists (`admin.parsers`) but there's no matching Text-Fabric converter — generic
|
|
57
|
+
XML has no fixed node-type vocabulary to map onto, unlike TEI's `<div>`/`<p>` convention. Add one the same way as the
|
|
58
|
+
others (pick an `otype_for`, wire it into
|
|
59
|
+
`converters/__init__.py`'s `CONVERTERS`) if a concrete need shows up; don't add it speculatively.
|
|
60
|
+
- `dataset_id`/`project_id`/`publisher_id`/`author_ids` in
|
|
61
|
+
`convert_to_corpus()` are caller-supplied and default to `""` — this package has no way to know them; they're assigned
|
|
62
|
+
by whatever backend calls it (the Corpora/Exegia app, not this converter).
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""admin.converters — conversion and packaging tools (admin / full runtime).
|
|
2
|
+
|
|
3
|
+
These modules depend on optional heavy deps (text-fabric, context-fabric,
|
|
4
|
+
under the `[full]` extra) and are intended for dataset ingestion / admin
|
|
5
|
+
tooling only.
|
|
6
|
+
|
|
7
|
+
Pipeline: a `_{format}_to_tf.py` converter turns a source document into a
|
|
8
|
+
Text-Fabric dataset; `convert_to_cfm` compiles that into Context-Fabric's
|
|
9
|
+
`.cfm` cache; `convert_to_corpus` packages both into a `.corpus` archive.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from ..parsers import SourceFormat
|
|
13
|
+
from ._epub_to_tf import convert_epub_to_tf
|
|
14
|
+
from ._html_to_tf import convert_html_to_tf
|
|
15
|
+
from ._pdf_to_tf import convert_pdf_to_tf
|
|
16
|
+
from ._tei_to_tf import convert_tei_to_tf
|
|
17
|
+
from ._tei_zip_to_tf import convert_tei_zip_to_tf
|
|
18
|
+
from ._text_to_tf import convert_text_to_tf
|
|
19
|
+
from ._tf_zip_to_tf import convert_tf_zip_to_tf
|
|
20
|
+
from .convert_to_cfm import convert_to_cfm
|
|
21
|
+
from .convert_to_corpus import convert_to_corpus
|
|
22
|
+
|
|
23
|
+
CONVERTERS = {
|
|
24
|
+
SourceFormat.EPUB: convert_epub_to_tf,
|
|
25
|
+
SourceFormat.HTML: convert_html_to_tf,
|
|
26
|
+
SourceFormat.PDF: convert_pdf_to_tf,
|
|
27
|
+
SourceFormat.TEI: convert_tei_to_tf,
|
|
28
|
+
SourceFormat.TEI_ZIP: convert_tei_zip_to_tf,
|
|
29
|
+
SourceFormat.PLAIN: convert_text_to_tf,
|
|
30
|
+
SourceFormat.TF_ZIP: convert_tf_zip_to_tf,
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
__all__ = [
|
|
34
|
+
"CONVERTERS",
|
|
35
|
+
"convert_epub_to_tf",
|
|
36
|
+
"convert_html_to_tf",
|
|
37
|
+
"convert_pdf_to_tf",
|
|
38
|
+
"convert_tei_to_tf",
|
|
39
|
+
"convert_tei_zip_to_tf",
|
|
40
|
+
"convert_text_to_tf",
|
|
41
|
+
"convert_tf_zip_to_tf",
|
|
42
|
+
"convert_to_cfm",
|
|
43
|
+
"convert_to_corpus",
|
|
44
|
+
]
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""
|
|
2
|
+
EPUB to Text-Fabric Converter
|
|
3
|
+
|
|
4
|
+
Converts EPUB ebook files into Text-Fabric datasets using the epub service
|
|
5
|
+
for parsing and the tf.convert.walker library for TF generation.
|
|
6
|
+
|
|
7
|
+
Features:
|
|
8
|
+
- Extracts EPUB metadata (title, author, publisher, etc.)
|
|
9
|
+
- Preserves book structure (chapters/pages)
|
|
10
|
+
- Converts HTML content to queryable nodes
|
|
11
|
+
- Creates semantic nodes from clean HTML
|
|
12
|
+
- Tracks conversion progress
|
|
13
|
+
|
|
14
|
+
Node Types:
|
|
15
|
+
- book: Root node for the entire EPUB
|
|
16
|
+
- chapter: Individual pages/chapters from the EPUB
|
|
17
|
+
- element: HTML elements from page content
|
|
18
|
+
- paragraph: Paragraph-like elements
|
|
19
|
+
- link: Link elements with href
|
|
20
|
+
- word: Individual words (slots)
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
from ..parsers import EpubParser
|
|
26
|
+
from ..parsers.schema import Unit
|
|
27
|
+
from ._walker import convert_document
|
|
28
|
+
|
|
29
|
+
_PARAGRAPH_TAGS = {"p", "blockquote"}
|
|
30
|
+
_LINK_TAGS = {"a"}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _otype_for(unit: Unit) -> str:
|
|
34
|
+
if unit.type == "chapter":
|
|
35
|
+
return "chapter"
|
|
36
|
+
if unit.type in _PARAGRAPH_TAGS:
|
|
37
|
+
return "paragraph"
|
|
38
|
+
if unit.type in _LINK_TAGS:
|
|
39
|
+
return "link"
|
|
40
|
+
return "element"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def convert_epub_to_tf(source: str, output_dir: str | Path) -> Path:
|
|
44
|
+
"""Convert an EPUB at `source` (path or URL) into a Text-Fabric dataset."""
|
|
45
|
+
return convert_document(
|
|
46
|
+
EpubParser(),
|
|
47
|
+
source,
|
|
48
|
+
output_dir,
|
|
49
|
+
root_type="book",
|
|
50
|
+
otype_for=_otype_for,
|
|
51
|
+
)
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""
|
|
2
|
+
HTML to Text-Fabric Converter
|
|
3
|
+
|
|
4
|
+
Converts HTML documents into Text-Fabric (TF) datasets using the tf.convert.walker library.
|
|
5
|
+
This allows HTML content to be queried using Context-Fabric's powerful graph query API.
|
|
6
|
+
|
|
7
|
+
Features:
|
|
8
|
+
- Preserves HTML structure as a node hierarchy
|
|
9
|
+
- Creates slots for text content (words/tokens)
|
|
10
|
+
- Stores HTML attributes as node features
|
|
11
|
+
- Supports nested HTML elements
|
|
12
|
+
- Generates valid Text-Fabric datasets
|
|
13
|
+
|
|
14
|
+
Node Types:
|
|
15
|
+
- document: Root node for each HTML document
|
|
16
|
+
- element: HTML tags (div, p, span, etc.)
|
|
17
|
+
- word: Individual words (slots)
|
|
18
|
+
|
|
19
|
+
Features:
|
|
20
|
+
- tag: HTML tag name (div, p, span, etc.)
|
|
21
|
+
- class: CSS class names
|
|
22
|
+
- id: HTML id attribute
|
|
23
|
+
- href: Link URLs (for <a> tags)
|
|
24
|
+
- src: Source URLs (for <img>, <script> tags)
|
|
25
|
+
- text: Raw text content
|
|
26
|
+
- * (any HTML attribute preserved as feature)
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
|
|
31
|
+
from ..parsers import HtmlParser
|
|
32
|
+
from ._walker import convert_document
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def convert_html_to_tf(source: str, output_dir: str | Path) -> Path:
|
|
36
|
+
"""Convert an HTML document at `source` (path or URL) into a Text-Fabric dataset."""
|
|
37
|
+
return convert_document(
|
|
38
|
+
HtmlParser(),
|
|
39
|
+
source,
|
|
40
|
+
output_dir,
|
|
41
|
+
root_type="document",
|
|
42
|
+
otype_for=lambda unit: "element",
|
|
43
|
+
)
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""
|
|
2
|
+
PDF to Text-Fabric Converter
|
|
3
|
+
|
|
4
|
+
Converts PDF files into Text-Fabric datasets using the pdf parser for
|
|
5
|
+
extraction and the tf.convert.walker library for TF generation.
|
|
6
|
+
|
|
7
|
+
Features:
|
|
8
|
+
- Extracts PDF metadata (title, author, producer, etc.)
|
|
9
|
+
- One node per page, so a large PDF converts one page at a time
|
|
10
|
+
- Tracks conversion progress
|
|
11
|
+
|
|
12
|
+
Node Types:
|
|
13
|
+
- book: Root node for the entire PDF
|
|
14
|
+
- page: Individual pages
|
|
15
|
+
- word: Individual words (slots)
|
|
16
|
+
|
|
17
|
+
Features:
|
|
18
|
+
- title, creators, date, subjects, ...: document metadata
|
|
19
|
+
- page_number: 1-based page number
|
|
20
|
+
- text, after: word text and its trailing whitespace
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
|
|
25
|
+
from ..parsers import PdfParser
|
|
26
|
+
from ._walker import convert_document
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def convert_pdf_to_tf(source: str, output_dir: str | Path) -> Path:
|
|
30
|
+
"""Convert a PDF at `source` (path or URL) into a Text-Fabric dataset."""
|
|
31
|
+
return convert_document(
|
|
32
|
+
PdfParser(),
|
|
33
|
+
source,
|
|
34
|
+
output_dir,
|
|
35
|
+
root_type="book",
|
|
36
|
+
otype_for=lambda unit: "page",
|
|
37
|
+
)
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""
|
|
2
|
+
TEI to Text-Fabric Converter
|
|
3
|
+
|
|
4
|
+
Converts TEI (Text Encoding Initiative) XML documents into Text-Fabric
|
|
5
|
+
datasets using the tei parser for extraction and the tf.convert.walker
|
|
6
|
+
library for TF generation.
|
|
7
|
+
|
|
8
|
+
Features:
|
|
9
|
+
- Extracts teiHeader metadata (title, author, publisher, date, ...)
|
|
10
|
+
- Preserves the TEI division hierarchy (<div> nesting)
|
|
11
|
+
- Lifts each division's <head> into a `label` feature
|
|
12
|
+
- One node per top-level division, so a large edition (a multi-book bible,
|
|
13
|
+
a multi-volume critical edition) converts one division at a time
|
|
14
|
+
|
|
15
|
+
Node Types:
|
|
16
|
+
- text: Root node for the entire TEI document
|
|
17
|
+
- div: A TEI division (chapter, book, ...); its `type` attribute survives
|
|
18
|
+
as an ordinary feature (e.g. `type="chapter"`)
|
|
19
|
+
- paragraph: <p> elements
|
|
20
|
+
- element: Any other TEI element (l, seg, note, ...)
|
|
21
|
+
- word: Individual words (slots)
|
|
22
|
+
|
|
23
|
+
Features:
|
|
24
|
+
- title, creators, language, publisher, date, identifier, rights: document metadata
|
|
25
|
+
- label: division heading, lifted from its <head> child
|
|
26
|
+
- type: TEI `@type` attribute (e.g. div type="chapter")
|
|
27
|
+
- text, after: word text and its trailing whitespace
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
|
|
32
|
+
from ..parsers import TeiParser
|
|
33
|
+
from ..parsers.schema import Unit
|
|
34
|
+
from ._walker import convert_document
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _otype_for(unit: Unit) -> str:
|
|
38
|
+
if unit.type == "div":
|
|
39
|
+
return "div"
|
|
40
|
+
if unit.type == "p":
|
|
41
|
+
return "paragraph"
|
|
42
|
+
return "element"
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def convert_tei_to_tf(source: str, output_dir: str | Path) -> Path:
|
|
46
|
+
"""Convert a TEI document at `source` (path or URL) into a Text-Fabric dataset."""
|
|
47
|
+
return convert_document(
|
|
48
|
+
TeiParser(),
|
|
49
|
+
source,
|
|
50
|
+
output_dir,
|
|
51
|
+
root_type="text",
|
|
52
|
+
otype_for=_otype_for,
|
|
53
|
+
)
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""
|
|
2
|
+
TEI ZIP to Text-Fabric Converter
|
|
3
|
+
|
|
4
|
+
Converts a ZIP archive of TEI (or TEI-in-plain-`.xml`) documents into a
|
|
5
|
+
single Text-Fabric dataset: each member document is parsed with the same
|
|
6
|
+
`TeiParser` the single-file `tei` converter uses, and all of them are walked
|
|
7
|
+
into one dataset via `convert_documents` -- one "text" root node per member
|
|
8
|
+
document, carrying that document's `<teiHeader>` metadata as features.
|
|
9
|
+
Members convert in archive-path order (sorted), so a corpus zipped as
|
|
10
|
+
`01-genesis.xml`, `02-exodus.xml`, ... keeps its intended sequence.
|
|
11
|
+
|
|
12
|
+
Node Types / Features: identical to `_tei_to_tf.py` (same parser, same
|
|
13
|
+
`otype_for`) -- the only difference is that the dataset can contain several
|
|
14
|
+
"text" roots instead of exactly one.
|
|
15
|
+
|
|
16
|
+
Archive safety mirrors `_tf_zip_to_tf.py`: path traversal, symlinks,
|
|
17
|
+
encrypted members, member count, and expanded size are all validated before
|
|
18
|
+
any bytes are written. macOS `__MACOSX/` resource forks and hidden dotfiles
|
|
19
|
+
are ignored.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import shutil
|
|
25
|
+
import tempfile
|
|
26
|
+
from pathlib import Path, PurePosixPath
|
|
27
|
+
from zipfile import BadZipFile, ZipFile, ZipInfo
|
|
28
|
+
|
|
29
|
+
from ..parsers import TeiParser
|
|
30
|
+
from ..parsers.schema import Document
|
|
31
|
+
from ._tei_to_tf import _otype_for
|
|
32
|
+
from ._tf_zip_to_tf import _MAX_FILES, _MAX_UNCOMPRESSED_BYTES, _safe_path
|
|
33
|
+
from ._walker import convert_documents
|
|
34
|
+
|
|
35
|
+
_TEI_SUFFIXES = frozenset({".tei", ".xml"})
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _is_noise(path: PurePosixPath) -> bool:
|
|
39
|
+
"""Housekeeping entries that say nothing about the corpus itself."""
|
|
40
|
+
return path.parts[0] == "__MACOSX" or path.name.startswith(".")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _tei_members(archive: ZipFile) -> list[ZipInfo]:
|
|
44
|
+
infos = [info for info in archive.infolist() if not info.is_dir()]
|
|
45
|
+
if len(infos) > _MAX_FILES:
|
|
46
|
+
raise ValueError(f"TEI ZIP contains more than {_MAX_FILES:,} files")
|
|
47
|
+
if sum(info.file_size for info in infos) > _MAX_UNCOMPRESSED_BYTES:
|
|
48
|
+
raise ValueError("Expanded TEI ZIP exceeds the 2 GiB limit")
|
|
49
|
+
|
|
50
|
+
members = [
|
|
51
|
+
info
|
|
52
|
+
for info in infos
|
|
53
|
+
for path in (_safe_path(info),)
|
|
54
|
+
if not _is_noise(path) and path.suffix.lower() in _TEI_SUFFIXES
|
|
55
|
+
]
|
|
56
|
+
if not members:
|
|
57
|
+
raise ValueError(
|
|
58
|
+
"ZIP does not contain any TEI documents (.tei or .xml files are required)"
|
|
59
|
+
)
|
|
60
|
+
return sorted(members, key=lambda info: info.filename)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def convert_tei_zip_to_tf(source: str, output_dir: str | Path) -> Path:
|
|
64
|
+
"""Convert every TEI/XML document inside ``source`` into one Text-Fabric
|
|
65
|
+
dataset at ``output_dir``."""
|
|
66
|
+
parser = TeiParser()
|
|
67
|
+
documents: list[Document] = []
|
|
68
|
+
|
|
69
|
+
try:
|
|
70
|
+
with (
|
|
71
|
+
ZipFile(source) as archive,
|
|
72
|
+
tempfile.TemporaryDirectory(prefix="tei-zip-") as scratch,
|
|
73
|
+
):
|
|
74
|
+
for info in _tei_members(archive):
|
|
75
|
+
# Flattened to the basename: members were validated against
|
|
76
|
+
# traversal above, and the parser only needs a readable file.
|
|
77
|
+
extracted = Path(scratch) / f"{len(documents)}-{PurePosixPath(info.filename).name}"
|
|
78
|
+
with (
|
|
79
|
+
archive.open(info) as member,
|
|
80
|
+
extracted.open("wb") as out,
|
|
81
|
+
):
|
|
82
|
+
shutil.copyfileobj(member, out)
|
|
83
|
+
documents.append(parser.parse(str(extracted)))
|
|
84
|
+
except BadZipFile as exc:
|
|
85
|
+
raise ValueError("Uploaded file is not a valid ZIP archive") from exc
|
|
86
|
+
|
|
87
|
+
return convert_documents(
|
|
88
|
+
documents,
|
|
89
|
+
output_dir,
|
|
90
|
+
root_type="text",
|
|
91
|
+
otype_for=_otype_for,
|
|
92
|
+
format_value=parser.format.value,
|
|
93
|
+
source_label=source,
|
|
94
|
+
)
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Plain Text to Text-Fabric Converter
|
|
3
|
+
|
|
4
|
+
Converts raw text files into Text-Fabric datasets using the plain-text
|
|
5
|
+
parser for extraction and the tf.convert.walker library for TF generation.
|
|
6
|
+
|
|
7
|
+
Features:
|
|
8
|
+
- One node per paragraph (blank-line-separated), so a large plain-text
|
|
9
|
+
corpus converts one paragraph at a time
|
|
10
|
+
- Minimal metadata (a title derived from the file name)
|
|
11
|
+
|
|
12
|
+
Node Types:
|
|
13
|
+
- book: Root node for the entire text file
|
|
14
|
+
- paragraph: A blank-line-separated block of text
|
|
15
|
+
- word: Individual words (slots)
|
|
16
|
+
|
|
17
|
+
Features:
|
|
18
|
+
- title: derived from the source file name
|
|
19
|
+
- text, after: word text and its trailing whitespace
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from ..parsers import PlainTextParser
|
|
25
|
+
from ._walker import convert_document
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def convert_text_to_tf(source: str, output_dir: str | Path) -> Path:
|
|
29
|
+
"""Convert a plain-text file at `source` (path or URL) into a Text-Fabric dataset."""
|
|
30
|
+
return convert_document(
|
|
31
|
+
PlainTextParser(),
|
|
32
|
+
source,
|
|
33
|
+
output_dir,
|
|
34
|
+
root_type="book",
|
|
35
|
+
otype_for=lambda unit: "paragraph",
|
|
36
|
+
)
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Import an existing Text-Fabric dataset from a ZIP archive."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import shutil
|
|
6
|
+
import stat
|
|
7
|
+
from pathlib import Path, PurePosixPath
|
|
8
|
+
from zipfile import BadZipFile, ZipFile, ZipInfo
|
|
9
|
+
|
|
10
|
+
_MAX_FILES = 10_000
|
|
11
|
+
_MAX_UNCOMPRESSED_BYTES = 2 * 1024 * 1024 * 1024
|
|
12
|
+
_REQUIRED_FILES = frozenset({"otype.tf", "oslots.tf"})
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _safe_path(info: ZipInfo) -> PurePosixPath:
|
|
16
|
+
path = PurePosixPath(info.filename)
|
|
17
|
+
mode = info.external_attr >> 16
|
|
18
|
+
if path.is_absolute() or ".." in path.parts:
|
|
19
|
+
raise ValueError(f"Unsafe path in Text-Fabric ZIP: {info.filename!r}")
|
|
20
|
+
if stat.S_ISLNK(mode):
|
|
21
|
+
raise ValueError(
|
|
22
|
+
f"Symbolic links are not allowed in Text-Fabric ZIPs: {info.filename!r}"
|
|
23
|
+
)
|
|
24
|
+
if info.flag_bits & 0x1:
|
|
25
|
+
raise ValueError("Encrypted Text-Fabric ZIPs are not supported")
|
|
26
|
+
return path
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _find_dataset_root(files: dict[PurePosixPath, ZipInfo]) -> PurePosixPath:
|
|
30
|
+
candidates: list[PurePosixPath] = []
|
|
31
|
+
for path in files:
|
|
32
|
+
if path.name != "otype.tf":
|
|
33
|
+
continue
|
|
34
|
+
if all(path.parent / required in files for required in _REQUIRED_FILES):
|
|
35
|
+
candidates.append(path.parent)
|
|
36
|
+
|
|
37
|
+
if not candidates:
|
|
38
|
+
raise ValueError(
|
|
39
|
+
"ZIP does not contain a Text-Fabric dataset (otype.tf and oslots.tf are required)"
|
|
40
|
+
)
|
|
41
|
+
if len(candidates) > 1:
|
|
42
|
+
roots = ", ".join(str(path) for path in sorted(candidates, key=str))
|
|
43
|
+
raise ValueError(f"ZIP contains multiple Text-Fabric datasets: {roots}")
|
|
44
|
+
return candidates[0]
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def convert_tf_zip_to_tf(source: str, output_dir: str | Path) -> Path:
|
|
48
|
+
"""Extract the single Text-Fabric dataset in ``source`` to ``output_dir``.
|
|
49
|
+
|
|
50
|
+
Archive paths, symlinks, member count, and expanded size are validated
|
|
51
|
+
before any bytes are written so a malformed upload cannot escape or fill
|
|
52
|
+
the conversion work directory.
|
|
53
|
+
"""
|
|
54
|
+
output_dir = Path(output_dir)
|
|
55
|
+
|
|
56
|
+
try:
|
|
57
|
+
with ZipFile(source) as archive:
|
|
58
|
+
infos = [info for info in archive.infolist() if not info.is_dir()]
|
|
59
|
+
if len(infos) > _MAX_FILES:
|
|
60
|
+
raise ValueError(
|
|
61
|
+
f"Text-Fabric ZIP contains more than {_MAX_FILES:,} files"
|
|
62
|
+
)
|
|
63
|
+
if sum(info.file_size for info in infos) > _MAX_UNCOMPRESSED_BYTES:
|
|
64
|
+
raise ValueError("Expanded Text-Fabric ZIP exceeds the 2 GiB limit")
|
|
65
|
+
|
|
66
|
+
files = {_safe_path(info): info for info in infos}
|
|
67
|
+
dataset_root = _find_dataset_root(files)
|
|
68
|
+
dataset_files = {
|
|
69
|
+
path: info
|
|
70
|
+
for path, info in files.items()
|
|
71
|
+
if path.is_relative_to(dataset_root) and path.suffix == ".tf"
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
output_dir.mkdir(parents=True, exist_ok=False)
|
|
75
|
+
for path, info in dataset_files.items():
|
|
76
|
+
relative = path.relative_to(dataset_root)
|
|
77
|
+
destination = output_dir.joinpath(*relative.parts)
|
|
78
|
+
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
79
|
+
with (
|
|
80
|
+
archive.open(info) as source_file,
|
|
81
|
+
destination.open("wb") as output_file,
|
|
82
|
+
):
|
|
83
|
+
shutil.copyfileobj(source_file, output_file)
|
|
84
|
+
except BadZipFile as exc:
|
|
85
|
+
raise ValueError("Uploaded file is not a valid ZIP archive") from exc
|
|
86
|
+
|
|
87
|
+
return output_dir
|