edocapi 0.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edocapi/__init__.py +49 -0
- edocapi/app.py +255 -0
- edocapi/cli/__init__.py +3 -0
- edocapi/cli/main.py +94 -0
- edocapi/config.py +76 -0
- edocapi/document.py +290 -0
- edocapi/exceptions.py +104 -0
- edocapi/files.py +97 -0
- edocapi/processors/__init__.py +16 -0
- edocapi/processors/base.py +101 -0
- edocapi/processors/docx.py +112 -0
- edocapi/processors/html.py +64 -0
- edocapi/processors/image.py +80 -0
- edocapi/processors/markdown.py +64 -0
- edocapi/processors/pdf.py +231 -0
- edocapi/processors/txt.py +58 -0
- edocapi/responses.py +88 -0
- edocapi/storage/__init__.py +3 -0
- edocapi/storage/temporary.py +132 -0
- edocapi/validation/__init__.py +13 -0
- edocapi/validation/files.py +209 -0
- edocapi-0.0.2.dist-info/METADATA +335 -0
- edocapi-0.0.2.dist-info/RECORD +27 -0
- edocapi-0.0.2.dist-info/WHEEL +5 -0
- edocapi-0.0.2.dist-info/entry_points.txt +2 -0
- edocapi-0.0.2.dist-info/licenses/LICENSE +21 -0
- edocapi-0.0.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
"""DOCX processor for eDocAPI (uses python-docx + weasyprint for PDF)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from docx import Document as DocxDocument
|
|
10
|
+
|
|
11
|
+
from edocapi.exceptions import ConversionError, InvalidDocument
|
|
12
|
+
from edocapi.processors.base import BaseProcessor, register_processor
|
|
13
|
+
from edocapi.storage.temporary import TemporaryStorage
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger("edocapi.processors.docx")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@register_processor("docx")
|
|
19
|
+
class DOCXProcessor(BaseProcessor):
|
|
20
|
+
"""Processor for Microsoft Word (.docx) documents."""
|
|
21
|
+
|
|
22
|
+
supported_input_types = {"docx"}
|
|
23
|
+
supported_output_types = {"pdf", "txt", "html"}
|
|
24
|
+
|
|
25
|
+
def __init__(self, path: Path, *, temp_storage: Any = None) -> None:
|
|
26
|
+
super().__init__(path, temp_storage=temp_storage)
|
|
27
|
+
try:
|
|
28
|
+
self._doc = DocxDocument(str(path))
|
|
29
|
+
except Exception as exc:
|
|
30
|
+
raise InvalidDocument(f"Invalid or corrupted DOCX: {exc}") from exc
|
|
31
|
+
|
|
32
|
+
def to_text(self) -> str:
|
|
33
|
+
"""Extract all paragraph text."""
|
|
34
|
+
paragraphs = [p.text for p in self._doc.paragraphs if p.text.strip()]
|
|
35
|
+
# Also extract tables
|
|
36
|
+
for table in self._doc.tables:
|
|
37
|
+
for row in table.rows:
|
|
38
|
+
cells = [cell.text.strip() for cell in row.cells]
|
|
39
|
+
paragraphs.append("\t".join(cells))
|
|
40
|
+
return "\n".join(paragraphs)
|
|
41
|
+
|
|
42
|
+
def to_html(self) -> str:
|
|
43
|
+
"""Convert DOCX to a simple HTML representation."""
|
|
44
|
+
parts: list[str] = ["<html><body>"]
|
|
45
|
+
for p in self._doc.paragraphs:
|
|
46
|
+
style = p.style.name.lower() if p.style else ""
|
|
47
|
+
text = (
|
|
48
|
+
p.text.replace("&", "&")
|
|
49
|
+
.replace("<", "<")
|
|
50
|
+
.replace(">", ">")
|
|
51
|
+
)
|
|
52
|
+
if not text.strip():
|
|
53
|
+
continue
|
|
54
|
+
if "heading 1" in style:
|
|
55
|
+
parts.append(f"<h1>{text}</h1>")
|
|
56
|
+
elif "heading 2" in style:
|
|
57
|
+
parts.append(f"<h2>{text}</h2>")
|
|
58
|
+
elif "heading 3" in style:
|
|
59
|
+
parts.append(f"<h3>{text}</h3>")
|
|
60
|
+
else:
|
|
61
|
+
parts.append(f"<p>{text}</p>")
|
|
62
|
+
|
|
63
|
+
for table in self._doc.tables:
|
|
64
|
+
parts.append("<table border='1'>")
|
|
65
|
+
for row in table.rows:
|
|
66
|
+
parts.append("<tr>")
|
|
67
|
+
for cell in row.cells:
|
|
68
|
+
cell_text = (
|
|
69
|
+
cell.text.replace("&", "&")
|
|
70
|
+
.replace("<", "<")
|
|
71
|
+
.replace(">", ">")
|
|
72
|
+
)
|
|
73
|
+
parts.append(f"<td>{cell_text}</td>")
|
|
74
|
+
parts.append("</tr>")
|
|
75
|
+
parts.append("</table>")
|
|
76
|
+
|
|
77
|
+
parts.append("</body></html>")
|
|
78
|
+
return "\n".join(parts)
|
|
79
|
+
|
|
80
|
+
def to_pdf(self) -> Path:
|
|
81
|
+
"""Convert DOCX -> HTML -> PDF via WeasyPrint."""
|
|
82
|
+
html = self.to_html()
|
|
83
|
+
try:
|
|
84
|
+
from weasyprint import HTML
|
|
85
|
+
except ImportError as exc:
|
|
86
|
+
raise ConversionError(
|
|
87
|
+
"DOCX -> PDF requires weasyprint. "
|
|
88
|
+
"Install with: pip install edocapi[html]",
|
|
89
|
+
source="docx",
|
|
90
|
+
target="pdf",
|
|
91
|
+
) from exc
|
|
92
|
+
|
|
93
|
+
storage = self.temp_storage or TemporaryStorage()
|
|
94
|
+
out = storage.create_file(suffix=".pdf", prefix="docx_")
|
|
95
|
+
HTML(string=html).write_pdf(str(out))
|
|
96
|
+
return out
|
|
97
|
+
|
|
98
|
+
def info(self) -> dict[str, Any]:
|
|
99
|
+
core = self._doc.core_properties
|
|
100
|
+
return {
|
|
101
|
+
"filename": self.path.name,
|
|
102
|
+
"type": "docx",
|
|
103
|
+
"size": self.path.stat().st_size,
|
|
104
|
+
"extension": ".docx",
|
|
105
|
+
"title": core.title or None,
|
|
106
|
+
"author": core.author or None,
|
|
107
|
+
"subject": core.subject or None,
|
|
108
|
+
"created": str(core.created) if core.created else None,
|
|
109
|
+
"modified": str(core.modified) if core.modified else None,
|
|
110
|
+
"paragraphs": len(self._doc.paragraphs),
|
|
111
|
+
"tables": len(self._doc.tables),
|
|
112
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""HTML processor for eDocAPI (uses weasyprint for PDF)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
import re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from edocapi.exceptions import ConversionError
|
|
11
|
+
from edocapi.processors.base import BaseProcessor, register_processor
|
|
12
|
+
from edocapi.storage.temporary import TemporaryStorage
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger("edocapi.processors.html")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@register_processor("html")
|
|
18
|
+
class HTMLProcessor(BaseProcessor):
|
|
19
|
+
"""Processor for HTML documents."""
|
|
20
|
+
|
|
21
|
+
supported_input_types = {"html"}
|
|
22
|
+
supported_output_types = {"pdf", "txt", "html"}
|
|
23
|
+
|
|
24
|
+
def to_text(self) -> str:
|
|
25
|
+
"""Strip tags and return approximate plain text."""
|
|
26
|
+
content = self.path.read_text(encoding="utf-8", errors="replace")
|
|
27
|
+
# Very simple tag stripping
|
|
28
|
+
text = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL | re.I)
|
|
29
|
+
text = re.sub(r"<style[^>]*>.*?</style>", "", text, flags=re.DOTALL | re.I)
|
|
30
|
+
text = re.sub(r"<[^>]+>", " ", text)
|
|
31
|
+
text = re.sub(r"\s+", " ", text)
|
|
32
|
+
return text.strip()
|
|
33
|
+
|
|
34
|
+
def to_html(self) -> str:
|
|
35
|
+
return self.path.read_text(encoding="utf-8", errors="replace")
|
|
36
|
+
|
|
37
|
+
def to_pdf(self) -> Path:
|
|
38
|
+
"""Convert HTML to PDF using WeasyPrint."""
|
|
39
|
+
try:
|
|
40
|
+
from weasyprint import HTML
|
|
41
|
+
except ImportError as exc:
|
|
42
|
+
raise ConversionError(
|
|
43
|
+
"HTML -> PDF requires weasyprint. "
|
|
44
|
+
"Install with: pip install edocapi[html]",
|
|
45
|
+
source="html",
|
|
46
|
+
target="pdf",
|
|
47
|
+
) from exc
|
|
48
|
+
|
|
49
|
+
storage = self.temp_storage or TemporaryStorage()
|
|
50
|
+
out = storage.create_file(suffix=".pdf", prefix="html_")
|
|
51
|
+
HTML(filename=str(self.path)).write_pdf(str(out))
|
|
52
|
+
return out
|
|
53
|
+
|
|
54
|
+
def info(self) -> dict[str, Any]:
|
|
55
|
+
content = self.path.read_text(encoding="utf-8", errors="replace")
|
|
56
|
+
title_match = re.search(r"<title[^>]*>(.*?)</title>", content, re.I | re.DOTALL)
|
|
57
|
+
title = title_match.group(1).strip() if title_match else None
|
|
58
|
+
return {
|
|
59
|
+
"filename": self.path.name,
|
|
60
|
+
"type": "html",
|
|
61
|
+
"size": self.path.stat().st_size,
|
|
62
|
+
"extension": self.path.suffix.lower(),
|
|
63
|
+
"title": title,
|
|
64
|
+
}
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Image processor for eDocAPI (uses Pillow + weasyprint / reportlab-like via PDF)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from PIL import Image
|
|
10
|
+
|
|
11
|
+
from edocapi.exceptions import ConversionError, InvalidDocument
|
|
12
|
+
from edocapi.processors.base import BaseProcessor, register_processor
|
|
13
|
+
from edocapi.storage.temporary import TemporaryStorage
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger("edocapi.processors.image")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@register_processor("jpeg", "jpg", "png", "webp")
|
|
19
|
+
class ImageProcessor(BaseProcessor):
|
|
20
|
+
"""Processor for common image formats."""
|
|
21
|
+
|
|
22
|
+
supported_input_types = {"jpeg", "jpg", "png", "webp"}
|
|
23
|
+
supported_output_types = {"pdf", "png", "jpg", "webp"}
|
|
24
|
+
|
|
25
|
+
def __init__(self, path: Path, *, temp_storage: Any = None) -> None:
|
|
26
|
+
super().__init__(path, temp_storage=temp_storage)
|
|
27
|
+
try:
|
|
28
|
+
self._img = Image.open(str(path))
|
|
29
|
+
self._img.load() # force load to validate
|
|
30
|
+
except Exception as exc:
|
|
31
|
+
raise InvalidDocument(f"Invalid or corrupted image: {exc}") from exc
|
|
32
|
+
|
|
33
|
+
def to_pdf(self) -> Path:
|
|
34
|
+
"""Convert image to a single-page PDF."""
|
|
35
|
+
storage = self.temp_storage or TemporaryStorage()
|
|
36
|
+
out = storage.create_file(suffix=".pdf", prefix="img_")
|
|
37
|
+
|
|
38
|
+
# Convert to RGB if necessary (PDF does not support all modes)
|
|
39
|
+
img = self._img
|
|
40
|
+
if img.mode in ("RGBA", "P", "LA"):
|
|
41
|
+
background = Image.new("RGB", img.size, (255, 255, 255))
|
|
42
|
+
if img.mode == "P":
|
|
43
|
+
img = img.convert("RGBA")
|
|
44
|
+
background.paste(img, mask=img.split()[-1] if img.mode in ("RGBA", "LA") else None)
|
|
45
|
+
img = background
|
|
46
|
+
elif img.mode != "RGB":
|
|
47
|
+
img = img.convert("RGB")
|
|
48
|
+
|
|
49
|
+
img.save(str(out), "PDF", resolution=100.0)
|
|
50
|
+
return out
|
|
51
|
+
|
|
52
|
+
def to_text(self) -> str:
|
|
53
|
+
raise NotImplementedError(
|
|
54
|
+
"Text extraction from images requires OCR, which is not available in v0.0.2."
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
def to_images(self, *, format: str = "png") -> list[Path]:
|
|
58
|
+
"""Return a copy of the image in the requested format."""
|
|
59
|
+
fmt = format.lower().lstrip(".")
|
|
60
|
+
if fmt not in {"png", "jpg", "jpeg", "webp"}:
|
|
61
|
+
raise ConversionError(f"Unsupported image format: {format}")
|
|
62
|
+
|
|
63
|
+
storage = self.temp_storage or TemporaryStorage()
|
|
64
|
+
suffix = f".{fmt if fmt != 'jpeg' else 'jpg'}"
|
|
65
|
+
out = storage.create_file(suffix=suffix, prefix="img_")
|
|
66
|
+
save_fmt = "JPEG" if fmt in ("jpg", "jpeg") else fmt.upper()
|
|
67
|
+
self._img.save(str(out), save_fmt)
|
|
68
|
+
return [out]
|
|
69
|
+
|
|
70
|
+
def info(self) -> dict[str, Any]:
|
|
71
|
+
return {
|
|
72
|
+
"filename": self.path.name,
|
|
73
|
+
"type": self.path.suffix.lstrip(".").lower(),
|
|
74
|
+
"size": self.path.stat().st_size,
|
|
75
|
+
"extension": self.path.suffix.lower(),
|
|
76
|
+
"width": self._img.width,
|
|
77
|
+
"height": self._img.height,
|
|
78
|
+
"mode": self._img.mode,
|
|
79
|
+
"format": self._img.format,
|
|
80
|
+
}
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Markdown processor for eDocAPI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
import markdown
|
|
10
|
+
|
|
11
|
+
from edocapi.exceptions import ConversionError
|
|
12
|
+
from edocapi.processors.base import BaseProcessor, register_processor
|
|
13
|
+
from edocapi.storage.temporary import TemporaryStorage
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger("edocapi.processors.markdown")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@register_processor("markdown", "md")
|
|
19
|
+
class MarkdownProcessor(BaseProcessor):
|
|
20
|
+
"""Processor for Markdown documents."""
|
|
21
|
+
|
|
22
|
+
supported_input_types = {"markdown", "md"}
|
|
23
|
+
supported_output_types = {"pdf", "html", "txt"}
|
|
24
|
+
|
|
25
|
+
def to_text(self) -> str:
|
|
26
|
+
"""Return the raw Markdown (or strip to plain text)."""
|
|
27
|
+
return self.path.read_text(encoding="utf-8", errors="replace")
|
|
28
|
+
|
|
29
|
+
def to_html(self) -> str:
|
|
30
|
+
"""Convert Markdown to HTML using the markdown library."""
|
|
31
|
+
md = self.path.read_text(encoding="utf-8", errors="replace")
|
|
32
|
+
html_body = markdown.markdown(
|
|
33
|
+
md,
|
|
34
|
+
extensions=["extra", "codehilite", "tables", "fenced_code"],
|
|
35
|
+
)
|
|
36
|
+
return f"<!DOCTYPE html><html><body>{html_body}</body></html>"
|
|
37
|
+
|
|
38
|
+
def to_pdf(self) -> Path:
|
|
39
|
+
"""Markdown -> HTML -> PDF."""
|
|
40
|
+
html = self.to_html()
|
|
41
|
+
try:
|
|
42
|
+
from weasyprint import HTML
|
|
43
|
+
except ImportError as exc:
|
|
44
|
+
raise ConversionError(
|
|
45
|
+
"Markdown -> PDF requires weasyprint. "
|
|
46
|
+
"Install with: pip install edocapi[html]",
|
|
47
|
+
source="markdown",
|
|
48
|
+
target="pdf",
|
|
49
|
+
) from exc
|
|
50
|
+
|
|
51
|
+
storage = self.temp_storage or TemporaryStorage()
|
|
52
|
+
out = storage.create_file(suffix=".pdf", prefix="md_")
|
|
53
|
+
HTML(string=html).write_pdf(str(out))
|
|
54
|
+
return out
|
|
55
|
+
|
|
56
|
+
def info(self) -> dict[str, Any]:
|
|
57
|
+
content = self.path.read_text(encoding="utf-8", errors="replace")
|
|
58
|
+
return {
|
|
59
|
+
"filename": self.path.name,
|
|
60
|
+
"type": "markdown",
|
|
61
|
+
"size": self.path.stat().st_size,
|
|
62
|
+
"extension": self.path.suffix.lower(),
|
|
63
|
+
"lines": content.count("\n") + 1,
|
|
64
|
+
}
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""PDF processor for eDocAPI (uses pypdf)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any, Sequence
|
|
8
|
+
|
|
9
|
+
from pypdf import PdfReader, PdfWriter
|
|
10
|
+
from pypdf.errors import PdfReadError
|
|
11
|
+
|
|
12
|
+
from edocapi.exceptions import InvalidDocument, ProcessingError
|
|
13
|
+
from edocapi.processors.base import BaseProcessor, register_processor
|
|
14
|
+
from edocapi.storage.temporary import TemporaryStorage
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger("edocapi.processors.pdf")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@register_processor("pdf")
|
|
20
|
+
class PDFProcessor(BaseProcessor):
|
|
21
|
+
"""Processor for PDF documents."""
|
|
22
|
+
|
|
23
|
+
supported_input_types = {"pdf"}
|
|
24
|
+
supported_output_types = {"pdf", "txt", "html", "png", "jpg", "webp"}
|
|
25
|
+
|
|
26
|
+
def __init__(self, path: Path, *, temp_storage: Any = None) -> None:
|
|
27
|
+
super().__init__(path, temp_storage=temp_storage)
|
|
28
|
+
try:
|
|
29
|
+
self._reader = PdfReader(str(path))
|
|
30
|
+
except PdfReadError as exc:
|
|
31
|
+
raise InvalidDocument(f"Invalid or corrupted PDF: {exc}") from exc
|
|
32
|
+
|
|
33
|
+
# ------------------------------------------------------------------
|
|
34
|
+
# Basic conversions
|
|
35
|
+
# ------------------------------------------------------------------
|
|
36
|
+
|
|
37
|
+
def to_pdf(self) -> Path:
|
|
38
|
+
"""Return a copy of the PDF (identity conversion)."""
|
|
39
|
+
out = self._temp_path(suffix=".pdf", prefix="copy_")
|
|
40
|
+
writer = PdfWriter()
|
|
41
|
+
for page in self._reader.pages:
|
|
42
|
+
writer.add_page(page)
|
|
43
|
+
with out.open("wb") as f:
|
|
44
|
+
writer.write(f)
|
|
45
|
+
return out
|
|
46
|
+
|
|
47
|
+
def to_text(self) -> str:
|
|
48
|
+
"""Extract text from all pages."""
|
|
49
|
+
parts: list[str] = []
|
|
50
|
+
for i, page in enumerate(self._reader.pages, start=1):
|
|
51
|
+
text = page.extract_text() or ""
|
|
52
|
+
if text.strip():
|
|
53
|
+
parts.append(f"--- Page {i} ---\n{text.strip()}")
|
|
54
|
+
return "\n\n".join(parts)
|
|
55
|
+
|
|
56
|
+
def to_html(self) -> str:
|
|
57
|
+
"""Simple HTML representation of extracted text."""
|
|
58
|
+
text = self.to_text()
|
|
59
|
+
# Very basic wrapping not a full visual conversion
|
|
60
|
+
escaped = (
|
|
61
|
+
text.replace("&", "&")
|
|
62
|
+
.replace("<", "<")
|
|
63
|
+
.replace(">", ">")
|
|
64
|
+
.replace("\n", "<br>\n")
|
|
65
|
+
)
|
|
66
|
+
return f"<html><body><pre>{escaped}</pre></body></html>"
|
|
67
|
+
|
|
68
|
+
def to_images(self, *, format: str = "png") -> list[Path]:
|
|
69
|
+
"""Convert PDF pages to images.
|
|
70
|
+
|
|
71
|
+
Requires the optional `pdf2image` package (and poppler).
|
|
72
|
+
If unavailable, raises a clear ProcessingError.
|
|
73
|
+
"""
|
|
74
|
+
try:
|
|
75
|
+
from pdf2image import convert_from_path
|
|
76
|
+
except ImportError as exc:
|
|
77
|
+
raise ProcessingError(
|
|
78
|
+
"PDF -> images requires the optional dependency 'pdf2image' "
|
|
79
|
+
"and the system package 'poppler'. "
|
|
80
|
+
"Install with: pip install pdf2image"
|
|
81
|
+
) from exc
|
|
82
|
+
|
|
83
|
+
fmt = format.lower().lstrip(".")
|
|
84
|
+
if fmt not in {"png", "jpg", "jpeg", "webp"}:
|
|
85
|
+
raise ProcessingError(f"Unsupported image format: {format}")
|
|
86
|
+
|
|
87
|
+
images = convert_from_path(str(self.path))
|
|
88
|
+
paths: list[Path] = []
|
|
89
|
+
for i, img in enumerate(images, start=1):
|
|
90
|
+
suffix = f".{fmt if fmt != 'jpeg' else 'jpg'}"
|
|
91
|
+
out = self._temp_path(suffix=suffix, prefix=f"page_{i}_")
|
|
92
|
+
img.save(str(out), fmt.upper() if fmt != "jpg" else "JPEG")
|
|
93
|
+
paths.append(out)
|
|
94
|
+
return paths
|
|
95
|
+
|
|
96
|
+
def info(self) -> dict[str, Any]:
|
|
97
|
+
meta = self._reader.metadata or {}
|
|
98
|
+
return {
|
|
99
|
+
"filename": self.path.name,
|
|
100
|
+
"type": "pdf",
|
|
101
|
+
"pages": len(self._reader.pages),
|
|
102
|
+
"size": self.path.stat().st_size,
|
|
103
|
+
"extension": ".pdf",
|
|
104
|
+
"title": getattr(meta, "title", None) or meta.get("/Title"),
|
|
105
|
+
"author": getattr(meta, "author", None) or meta.get("/Author"),
|
|
106
|
+
"subject": getattr(meta, "subject", None) or meta.get("/Subject"),
|
|
107
|
+
"creator": getattr(meta, "creator", None) or meta.get("/Creator"),
|
|
108
|
+
"producer": getattr(meta, "producer", None) or meta.get("/Producer"),
|
|
109
|
+
"creation_date": str(getattr(meta, "creation_date", None) or meta.get("/CreationDate") or ""),
|
|
110
|
+
"modification_date": str(getattr(meta, "modification_date", None) or meta.get("/ModDate") or ""),
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
# ------------------------------------------------------------------
|
|
114
|
+
# PDF operations
|
|
115
|
+
# ------------------------------------------------------------------
|
|
116
|
+
|
|
117
|
+
def compress(self, level: str = "medium") -> Path:
|
|
118
|
+
"""Compress the PDF.
|
|
119
|
+
|
|
120
|
+
Levels:
|
|
121
|
+
low minimal compression (mostly remove unused objects)
|
|
122
|
+
medium default, moderate image quality reduction if possible
|
|
123
|
+
high aggressive (may reduce quality)
|
|
124
|
+
|
|
125
|
+
Note: pypdf's compression is limited; this is a best-effort implementation.
|
|
126
|
+
"""
|
|
127
|
+
level = level.lower()
|
|
128
|
+
if level not in {"low", "medium", "high"}:
|
|
129
|
+
raise ProcessingError(f"Unknown compression level: {level}")
|
|
130
|
+
|
|
131
|
+
writer = PdfWriter()
|
|
132
|
+
for page in self._reader.pages:
|
|
133
|
+
# Basic content stream compression
|
|
134
|
+
page.compress_content_streams()
|
|
135
|
+
writer.add_page(page)
|
|
136
|
+
|
|
137
|
+
# Remove unused objects / metadata for higher levels
|
|
138
|
+
if level in {"medium", "high"}:
|
|
139
|
+
try:
|
|
140
|
+
writer.compress_identical_objects(
|
|
141
|
+
remove_duplicates=True, remove_unreferenced=True
|
|
142
|
+
)
|
|
143
|
+
except TypeError:
|
|
144
|
+
# older pypdf API
|
|
145
|
+
writer.compress_identical_objects(
|
|
146
|
+
remove_identicals=True, remove_orphans=True
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
out = self._temp_path(suffix=".pdf", prefix="compressed_")
|
|
150
|
+
with out.open("wb") as f:
|
|
151
|
+
writer.write(f)
|
|
152
|
+
return out
|
|
153
|
+
|
|
154
|
+
def split(self) -> list[Path]:
|
|
155
|
+
"""Split into one PDF per page."""
|
|
156
|
+
paths: list[Path] = []
|
|
157
|
+
for i, page in enumerate(self._reader.pages, start=1):
|
|
158
|
+
writer = PdfWriter()
|
|
159
|
+
writer.add_page(page)
|
|
160
|
+
out = self._temp_path(suffix=".pdf", prefix=f"page_{i}_")
|
|
161
|
+
with out.open("wb") as f:
|
|
162
|
+
writer.write(f)
|
|
163
|
+
paths.append(out)
|
|
164
|
+
return paths
|
|
165
|
+
|
|
166
|
+
def extract_pages(self, start: int, end: int | None = None) -> Path:
|
|
167
|
+
"""Extract pages start..end (1-based inclusive)."""
|
|
168
|
+
total = len(self._reader.pages)
|
|
169
|
+
if start < 1 or start > total:
|
|
170
|
+
raise ProcessingError(f"Start page {start} out of range (1-{total}).")
|
|
171
|
+
if end is None:
|
|
172
|
+
end = total
|
|
173
|
+
if end < start or end > total:
|
|
174
|
+
raise ProcessingError(f"End page {end} out of range ({start}-{total}).")
|
|
175
|
+
|
|
176
|
+
writer = PdfWriter()
|
|
177
|
+
for i in range(start - 1, end): # convert to 0-based
|
|
178
|
+
writer.add_page(self._reader.pages[i])
|
|
179
|
+
|
|
180
|
+
out = self._temp_path(
|
|
181
|
+
suffix=".pdf",
|
|
182
|
+
prefix=f"pages_{start}-{end}_",
|
|
183
|
+
)
|
|
184
|
+
with out.open("wb") as f:
|
|
185
|
+
writer.write(f)
|
|
186
|
+
return out
|
|
187
|
+
|
|
188
|
+
def rotate(self, degrees: int = 90, pages: Sequence[int] | None = None) -> Path:
|
|
189
|
+
"""Rotate pages by degrees (90, 180, 270). pages is 1-based."""
|
|
190
|
+
if degrees % 90 != 0:
|
|
191
|
+
raise ProcessingError("Rotation must be a multiple of 90 degrees.")
|
|
192
|
+
|
|
193
|
+
writer = PdfWriter()
|
|
194
|
+
total = len(self._reader.pages)
|
|
195
|
+
page_set = set(pages) if pages else None
|
|
196
|
+
|
|
197
|
+
for i, page in enumerate(self._reader.pages, start=1):
|
|
198
|
+
if page_set is None or i in page_set:
|
|
199
|
+
page.rotate(degrees)
|
|
200
|
+
writer.add_page(page)
|
|
201
|
+
|
|
202
|
+
out = self._temp_path(suffix=".pdf", prefix="rotated_")
|
|
203
|
+
with out.open("wb") as f:
|
|
204
|
+
writer.write(f)
|
|
205
|
+
return out
|
|
206
|
+
|
|
207
|
+
# ------------------------------------------------------------------
|
|
208
|
+
# Helpers
|
|
209
|
+
# ------------------------------------------------------------------
|
|
210
|
+
|
|
211
|
+
def _temp_path(self, suffix: str = "", prefix: str = "pdf_") -> Path:
|
|
212
|
+
storage = self.temp_storage or TemporaryStorage()
|
|
213
|
+
return storage.create_file(suffix=suffix, prefix=prefix)
|
|
214
|
+
|
|
215
|
+
@staticmethod
|
|
216
|
+
def merge(paths: Sequence[Path], *, temp_storage: TemporaryStorage | None = None) -> Path:
|
|
217
|
+
"""Merge multiple PDFs into one."""
|
|
218
|
+
writer = PdfWriter()
|
|
219
|
+
for p in paths:
|
|
220
|
+
try:
|
|
221
|
+
reader = PdfReader(str(p))
|
|
222
|
+
for page in reader.pages:
|
|
223
|
+
writer.add_page(page)
|
|
224
|
+
except PdfReadError as exc:
|
|
225
|
+
raise InvalidDocument(f"Cannot read PDF for merge: {p.name}: {exc}") from exc
|
|
226
|
+
|
|
227
|
+
storage = temp_storage or TemporaryStorage()
|
|
228
|
+
out = storage.create_file(suffix=".pdf", prefix="merged_")
|
|
229
|
+
with out.open("wb") as f:
|
|
230
|
+
writer.write(f)
|
|
231
|
+
return out
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""Plain-text processor for eDocAPI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from edocapi.processors.base import BaseProcessor, register_processor
|
|
9
|
+
from edocapi.storage.temporary import TemporaryStorage
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@register_processor("txt")
|
|
13
|
+
class TXTProcessor(BaseProcessor):
|
|
14
|
+
"""Processor for plain-text documents."""
|
|
15
|
+
|
|
16
|
+
supported_input_types = {"txt"}
|
|
17
|
+
supported_output_types = {"txt", "html", "pdf"}
|
|
18
|
+
|
|
19
|
+
def to_text(self) -> str:
|
|
20
|
+
return self.path.read_text(encoding="utf-8", errors="replace")
|
|
21
|
+
|
|
22
|
+
def to_html(self) -> str:
|
|
23
|
+
text = (
|
|
24
|
+
self.to_text()
|
|
25
|
+
.replace("&", "&")
|
|
26
|
+
.replace("<", "<")
|
|
27
|
+
.replace(">", ">")
|
|
28
|
+
.replace("\n", "<br>\n")
|
|
29
|
+
)
|
|
30
|
+
return f"<html><body><pre>{text}</pre></body></html>"
|
|
31
|
+
|
|
32
|
+
def to_pdf(self) -> Path:
|
|
33
|
+
html = self.to_html()
|
|
34
|
+
try:
|
|
35
|
+
from weasyprint import HTML
|
|
36
|
+
except ImportError as exc:
|
|
37
|
+
from edocapi.exceptions import ConversionError
|
|
38
|
+
raise ConversionError(
|
|
39
|
+
"TXT -> PDF requires weasyprint. Install with: pip install edocapi[html]",
|
|
40
|
+
source="txt",
|
|
41
|
+
target="pdf",
|
|
42
|
+
) from exc
|
|
43
|
+
|
|
44
|
+
storage = self.temp_storage or TemporaryStorage()
|
|
45
|
+
out = storage.create_file(suffix=".pdf", prefix="txt_")
|
|
46
|
+
HTML(string=html).write_pdf(str(out))
|
|
47
|
+
return out
|
|
48
|
+
|
|
49
|
+
def info(self) -> dict[str, Any]:
|
|
50
|
+
content = self.to_text()
|
|
51
|
+
return {
|
|
52
|
+
"filename": self.path.name,
|
|
53
|
+
"type": "txt",
|
|
54
|
+
"size": self.path.stat().st_size,
|
|
55
|
+
"extension": ".txt",
|
|
56
|
+
"lines": content.count("\n") + 1,
|
|
57
|
+
"characters": len(content),
|
|
58
|
+
}
|