edocapi 0.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,112 @@
1
+ """DOCX processor for eDocAPI (uses python-docx + weasyprint for PDF)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from docx import Document as DocxDocument
10
+
11
+ from edocapi.exceptions import ConversionError, InvalidDocument
12
+ from edocapi.processors.base import BaseProcessor, register_processor
13
+ from edocapi.storage.temporary import TemporaryStorage
14
+
15
+ logger = logging.getLogger("edocapi.processors.docx")
16
+
17
+
18
+ @register_processor("docx")
19
+ class DOCXProcessor(BaseProcessor):
20
+ """Processor for Microsoft Word (.docx) documents."""
21
+
22
+ supported_input_types = {"docx"}
23
+ supported_output_types = {"pdf", "txt", "html"}
24
+
25
+ def __init__(self, path: Path, *, temp_storage: Any = None) -> None:
26
+ super().__init__(path, temp_storage=temp_storage)
27
+ try:
28
+ self._doc = DocxDocument(str(path))
29
+ except Exception as exc:
30
+ raise InvalidDocument(f"Invalid or corrupted DOCX: {exc}") from exc
31
+
32
+ def to_text(self) -> str:
33
+ """Extract all paragraph text."""
34
+ paragraphs = [p.text for p in self._doc.paragraphs if p.text.strip()]
35
+ # Also extract tables
36
+ for table in self._doc.tables:
37
+ for row in table.rows:
38
+ cells = [cell.text.strip() for cell in row.cells]
39
+ paragraphs.append("\t".join(cells))
40
+ return "\n".join(paragraphs)
41
+
42
+ def to_html(self) -> str:
43
+ """Convert DOCX to a simple HTML representation."""
44
+ parts: list[str] = ["<html><body>"]
45
+ for p in self._doc.paragraphs:
46
+ style = p.style.name.lower() if p.style else ""
47
+ text = (
48
+ p.text.replace("&", "&amp;")
49
+ .replace("<", "&lt;")
50
+ .replace(">", "&gt;")
51
+ )
52
+ if not text.strip():
53
+ continue
54
+ if "heading 1" in style:
55
+ parts.append(f"<h1>{text}</h1>")
56
+ elif "heading 2" in style:
57
+ parts.append(f"<h2>{text}</h2>")
58
+ elif "heading 3" in style:
59
+ parts.append(f"<h3>{text}</h3>")
60
+ else:
61
+ parts.append(f"<p>{text}</p>")
62
+
63
+ for table in self._doc.tables:
64
+ parts.append("<table border='1'>")
65
+ for row in table.rows:
66
+ parts.append("<tr>")
67
+ for cell in row.cells:
68
+ cell_text = (
69
+ cell.text.replace("&", "&amp;")
70
+ .replace("<", "&lt;")
71
+ .replace(">", "&gt;")
72
+ )
73
+ parts.append(f"<td>{cell_text}</td>")
74
+ parts.append("</tr>")
75
+ parts.append("</table>")
76
+
77
+ parts.append("</body></html>")
78
+ return "\n".join(parts)
79
+
80
+ def to_pdf(self) -> Path:
81
+ """Convert DOCX -> HTML -> PDF via WeasyPrint."""
82
+ html = self.to_html()
83
+ try:
84
+ from weasyprint import HTML
85
+ except ImportError as exc:
86
+ raise ConversionError(
87
+ "DOCX -> PDF requires weasyprint. "
88
+ "Install with: pip install edocapi[html]",
89
+ source="docx",
90
+ target="pdf",
91
+ ) from exc
92
+
93
+ storage = self.temp_storage or TemporaryStorage()
94
+ out = storage.create_file(suffix=".pdf", prefix="docx_")
95
+ HTML(string=html).write_pdf(str(out))
96
+ return out
97
+
98
+ def info(self) -> dict[str, Any]:
99
+ core = self._doc.core_properties
100
+ return {
101
+ "filename": self.path.name,
102
+ "type": "docx",
103
+ "size": self.path.stat().st_size,
104
+ "extension": ".docx",
105
+ "title": core.title or None,
106
+ "author": core.author or None,
107
+ "subject": core.subject or None,
108
+ "created": str(core.created) if core.created else None,
109
+ "modified": str(core.modified) if core.modified else None,
110
+ "paragraphs": len(self._doc.paragraphs),
111
+ "tables": len(self._doc.tables),
112
+ }
@@ -0,0 +1,64 @@
1
+ """HTML processor for eDocAPI (uses weasyprint for PDF)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ import re
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+ from edocapi.exceptions import ConversionError
11
+ from edocapi.processors.base import BaseProcessor, register_processor
12
+ from edocapi.storage.temporary import TemporaryStorage
13
+
14
+ logger = logging.getLogger("edocapi.processors.html")
15
+
16
+
17
+ @register_processor("html")
18
+ class HTMLProcessor(BaseProcessor):
19
+ """Processor for HTML documents."""
20
+
21
+ supported_input_types = {"html"}
22
+ supported_output_types = {"pdf", "txt", "html"}
23
+
24
+ def to_text(self) -> str:
25
+ """Strip tags and return approximate plain text."""
26
+ content = self.path.read_text(encoding="utf-8", errors="replace")
27
+ # Very simple tag stripping
28
+ text = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL | re.I)
29
+ text = re.sub(r"<style[^>]*>.*?</style>", "", text, flags=re.DOTALL | re.I)
30
+ text = re.sub(r"<[^>]+>", " ", text)
31
+ text = re.sub(r"\s+", " ", text)
32
+ return text.strip()
33
+
34
+ def to_html(self) -> str:
35
+ return self.path.read_text(encoding="utf-8", errors="replace")
36
+
37
+ def to_pdf(self) -> Path:
38
+ """Convert HTML to PDF using WeasyPrint."""
39
+ try:
40
+ from weasyprint import HTML
41
+ except ImportError as exc:
42
+ raise ConversionError(
43
+ "HTML -> PDF requires weasyprint. "
44
+ "Install with: pip install edocapi[html]",
45
+ source="html",
46
+ target="pdf",
47
+ ) from exc
48
+
49
+ storage = self.temp_storage or TemporaryStorage()
50
+ out = storage.create_file(suffix=".pdf", prefix="html_")
51
+ HTML(filename=str(self.path)).write_pdf(str(out))
52
+ return out
53
+
54
+ def info(self) -> dict[str, Any]:
55
+ content = self.path.read_text(encoding="utf-8", errors="replace")
56
+ title_match = re.search(r"<title[^>]*>(.*?)</title>", content, re.I | re.DOTALL)
57
+ title = title_match.group(1).strip() if title_match else None
58
+ return {
59
+ "filename": self.path.name,
60
+ "type": "html",
61
+ "size": self.path.stat().st_size,
62
+ "extension": self.path.suffix.lower(),
63
+ "title": title,
64
+ }
@@ -0,0 +1,80 @@
1
+ """Image processor for eDocAPI (uses Pillow + weasyprint / reportlab-like via PDF)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ from PIL import Image
10
+
11
+ from edocapi.exceptions import ConversionError, InvalidDocument
12
+ from edocapi.processors.base import BaseProcessor, register_processor
13
+ from edocapi.storage.temporary import TemporaryStorage
14
+
15
+ logger = logging.getLogger("edocapi.processors.image")
16
+
17
+
18
+ @register_processor("jpeg", "jpg", "png", "webp")
19
+ class ImageProcessor(BaseProcessor):
20
+ """Processor for common image formats."""
21
+
22
+ supported_input_types = {"jpeg", "jpg", "png", "webp"}
23
+ supported_output_types = {"pdf", "png", "jpg", "webp"}
24
+
25
+ def __init__(self, path: Path, *, temp_storage: Any = None) -> None:
26
+ super().__init__(path, temp_storage=temp_storage)
27
+ try:
28
+ self._img = Image.open(str(path))
29
+ self._img.load() # force load to validate
30
+ except Exception as exc:
31
+ raise InvalidDocument(f"Invalid or corrupted image: {exc}") from exc
32
+
33
+ def to_pdf(self) -> Path:
34
+ """Convert image to a single-page PDF."""
35
+ storage = self.temp_storage or TemporaryStorage()
36
+ out = storage.create_file(suffix=".pdf", prefix="img_")
37
+
38
+ # Convert to RGB if necessary (PDF does not support all modes)
39
+ img = self._img
40
+ if img.mode in ("RGBA", "P", "LA"):
41
+ background = Image.new("RGB", img.size, (255, 255, 255))
42
+ if img.mode == "P":
43
+ img = img.convert("RGBA")
44
+ background.paste(img, mask=img.split()[-1] if img.mode in ("RGBA", "LA") else None)
45
+ img = background
46
+ elif img.mode != "RGB":
47
+ img = img.convert("RGB")
48
+
49
+ img.save(str(out), "PDF", resolution=100.0)
50
+ return out
51
+
52
+ def to_text(self) -> str:
53
+ raise NotImplementedError(
54
+ "Text extraction from images requires OCR, which is not available in v0.0.2."
55
+ )
56
+
57
+ def to_images(self, *, format: str = "png") -> list[Path]:
58
+ """Return a copy of the image in the requested format."""
59
+ fmt = format.lower().lstrip(".")
60
+ if fmt not in {"png", "jpg", "jpeg", "webp"}:
61
+ raise ConversionError(f"Unsupported image format: {format}")
62
+
63
+ storage = self.temp_storage or TemporaryStorage()
64
+ suffix = f".{fmt if fmt != 'jpeg' else 'jpg'}"
65
+ out = storage.create_file(suffix=suffix, prefix="img_")
66
+ save_fmt = "JPEG" if fmt in ("jpg", "jpeg") else fmt.upper()
67
+ self._img.save(str(out), save_fmt)
68
+ return [out]
69
+
70
+ def info(self) -> dict[str, Any]:
71
+ return {
72
+ "filename": self.path.name,
73
+ "type": self.path.suffix.lstrip(".").lower(),
74
+ "size": self.path.stat().st_size,
75
+ "extension": self.path.suffix.lower(),
76
+ "width": self._img.width,
77
+ "height": self._img.height,
78
+ "mode": self._img.mode,
79
+ "format": self._img.format,
80
+ }
@@ -0,0 +1,64 @@
1
+ """Markdown processor for eDocAPI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from pathlib import Path
7
+ from typing import Any
8
+
9
+ import markdown
10
+
11
+ from edocapi.exceptions import ConversionError
12
+ from edocapi.processors.base import BaseProcessor, register_processor
13
+ from edocapi.storage.temporary import TemporaryStorage
14
+
15
+ logger = logging.getLogger("edocapi.processors.markdown")
16
+
17
+
18
+ @register_processor("markdown", "md")
19
+ class MarkdownProcessor(BaseProcessor):
20
+ """Processor for Markdown documents."""
21
+
22
+ supported_input_types = {"markdown", "md"}
23
+ supported_output_types = {"pdf", "html", "txt"}
24
+
25
+ def to_text(self) -> str:
26
+ """Return the raw Markdown (or strip to plain text)."""
27
+ return self.path.read_text(encoding="utf-8", errors="replace")
28
+
29
+ def to_html(self) -> str:
30
+ """Convert Markdown to HTML using the markdown library."""
31
+ md = self.path.read_text(encoding="utf-8", errors="replace")
32
+ html_body = markdown.markdown(
33
+ md,
34
+ extensions=["extra", "codehilite", "tables", "fenced_code"],
35
+ )
36
+ return f"<!DOCTYPE html><html><body>{html_body}</body></html>"
37
+
38
+ def to_pdf(self) -> Path:
39
+ """Markdown -> HTML -> PDF."""
40
+ html = self.to_html()
41
+ try:
42
+ from weasyprint import HTML
43
+ except ImportError as exc:
44
+ raise ConversionError(
45
+ "Markdown -> PDF requires weasyprint. "
46
+ "Install with: pip install edocapi[html]",
47
+ source="markdown",
48
+ target="pdf",
49
+ ) from exc
50
+
51
+ storage = self.temp_storage or TemporaryStorage()
52
+ out = storage.create_file(suffix=".pdf", prefix="md_")
53
+ HTML(string=html).write_pdf(str(out))
54
+ return out
55
+
56
+ def info(self) -> dict[str, Any]:
57
+ content = self.path.read_text(encoding="utf-8", errors="replace")
58
+ return {
59
+ "filename": self.path.name,
60
+ "type": "markdown",
61
+ "size": self.path.stat().st_size,
62
+ "extension": self.path.suffix.lower(),
63
+ "lines": content.count("\n") + 1,
64
+ }
@@ -0,0 +1,231 @@
1
+ """PDF processor for eDocAPI (uses pypdf)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from pathlib import Path
7
+ from typing import Any, Sequence
8
+
9
+ from pypdf import PdfReader, PdfWriter
10
+ from pypdf.errors import PdfReadError
11
+
12
+ from edocapi.exceptions import InvalidDocument, ProcessingError
13
+ from edocapi.processors.base import BaseProcessor, register_processor
14
+ from edocapi.storage.temporary import TemporaryStorage
15
+
16
+ logger = logging.getLogger("edocapi.processors.pdf")
17
+
18
+
19
+ @register_processor("pdf")
20
+ class PDFProcessor(BaseProcessor):
21
+ """Processor for PDF documents."""
22
+
23
+ supported_input_types = {"pdf"}
24
+ supported_output_types = {"pdf", "txt", "html", "png", "jpg", "webp"}
25
+
26
+ def __init__(self, path: Path, *, temp_storage: Any = None) -> None:
27
+ super().__init__(path, temp_storage=temp_storage)
28
+ try:
29
+ self._reader = PdfReader(str(path))
30
+ except PdfReadError as exc:
31
+ raise InvalidDocument(f"Invalid or corrupted PDF: {exc}") from exc
32
+
33
+ # ------------------------------------------------------------------
34
+ # Basic conversions
35
+ # ------------------------------------------------------------------
36
+
37
+ def to_pdf(self) -> Path:
38
+ """Return a copy of the PDF (identity conversion)."""
39
+ out = self._temp_path(suffix=".pdf", prefix="copy_")
40
+ writer = PdfWriter()
41
+ for page in self._reader.pages:
42
+ writer.add_page(page)
43
+ with out.open("wb") as f:
44
+ writer.write(f)
45
+ return out
46
+
47
+ def to_text(self) -> str:
48
+ """Extract text from all pages."""
49
+ parts: list[str] = []
50
+ for i, page in enumerate(self._reader.pages, start=1):
51
+ text = page.extract_text() or ""
52
+ if text.strip():
53
+ parts.append(f"--- Page {i} ---\n{text.strip()}")
54
+ return "\n\n".join(parts)
55
+
56
+ def to_html(self) -> str:
57
+ """Simple HTML representation of extracted text."""
58
+ text = self.to_text()
59
+ # Very basic wrapping  not a full visual conversion
60
+ escaped = (
61
+ text.replace("&", "&amp;")
62
+ .replace("<", "&lt;")
63
+ .replace(">", "&gt;")
64
+ .replace("\n", "<br>\n")
65
+ )
66
+ return f"<html><body><pre>{escaped}</pre></body></html>"
67
+
68
+ def to_images(self, *, format: str = "png") -> list[Path]:
69
+ """Convert PDF pages to images.
70
+
71
+ Requires the optional `pdf2image` package (and poppler).
72
+ If unavailable, raises a clear ProcessingError.
73
+ """
74
+ try:
75
+ from pdf2image import convert_from_path
76
+ except ImportError as exc:
77
+ raise ProcessingError(
78
+ "PDF -> images requires the optional dependency 'pdf2image' "
79
+ "and the system package 'poppler'. "
80
+ "Install with: pip install pdf2image"
81
+ ) from exc
82
+
83
+ fmt = format.lower().lstrip(".")
84
+ if fmt not in {"png", "jpg", "jpeg", "webp"}:
85
+ raise ProcessingError(f"Unsupported image format: {format}")
86
+
87
+ images = convert_from_path(str(self.path))
88
+ paths: list[Path] = []
89
+ for i, img in enumerate(images, start=1):
90
+ suffix = f".{fmt if fmt != 'jpeg' else 'jpg'}"
91
+ out = self._temp_path(suffix=suffix, prefix=f"page_{i}_")
92
+ img.save(str(out), fmt.upper() if fmt != "jpg" else "JPEG")
93
+ paths.append(out)
94
+ return paths
95
+
96
+ def info(self) -> dict[str, Any]:
97
+ meta = self._reader.metadata or {}
98
+ return {
99
+ "filename": self.path.name,
100
+ "type": "pdf",
101
+ "pages": len(self._reader.pages),
102
+ "size": self.path.stat().st_size,
103
+ "extension": ".pdf",
104
+ "title": getattr(meta, "title", None) or meta.get("/Title"),
105
+ "author": getattr(meta, "author", None) or meta.get("/Author"),
106
+ "subject": getattr(meta, "subject", None) or meta.get("/Subject"),
107
+ "creator": getattr(meta, "creator", None) or meta.get("/Creator"),
108
+ "producer": getattr(meta, "producer", None) or meta.get("/Producer"),
109
+ "creation_date": str(getattr(meta, "creation_date", None) or meta.get("/CreationDate") or ""),
110
+ "modification_date": str(getattr(meta, "modification_date", None) or meta.get("/ModDate") or ""),
111
+ }
112
+
113
+ # ------------------------------------------------------------------
114
+ # PDF operations
115
+ # ------------------------------------------------------------------
116
+
117
+ def compress(self, level: str = "medium") -> Path:
118
+ """Compress the PDF.
119
+
120
+ Levels:
121
+ low  minimal compression (mostly remove unused objects)
122
+ medium  default, moderate image quality reduction if possible
123
+ high  aggressive (may reduce quality)
124
+
125
+ Note: pypdf's compression is limited; this is a best-effort implementation.
126
+ """
127
+ level = level.lower()
128
+ if level not in {"low", "medium", "high"}:
129
+ raise ProcessingError(f"Unknown compression level: {level}")
130
+
131
+ writer = PdfWriter()
132
+ for page in self._reader.pages:
133
+ # Basic content stream compression
134
+ page.compress_content_streams()
135
+ writer.add_page(page)
136
+
137
+ # Remove unused objects / metadata for higher levels
138
+ if level in {"medium", "high"}:
139
+ try:
140
+ writer.compress_identical_objects(
141
+ remove_duplicates=True, remove_unreferenced=True
142
+ )
143
+ except TypeError:
144
+ # older pypdf API
145
+ writer.compress_identical_objects(
146
+ remove_identicals=True, remove_orphans=True
147
+ )
148
+
149
+ out = self._temp_path(suffix=".pdf", prefix="compressed_")
150
+ with out.open("wb") as f:
151
+ writer.write(f)
152
+ return out
153
+
154
+ def split(self) -> list[Path]:
155
+ """Split into one PDF per page."""
156
+ paths: list[Path] = []
157
+ for i, page in enumerate(self._reader.pages, start=1):
158
+ writer = PdfWriter()
159
+ writer.add_page(page)
160
+ out = self._temp_path(suffix=".pdf", prefix=f"page_{i}_")
161
+ with out.open("wb") as f:
162
+ writer.write(f)
163
+ paths.append(out)
164
+ return paths
165
+
166
+ def extract_pages(self, start: int, end: int | None = None) -> Path:
167
+ """Extract pages start..end (1-based inclusive)."""
168
+ total = len(self._reader.pages)
169
+ if start < 1 or start > total:
170
+ raise ProcessingError(f"Start page {start} out of range (1-{total}).")
171
+ if end is None:
172
+ end = total
173
+ if end < start or end > total:
174
+ raise ProcessingError(f"End page {end} out of range ({start}-{total}).")
175
+
176
+ writer = PdfWriter()
177
+ for i in range(start - 1, end): # convert to 0-based
178
+ writer.add_page(self._reader.pages[i])
179
+
180
+ out = self._temp_path(
181
+ suffix=".pdf",
182
+ prefix=f"pages_{start}-{end}_",
183
+ )
184
+ with out.open("wb") as f:
185
+ writer.write(f)
186
+ return out
187
+
188
+ def rotate(self, degrees: int = 90, pages: Sequence[int] | None = None) -> Path:
189
+ """Rotate pages by degrees (90, 180, 270). pages is 1-based."""
190
+ if degrees % 90 != 0:
191
+ raise ProcessingError("Rotation must be a multiple of 90 degrees.")
192
+
193
+ writer = PdfWriter()
194
+ total = len(self._reader.pages)
195
+ page_set = set(pages) if pages else None
196
+
197
+ for i, page in enumerate(self._reader.pages, start=1):
198
+ if page_set is None or i in page_set:
199
+ page.rotate(degrees)
200
+ writer.add_page(page)
201
+
202
+ out = self._temp_path(suffix=".pdf", prefix="rotated_")
203
+ with out.open("wb") as f:
204
+ writer.write(f)
205
+ return out
206
+
207
+ # ------------------------------------------------------------------
208
+ # Helpers
209
+ # ------------------------------------------------------------------
210
+
211
+ def _temp_path(self, suffix: str = "", prefix: str = "pdf_") -> Path:
212
+ storage = self.temp_storage or TemporaryStorage()
213
+ return storage.create_file(suffix=suffix, prefix=prefix)
214
+
215
+ @staticmethod
216
+ def merge(paths: Sequence[Path], *, temp_storage: TemporaryStorage | None = None) -> Path:
217
+ """Merge multiple PDFs into one."""
218
+ writer = PdfWriter()
219
+ for p in paths:
220
+ try:
221
+ reader = PdfReader(str(p))
222
+ for page in reader.pages:
223
+ writer.add_page(page)
224
+ except PdfReadError as exc:
225
+ raise InvalidDocument(f"Cannot read PDF for merge: {p.name}: {exc}") from exc
226
+
227
+ storage = temp_storage or TemporaryStorage()
228
+ out = storage.create_file(suffix=".pdf", prefix="merged_")
229
+ with out.open("wb") as f:
230
+ writer.write(f)
231
+ return out
@@ -0,0 +1,58 @@
1
+ """Plain-text processor for eDocAPI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ from edocapi.processors.base import BaseProcessor, register_processor
9
+ from edocapi.storage.temporary import TemporaryStorage
10
+
11
+
12
+ @register_processor("txt")
13
+ class TXTProcessor(BaseProcessor):
14
+ """Processor for plain-text documents."""
15
+
16
+ supported_input_types = {"txt"}
17
+ supported_output_types = {"txt", "html", "pdf"}
18
+
19
+ def to_text(self) -> str:
20
+ return self.path.read_text(encoding="utf-8", errors="replace")
21
+
22
+ def to_html(self) -> str:
23
+ text = (
24
+ self.to_text()
25
+ .replace("&", "&amp;")
26
+ .replace("<", "&lt;")
27
+ .replace(">", "&gt;")
28
+ .replace("\n", "<br>\n")
29
+ )
30
+ return f"<html><body><pre>{text}</pre></body></html>"
31
+
32
+ def to_pdf(self) -> Path:
33
+ html = self.to_html()
34
+ try:
35
+ from weasyprint import HTML
36
+ except ImportError as exc:
37
+ from edocapi.exceptions import ConversionError
38
+ raise ConversionError(
39
+ "TXT -> PDF requires weasyprint. Install with: pip install edocapi[html]",
40
+ source="txt",
41
+ target="pdf",
42
+ ) from exc
43
+
44
+ storage = self.temp_storage or TemporaryStorage()
45
+ out = storage.create_file(suffix=".pdf", prefix="txt_")
46
+ HTML(string=html).write_pdf(str(out))
47
+ return out
48
+
49
+ def info(self) -> dict[str, Any]:
50
+ content = self.to_text()
51
+ return {
52
+ "filename": self.path.name,
53
+ "type": "txt",
54
+ "size": self.path.stat().st_size,
55
+ "extension": ".txt",
56
+ "lines": content.count("\n") + 1,
57
+ "characters": len(content),
58
+ }