edocapi 0.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
edocapi/document.py ADDED
@@ -0,0 +1,290 @@
1
+ """Document abstraction for eDocAPI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from pathlib import Path
7
+ from typing import Any, Sequence
8
+
9
+ from edocapi.exceptions import ConversionError, InvalidDocument, ProcessingError
10
+ from edocapi.processors.base import get_processor_class, list_supported_types
11
+ from edocapi.storage.temporary import TemporaryStorage, safe_filename
12
+ from edocapi.validation.files import validate_file
13
+
14
+ logger = logging.getLogger("edocapi.document")
15
+
16
+
17
+ class Document:
18
+ """High-level document abstraction.
19
+
20
+ Developers interact primarily with this class::
21
+
22
+ Document(file).to_pdf()
23
+ Document(file).extract_text()
24
+ Document.merge([f1, f2])
25
+ """
26
+
27
+ def __init__(
28
+ self,
29
+ source: Path | str | bytes | None = None,
30
+ *,
31
+ filename: str | None = None,
32
+ doc_type: str | None = None,
33
+ temp_storage: TemporaryStorage | None = None,
34
+ max_size: int = 50 * 1024 * 1024,
35
+ ) -> None:
36
+ self._temp = temp_storage or TemporaryStorage()
37
+ self._max_size = max_size
38
+ self._path: Path | None = None
39
+ self._type: str | None = doc_type
40
+ self._name: str | None = filename
41
+
42
+ if source is None:
43
+ return # empty document (used by classmethods)
44
+
45
+ if isinstance(source, (str, Path)):
46
+ path = Path(source)
47
+ self._path = path
48
+ self._name = filename or path.name
49
+ self._type = validate_file(path, max_size=max_size)
50
+ elif isinstance(source, bytes):
51
+ if not filename:
52
+ raise InvalidDocument("filename is required when creating Document from bytes")
53
+ safe = safe_filename(filename)
54
+ path = self._temp.create_file(
55
+ suffix=Path(safe).suffix,
56
+ prefix="doc_",
57
+ content=source,
58
+ )
59
+ # Rename to keep original name
60
+ target = path.with_name(safe)
61
+ if target != path:
62
+ path.rename(target)
63
+ path = target
64
+ self._path = path
65
+ self._name = safe
66
+ self._type = validate_file(path, max_size=max_size)
67
+ else:
68
+ raise TypeError(
69
+ f"Unsupported source type for Document: {type(source)}. "
70
+ "Expected Path, str, or bytes."
71
+ )
72
+
73
+ logger.debug("Document created: %s (type=%s)", self._name, self._type)
74
+
75
+ # ------------------------------------------------------------------
76
+ # Properties
77
+ # ------------------------------------------------------------------
78
+
79
+ @property
80
+ def path(self) -> Path:
81
+ if self._path is None:
82
+ raise InvalidDocument("Document has no underlying file path.")
83
+ return self._path
84
+
85
+ @property
86
+ def name(self) -> str:
87
+ return self._name or (self._path.name if self._path else "unnamed")
88
+
89
+ @property
90
+ def size(self) -> int:
91
+ return self.path.stat().st_size
92
+
93
+ @property
94
+ def type(self) -> str:
95
+ return self._type or "unknown"
96
+
97
+ @property
98
+ def extension(self) -> str:
99
+ return self.path.suffix.lower()
100
+
101
+ @property
102
+ def mime_type(self) -> str:
103
+ import mimetypes
104
+
105
+ mime, _ = mimetypes.guess_type(self.name)
106
+ return mime or "application/octet-stream"
107
+
108
+ # ------------------------------------------------------------------
109
+ # Class constructors for in-memory content
110
+ # ------------------------------------------------------------------
111
+
112
+ @classmethod
113
+ def html(cls, content: str, *, filename: str = "document.html") -> "Document":
114
+ """Create a Document from an HTML string."""
115
+ data = content.encode("utf-8")
116
+ return cls(data, filename=filename, doc_type="html")
117
+
118
+ @classmethod
119
+ def markdown(cls, content: str, *, filename: str = "document.md") -> "Document":
120
+ """Create a Document from a Markdown string."""
121
+ data = content.encode("utf-8")
122
+ return cls(data, filename=filename, doc_type="markdown")
123
+
124
+ @classmethod
125
+ def text(cls, content: str, *, filename: str = "document.txt") -> "Document":
126
+ """Create a Document from plain text."""
127
+ data = content.encode("utf-8")
128
+ return cls(data, filename=filename, doc_type="txt")
129
+
130
+ # ------------------------------------------------------------------
131
+ # Processor access
132
+ # ------------------------------------------------------------------
133
+
134
+ def _processor(self):
135
+ cls = get_processor_class(self.type)
136
+ return cls(self.path, temp_storage=self._temp)
137
+
138
+ # ------------------------------------------------------------------
139
+ # Conversion methods
140
+ # ------------------------------------------------------------------
141
+
142
+ def to_pdf(self) -> "Document":
143
+ """Convert this document to PDF and return a new Document."""
144
+ logger.info("Converting %s -> PDF", self.name)
145
+ try:
146
+ out_path = self._processor().to_pdf()
147
+ return Document(out_path, temp_storage=self._temp)
148
+ except Exception as exc:
149
+ raise ConversionError(
150
+ str(exc), source=self.type, target="pdf"
151
+ ) from exc
152
+
153
+ def to_text(self) -> str:
154
+ """Extract plain text content."""
155
+ logger.info("Extracting text from %s", self.name)
156
+ try:
157
+ return self._processor().to_text()
158
+ except NotImplementedError:
159
+ raise ConversionError(
160
+ f"Text extraction not supported for type '{self.type}'.",
161
+ source=self.type,
162
+ target="txt",
163
+ )
164
+ except Exception as exc:
165
+ raise ConversionError(str(exc), source=self.type, target="txt") from exc
166
+
167
+ def extract_text(self) -> str:
168
+ """Alias for to_text()."""
169
+ return self.to_text()
170
+
171
+ def to_html(self) -> str:
172
+ """Convert to HTML string."""
173
+ logger.info("Converting %s -> HTML", self.name)
174
+ try:
175
+ return self._processor().to_html()
176
+ except NotImplementedError:
177
+ raise ConversionError(
178
+ f"HTML conversion not supported for type '{self.type}'.",
179
+ source=self.type,
180
+ target="html",
181
+ )
182
+ except Exception as exc:
183
+ raise ConversionError(str(exc), source=self.type, target="html") from exc
184
+
185
+ def to_images(self, *, format: str = "png") -> list["Document"]:
186
+ """Convert pages to images. Returns a list of Document objects."""
187
+ logger.info("Converting %s -> images (%s)", self.name, format)
188
+ try:
189
+ paths = self._processor().to_images(format=format)
190
+ return [Document(p, temp_storage=self._temp) for p in paths]
191
+ except NotImplementedError:
192
+ raise ConversionError(
193
+ f"Image conversion not supported for type '{self.type}'.",
194
+ source=self.type,
195
+ target="images",
196
+ )
197
+ except Exception as exc:
198
+ raise ConversionError(str(exc), source=self.type, target="images") from exc
199
+
200
+ # ------------------------------------------------------------------
201
+ # PDF operations (delegated when type is PDF)
202
+ # ------------------------------------------------------------------
203
+
204
+ def info(self) -> dict[str, Any]:
205
+ """Return document metadata / information."""
206
+ try:
207
+ return self._processor().info()
208
+ except Exception:
209
+ # Fallback basic info
210
+ return {
211
+ "filename": self.name,
212
+ "type": self.type,
213
+ "size": self.size,
214
+ "extension": self.extension,
215
+ "mime_type": self.mime_type,
216
+ }
217
+
218
+ def compress(self, level: str = "medium") -> "Document":
219
+ """Compress a PDF. Returns a new Document."""
220
+ if self.type != "pdf":
221
+ raise ProcessingError("compress() is only available for PDF documents.")
222
+ from edocapi.processors.pdf import PDFProcessor
223
+
224
+ proc = PDFProcessor(self.path, temp_storage=self._temp)
225
+ out = proc.compress(level=level)
226
+ return Document(out, temp_storage=self._temp)
227
+
228
+ def split(self) -> list["Document"]:
229
+ """Split a PDF into individual pages."""
230
+ if self.type != "pdf":
231
+ raise ProcessingError("split() is only available for PDF documents.")
232
+ from edocapi.processors.pdf import PDFProcessor
233
+
234
+ proc = PDFProcessor(self.path, temp_storage=self._temp)
235
+ paths = proc.split()
236
+ return [Document(p, temp_storage=self._temp) for p in paths]
237
+
238
+ def extract_pages(self, start: int, end: int | None = None) -> "Document":
239
+ """Extract a page range (1-based inclusive)."""
240
+ if self.type != "pdf":
241
+ raise ProcessingError("extract_pages() is only available for PDF documents.")
242
+ from edocapi.processors.pdf import PDFProcessor
243
+
244
+ proc = PDFProcessor(self.path, temp_storage=self._temp)
245
+ out = proc.extract_pages(start, end)
246
+ return Document(out, temp_storage=self._temp)
247
+
248
+ def rotate(self, degrees: int = 90, pages: Sequence[int] | None = None) -> "Document":
249
+ """Rotate pages (1-based). Default rotates all pages by 90 degrees."""
250
+ if self.type != "pdf":
251
+ raise ProcessingError("rotate() is only available for PDF documents.")
252
+ from edocapi.processors.pdf import PDFProcessor
253
+
254
+ proc = PDFProcessor(self.path, temp_storage=self._temp)
255
+ out = proc.rotate(degrees, pages=pages)
256
+ return Document(out, temp_storage=self._temp)
257
+
258
+ # ------------------------------------------------------------------
259
+ # Class-level operations
260
+ # ------------------------------------------------------------------
261
+
262
+ @classmethod
263
+ def merge(cls, files: Sequence[Path | str | "Document"]) -> "Document":
264
+ """Merge multiple PDF documents into one."""
265
+ from edocapi.processors.pdf import PDFProcessor
266
+
267
+ paths: list[Path] = []
268
+ for f in files:
269
+ if isinstance(f, Document):
270
+ paths.append(f.path)
271
+ else:
272
+ paths.append(Path(f))
273
+
274
+ if not paths:
275
+ raise ProcessingError("No files provided for merge.")
276
+
277
+ temp = TemporaryStorage()
278
+ out = PDFProcessor.merge(paths, temp_storage=temp)
279
+ return cls(out, temp_storage=temp)
280
+
281
+ # ------------------------------------------------------------------
282
+ # Utility
283
+ # ------------------------------------------------------------------
284
+
285
+ def __repr__(self) -> str:
286
+ return f"<Document name={self.name!r} type={self.type!r} size={self.size}>"
287
+
288
+ @staticmethod
289
+ def supported_types() -> list[str]:
290
+ return list_supported_types()
edocapi/exceptions.py ADDED
@@ -0,0 +1,104 @@
1
+ """eDocAPI exception hierarchy."""
2
+
3
+ from __future__ import annotations
4
+
5
+
6
+ class EdocAPIError(Exception):
7
+ """Base exception for all eDocAPI errors."""
8
+
9
+ def __init__(self, message: str = "An error occurred in eDocAPI.") -> None:
10
+ self.message = message
11
+ super().__init__(self.message)
12
+
13
+ def to_dict(self) -> dict:
14
+ return {
15
+ "error": self.__class__.__name__,
16
+ "message": self.message,
17
+ }
18
+
19
+
20
+ class UnsupportedFileType(EdocAPIError):
21
+ """Raised when the uploaded or provided file type is not supported."""
22
+
23
+ def __init__(
24
+ self,
25
+ message: str = "The uploaded file type is not supported.",
26
+ *,
27
+ mime_type: str | None = None,
28
+ extension: str | None = None,
29
+ ) -> None:
30
+ if mime_type or extension:
31
+ details = []
32
+ if mime_type:
33
+ details.append(f"MIME type: {mime_type}")
34
+ if extension:
35
+ details.append(f"extension: {extension}")
36
+ message = f"{message} ({', '.join(details)})"
37
+ super().__init__(message)
38
+ self.mime_type = mime_type
39
+ self.extension = extension
40
+
41
+
42
+ class InvalidDocument(EdocAPIError):
43
+ """Raised when a file cannot be parsed as a valid document."""
44
+
45
+ def __init__(self, message: str = "The file is not a valid document.") -> None:
46
+ super().__init__(message)
47
+
48
+
49
+ class FileTooLarge(EdocAPIError):
50
+ """Raised when an uploaded file exceeds the configured size limit."""
51
+
52
+ def __init__(
53
+ self,
54
+ message: str = "The uploaded file exceeds the maximum allowed size.",
55
+ *,
56
+ size: int | None = None,
57
+ max_size: int | None = None,
58
+ ) -> None:
59
+ if size is not None and max_size is not None:
60
+ message = (
61
+ f"File size {size} bytes exceeds maximum allowed size "
62
+ f"of {max_size} bytes."
63
+ )
64
+ super().__init__(message)
65
+ self.size = size
66
+ self.max_size = max_size
67
+
68
+
69
+ class FileNotFound(EdocAPIError):
70
+ """Raised when a required file cannot be found."""
71
+
72
+ def __init__(self, message: str = "The requested file was not found.") -> None:
73
+ super().__init__(message)
74
+
75
+
76
+ class ConversionError(EdocAPIError):
77
+ """Raised when a document conversion fails."""
78
+
79
+ def __init__(
80
+ self,
81
+ message: str = "Document conversion failed.",
82
+ *,
83
+ source: str | None = None,
84
+ target: str | None = None,
85
+ ) -> None:
86
+ if source and target:
87
+ message = f"Failed to convert from {source} to {target}: {message}"
88
+ super().__init__(message)
89
+ self.source = source
90
+ self.target = target
91
+
92
+
93
+ class ProcessingError(EdocAPIError):
94
+ """Raised when a document processing operation fails."""
95
+
96
+ def __init__(self, message: str = "Document processing failed.") -> None:
97
+ super().__init__(message)
98
+
99
+
100
+ class ValidationError(EdocAPIError):
101
+ """Raised when file or input validation fails."""
102
+
103
+ def __init__(self, message: str = "Validation failed.") -> None:
104
+ super().__init__(message)
edocapi/files.py ADDED
@@ -0,0 +1,97 @@
1
+ """File upload helpers for eDocAPI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from pathlib import Path
7
+ import uuid
8
+ from typing import Any
9
+
10
+ from starlette.datastructures import UploadFile
11
+ from starlette.requests import Request
12
+
13
+ from edocapi.exceptions import FileTooLarge, ValidationError
14
+ from edocapi.storage.temporary import TemporaryStorage
15
+ from edocapi.validation.files import detect_type_from_bytes, validate_extension
16
+
17
+ logger = logging.getLogger("edocapi.files")
18
+
19
+
20
+ async def extract_uploads(
21
+ request: Request,
22
+ *,
23
+ max_size: int,
24
+ max_files: int = 20,
25
+ temp_storage: TemporaryStorage | None = None,
26
+ ) -> list[Path]:
27
+ """Extract and validate uploaded files from a multipart request.
28
+
29
+ Returns a list of paths to temporary files that contain the uploaded content.
30
+ """
31
+ content_type = request.headers.get("content-type", "")
32
+ if "multipart/form-data" not in content_type:
33
+ raise ValidationError(
34
+ "Expected multipart/form-data for file uploads. "
35
+ "Did you forget to set enctype='multipart/form-data'?"
36
+ )
37
+
38
+ form = await request.form()
39
+ uploads: list[Path] = []
40
+
41
+ # Collect all UploadFile instances (single 'file' or multiple 'files')
42
+ candidates: list[UploadFile] = []
43
+ for key in form:
44
+ value = form.getlist(key)
45
+ for item in value:
46
+ if isinstance(item, UploadFile):
47
+ candidates.append(item)
48
+
49
+ if not candidates:
50
+ raise ValidationError("No file part found in the request.")
51
+
52
+ if len(candidates) > max_files:
53
+ raise ValidationError(
54
+ f"Too many files uploaded ({len(candidates)}). Maximum is {max_files}."
55
+ )
56
+
57
+ storage = temp_storage or TemporaryStorage()
58
+
59
+ for upload in candidates:
60
+ filename = upload.filename or "unnamed"
61
+ path: Path | None = None
62
+ try:
63
+ safe_name = validate_extension(filename)
64
+ detected = bytearray()
65
+ path = storage.create_file(suffix=Path(safe_name).suffix, prefix="upload_")
66
+ size = 0
67
+ with path.open("wb") as destination:
68
+ while chunk := await upload.read(min(64 * 1024, max_size - size + 1)):
69
+ size += len(chunk)
70
+ if size > max_size:
71
+ raise FileTooLarge(size=size, max_size=max_size)
72
+ destination.write(chunk)
73
+ if len(detected) < 4096:
74
+ detected.extend(chunk[:4096 - len(detected)])
75
+
76
+ kind = detect_type_from_bytes(bytes(detected))
77
+ if kind is None and Path(safe_name).suffix.lower() not in {
78
+ ".txt", ".md", ".markdown", ".html", ".htm"
79
+ }:
80
+ raise ValidationError("Unable to determine a supported file type from content.")
81
+
82
+ named_path = path.with_name(f"{uuid.uuid4().hex}_{safe_name}")
83
+ path.rename(named_path)
84
+ storage._files.discard(path)
85
+ path = storage.register(named_path)
86
+
87
+ uploads.append(path)
88
+ logger.info("Document uploaded: %s (%d bytes)", safe_name, size)
89
+ except Exception:
90
+ # A rejected or partially written upload must not remain on disk.
91
+ if path is not None:
92
+ storage.discard(path)
93
+ raise
94
+ finally:
95
+ await upload.close()
96
+
97
+ return uploads
@@ -0,0 +1,16 @@
1
+ """Document processors  import side-effects register the processors."""
2
+
3
+ from edocapi.processors.base import (
4
+ BaseProcessor,
5
+ get_processor_class,
6
+ list_supported_types,
7
+ register_processor,
8
+ )
9
+ from edocapi.processors import pdf, docx, html, markdown, image, txt # noqa: F401
10
+
11
+ __all__ = [
12
+ "BaseProcessor",
13
+ "get_processor_class",
14
+ "list_supported_types",
15
+ "register_processor",
16
+ ]
@@ -0,0 +1,101 @@
1
+ """Base processor and registry for eDocAPI document processors."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from abc import ABC, abstractmethod
7
+ from pathlib import Path
8
+ from typing import Any, Type
9
+
10
+ logger = logging.getLogger("edocapi.processors")
11
+
12
+
13
+ class BaseProcessor(ABC):
14
+ """Abstract base class for format-specific document processors."""
15
+
16
+ # Formats this processor can handle as input (e.g. {"pdf"}, {"docx", "doc"})
17
+ supported_input_types: set[str] = set()
18
+
19
+ # Formats this processor can produce as output
20
+ supported_output_types: set[str] = set()
21
+
22
+ def __init__(self, path: Path, *, temp_storage: Any = None) -> None:
23
+ self.path = Path(path)
24
+ self.temp_storage = temp_storage
25
+
26
+ @abstractmethod
27
+ def to_pdf(self) -> Path:
28
+ """Convert the document to PDF. Returns path to the PDF file."""
29
+ ...
30
+
31
+ def to_text(self) -> str:
32
+ """Extract plain text from the document."""
33
+ raise NotImplementedError(
34
+ f"{self.__class__.__name__} does not support text extraction."
35
+ )
36
+
37
+ def to_html(self) -> str:
38
+ """Convert the document to HTML string."""
39
+ raise NotImplementedError(
40
+ f"{self.__class__.__name__} does not support HTML conversion."
41
+ )
42
+
43
+ def to_images(self, *, format: str = "png") -> list[Path]:
44
+ """Convert document pages to images. Returns list of image paths."""
45
+ raise NotImplementedError(
46
+ f"{self.__class__.__name__} does not support image conversion."
47
+ )
48
+
49
+ def info(self) -> dict[str, Any]:
50
+ """Return basic information about the document."""
51
+ stat = self.path.stat()
52
+ return {
53
+ "filename": self.path.name,
54
+ "type": next(iter(self.supported_input_types), "unknown"),
55
+ "size": stat.st_size,
56
+ "extension": self.path.suffix.lower(),
57
+ }
58
+
59
+ def extract_text(self) -> str:
60
+ """Alias for to_text() for clearer naming."""
61
+ return self.to_text()
62
+
63
+
64
+ # ---------------------------------------------------------------------------
65
+ # Processor registry
66
+ # ---------------------------------------------------------------------------
67
+
68
+ _REGISTRY: dict[str, Type[BaseProcessor]] = {}
69
+
70
+
71
+ def register_processor(
72
+ *types: str,
73
+ ) -> callable:
74
+ """Decorator to register a processor for one or more input types."""
75
+
76
+ def decorator(cls: Type[BaseProcessor]) -> Type[BaseProcessor]:
77
+ for t in types:
78
+ key = t.lower().lstrip(".")
79
+ _REGISTRY[key] = cls
80
+ logger.debug("Registered processor %s for type %s", cls.__name__, key)
81
+ return cls
82
+
83
+ return decorator
84
+
85
+
86
+ def get_processor_class(doc_type: str) -> Type[BaseProcessor]:
87
+ """Return the processor class for a given document type."""
88
+ key = doc_type.lower().lstrip(".")
89
+ if key not in _REGISTRY:
90
+ from edocapi.exceptions import UnsupportedFileType
91
+
92
+ raise UnsupportedFileType(
93
+ f"No processor registered for document type '{doc_type}'.",
94
+ extension=key,
95
+ )
96
+ return _REGISTRY[key]
97
+
98
+
99
+ def list_supported_types() -> list[str]:
100
+ """Return a sorted list of currently supported input types."""
101
+ return sorted(_REGISTRY.keys())