edocapi 0.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edocapi/__init__.py +49 -0
- edocapi/app.py +255 -0
- edocapi/cli/__init__.py +3 -0
- edocapi/cli/main.py +94 -0
- edocapi/config.py +76 -0
- edocapi/document.py +290 -0
- edocapi/exceptions.py +104 -0
- edocapi/files.py +97 -0
- edocapi/processors/__init__.py +16 -0
- edocapi/processors/base.py +101 -0
- edocapi/processors/docx.py +112 -0
- edocapi/processors/html.py +64 -0
- edocapi/processors/image.py +80 -0
- edocapi/processors/markdown.py +64 -0
- edocapi/processors/pdf.py +231 -0
- edocapi/processors/txt.py +58 -0
- edocapi/responses.py +88 -0
- edocapi/storage/__init__.py +3 -0
- edocapi/storage/temporary.py +132 -0
- edocapi/validation/__init__.py +13 -0
- edocapi/validation/files.py +209 -0
- edocapi-0.0.2.dist-info/METADATA +335 -0
- edocapi-0.0.2.dist-info/RECORD +27 -0
- edocapi-0.0.2.dist-info/WHEEL +5 -0
- edocapi-0.0.2.dist-info/entry_points.txt +2 -0
- edocapi-0.0.2.dist-info/licenses/LICENSE +21 -0
- edocapi-0.0.2.dist-info/top_level.txt +1 -0
edocapi/document.py
ADDED
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
"""Document abstraction for eDocAPI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any, Sequence
|
|
8
|
+
|
|
9
|
+
from edocapi.exceptions import ConversionError, InvalidDocument, ProcessingError
|
|
10
|
+
from edocapi.processors.base import get_processor_class, list_supported_types
|
|
11
|
+
from edocapi.storage.temporary import TemporaryStorage, safe_filename
|
|
12
|
+
from edocapi.validation.files import validate_file
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger("edocapi.document")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Document:
|
|
18
|
+
"""High-level document abstraction.
|
|
19
|
+
|
|
20
|
+
Developers interact primarily with this class::
|
|
21
|
+
|
|
22
|
+
Document(file).to_pdf()
|
|
23
|
+
Document(file).extract_text()
|
|
24
|
+
Document.merge([f1, f2])
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
def __init__(
|
|
28
|
+
self,
|
|
29
|
+
source: Path | str | bytes | None = None,
|
|
30
|
+
*,
|
|
31
|
+
filename: str | None = None,
|
|
32
|
+
doc_type: str | None = None,
|
|
33
|
+
temp_storage: TemporaryStorage | None = None,
|
|
34
|
+
max_size: int = 50 * 1024 * 1024,
|
|
35
|
+
) -> None:
|
|
36
|
+
self._temp = temp_storage or TemporaryStorage()
|
|
37
|
+
self._max_size = max_size
|
|
38
|
+
self._path: Path | None = None
|
|
39
|
+
self._type: str | None = doc_type
|
|
40
|
+
self._name: str | None = filename
|
|
41
|
+
|
|
42
|
+
if source is None:
|
|
43
|
+
return # empty document (used by classmethods)
|
|
44
|
+
|
|
45
|
+
if isinstance(source, (str, Path)):
|
|
46
|
+
path = Path(source)
|
|
47
|
+
self._path = path
|
|
48
|
+
self._name = filename or path.name
|
|
49
|
+
self._type = validate_file(path, max_size=max_size)
|
|
50
|
+
elif isinstance(source, bytes):
|
|
51
|
+
if not filename:
|
|
52
|
+
raise InvalidDocument("filename is required when creating Document from bytes")
|
|
53
|
+
safe = safe_filename(filename)
|
|
54
|
+
path = self._temp.create_file(
|
|
55
|
+
suffix=Path(safe).suffix,
|
|
56
|
+
prefix="doc_",
|
|
57
|
+
content=source,
|
|
58
|
+
)
|
|
59
|
+
# Rename to keep original name
|
|
60
|
+
target = path.with_name(safe)
|
|
61
|
+
if target != path:
|
|
62
|
+
path.rename(target)
|
|
63
|
+
path = target
|
|
64
|
+
self._path = path
|
|
65
|
+
self._name = safe
|
|
66
|
+
self._type = validate_file(path, max_size=max_size)
|
|
67
|
+
else:
|
|
68
|
+
raise TypeError(
|
|
69
|
+
f"Unsupported source type for Document: {type(source)}. "
|
|
70
|
+
"Expected Path, str, or bytes."
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
logger.debug("Document created: %s (type=%s)", self._name, self._type)
|
|
74
|
+
|
|
75
|
+
# ------------------------------------------------------------------
|
|
76
|
+
# Properties
|
|
77
|
+
# ------------------------------------------------------------------
|
|
78
|
+
|
|
79
|
+
@property
|
|
80
|
+
def path(self) -> Path:
|
|
81
|
+
if self._path is None:
|
|
82
|
+
raise InvalidDocument("Document has no underlying file path.")
|
|
83
|
+
return self._path
|
|
84
|
+
|
|
85
|
+
@property
|
|
86
|
+
def name(self) -> str:
|
|
87
|
+
return self._name or (self._path.name if self._path else "unnamed")
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def size(self) -> int:
|
|
91
|
+
return self.path.stat().st_size
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def type(self) -> str:
|
|
95
|
+
return self._type or "unknown"
|
|
96
|
+
|
|
97
|
+
@property
|
|
98
|
+
def extension(self) -> str:
|
|
99
|
+
return self.path.suffix.lower()
|
|
100
|
+
|
|
101
|
+
@property
|
|
102
|
+
def mime_type(self) -> str:
|
|
103
|
+
import mimetypes
|
|
104
|
+
|
|
105
|
+
mime, _ = mimetypes.guess_type(self.name)
|
|
106
|
+
return mime or "application/octet-stream"
|
|
107
|
+
|
|
108
|
+
# ------------------------------------------------------------------
|
|
109
|
+
# Class constructors for in-memory content
|
|
110
|
+
# ------------------------------------------------------------------
|
|
111
|
+
|
|
112
|
+
@classmethod
|
|
113
|
+
def html(cls, content: str, *, filename: str = "document.html") -> "Document":
|
|
114
|
+
"""Create a Document from an HTML string."""
|
|
115
|
+
data = content.encode("utf-8")
|
|
116
|
+
return cls(data, filename=filename, doc_type="html")
|
|
117
|
+
|
|
118
|
+
@classmethod
|
|
119
|
+
def markdown(cls, content: str, *, filename: str = "document.md") -> "Document":
|
|
120
|
+
"""Create a Document from a Markdown string."""
|
|
121
|
+
data = content.encode("utf-8")
|
|
122
|
+
return cls(data, filename=filename, doc_type="markdown")
|
|
123
|
+
|
|
124
|
+
@classmethod
|
|
125
|
+
def text(cls, content: str, *, filename: str = "document.txt") -> "Document":
|
|
126
|
+
"""Create a Document from plain text."""
|
|
127
|
+
data = content.encode("utf-8")
|
|
128
|
+
return cls(data, filename=filename, doc_type="txt")
|
|
129
|
+
|
|
130
|
+
# ------------------------------------------------------------------
|
|
131
|
+
# Processor access
|
|
132
|
+
# ------------------------------------------------------------------
|
|
133
|
+
|
|
134
|
+
def _processor(self):
|
|
135
|
+
cls = get_processor_class(self.type)
|
|
136
|
+
return cls(self.path, temp_storage=self._temp)
|
|
137
|
+
|
|
138
|
+
# ------------------------------------------------------------------
|
|
139
|
+
# Conversion methods
|
|
140
|
+
# ------------------------------------------------------------------
|
|
141
|
+
|
|
142
|
+
def to_pdf(self) -> "Document":
|
|
143
|
+
"""Convert this document to PDF and return a new Document."""
|
|
144
|
+
logger.info("Converting %s -> PDF", self.name)
|
|
145
|
+
try:
|
|
146
|
+
out_path = self._processor().to_pdf()
|
|
147
|
+
return Document(out_path, temp_storage=self._temp)
|
|
148
|
+
except Exception as exc:
|
|
149
|
+
raise ConversionError(
|
|
150
|
+
str(exc), source=self.type, target="pdf"
|
|
151
|
+
) from exc
|
|
152
|
+
|
|
153
|
+
def to_text(self) -> str:
|
|
154
|
+
"""Extract plain text content."""
|
|
155
|
+
logger.info("Extracting text from %s", self.name)
|
|
156
|
+
try:
|
|
157
|
+
return self._processor().to_text()
|
|
158
|
+
except NotImplementedError:
|
|
159
|
+
raise ConversionError(
|
|
160
|
+
f"Text extraction not supported for type '{self.type}'.",
|
|
161
|
+
source=self.type,
|
|
162
|
+
target="txt",
|
|
163
|
+
)
|
|
164
|
+
except Exception as exc:
|
|
165
|
+
raise ConversionError(str(exc), source=self.type, target="txt") from exc
|
|
166
|
+
|
|
167
|
+
def extract_text(self) -> str:
|
|
168
|
+
"""Alias for to_text()."""
|
|
169
|
+
return self.to_text()
|
|
170
|
+
|
|
171
|
+
def to_html(self) -> str:
|
|
172
|
+
"""Convert to HTML string."""
|
|
173
|
+
logger.info("Converting %s -> HTML", self.name)
|
|
174
|
+
try:
|
|
175
|
+
return self._processor().to_html()
|
|
176
|
+
except NotImplementedError:
|
|
177
|
+
raise ConversionError(
|
|
178
|
+
f"HTML conversion not supported for type '{self.type}'.",
|
|
179
|
+
source=self.type,
|
|
180
|
+
target="html",
|
|
181
|
+
)
|
|
182
|
+
except Exception as exc:
|
|
183
|
+
raise ConversionError(str(exc), source=self.type, target="html") from exc
|
|
184
|
+
|
|
185
|
+
def to_images(self, *, format: str = "png") -> list["Document"]:
|
|
186
|
+
"""Convert pages to images. Returns a list of Document objects."""
|
|
187
|
+
logger.info("Converting %s -> images (%s)", self.name, format)
|
|
188
|
+
try:
|
|
189
|
+
paths = self._processor().to_images(format=format)
|
|
190
|
+
return [Document(p, temp_storage=self._temp) for p in paths]
|
|
191
|
+
except NotImplementedError:
|
|
192
|
+
raise ConversionError(
|
|
193
|
+
f"Image conversion not supported for type '{self.type}'.",
|
|
194
|
+
source=self.type,
|
|
195
|
+
target="images",
|
|
196
|
+
)
|
|
197
|
+
except Exception as exc:
|
|
198
|
+
raise ConversionError(str(exc), source=self.type, target="images") from exc
|
|
199
|
+
|
|
200
|
+
# ------------------------------------------------------------------
|
|
201
|
+
# PDF operations (delegated when type is PDF)
|
|
202
|
+
# ------------------------------------------------------------------
|
|
203
|
+
|
|
204
|
+
def info(self) -> dict[str, Any]:
|
|
205
|
+
"""Return document metadata / information."""
|
|
206
|
+
try:
|
|
207
|
+
return self._processor().info()
|
|
208
|
+
except Exception:
|
|
209
|
+
# Fallback basic info
|
|
210
|
+
return {
|
|
211
|
+
"filename": self.name,
|
|
212
|
+
"type": self.type,
|
|
213
|
+
"size": self.size,
|
|
214
|
+
"extension": self.extension,
|
|
215
|
+
"mime_type": self.mime_type,
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
def compress(self, level: str = "medium") -> "Document":
|
|
219
|
+
"""Compress a PDF. Returns a new Document."""
|
|
220
|
+
if self.type != "pdf":
|
|
221
|
+
raise ProcessingError("compress() is only available for PDF documents.")
|
|
222
|
+
from edocapi.processors.pdf import PDFProcessor
|
|
223
|
+
|
|
224
|
+
proc = PDFProcessor(self.path, temp_storage=self._temp)
|
|
225
|
+
out = proc.compress(level=level)
|
|
226
|
+
return Document(out, temp_storage=self._temp)
|
|
227
|
+
|
|
228
|
+
def split(self) -> list["Document"]:
|
|
229
|
+
"""Split a PDF into individual pages."""
|
|
230
|
+
if self.type != "pdf":
|
|
231
|
+
raise ProcessingError("split() is only available for PDF documents.")
|
|
232
|
+
from edocapi.processors.pdf import PDFProcessor
|
|
233
|
+
|
|
234
|
+
proc = PDFProcessor(self.path, temp_storage=self._temp)
|
|
235
|
+
paths = proc.split()
|
|
236
|
+
return [Document(p, temp_storage=self._temp) for p in paths]
|
|
237
|
+
|
|
238
|
+
def extract_pages(self, start: int, end: int | None = None) -> "Document":
|
|
239
|
+
"""Extract a page range (1-based inclusive)."""
|
|
240
|
+
if self.type != "pdf":
|
|
241
|
+
raise ProcessingError("extract_pages() is only available for PDF documents.")
|
|
242
|
+
from edocapi.processors.pdf import PDFProcessor
|
|
243
|
+
|
|
244
|
+
proc = PDFProcessor(self.path, temp_storage=self._temp)
|
|
245
|
+
out = proc.extract_pages(start, end)
|
|
246
|
+
return Document(out, temp_storage=self._temp)
|
|
247
|
+
|
|
248
|
+
def rotate(self, degrees: int = 90, pages: Sequence[int] | None = None) -> "Document":
|
|
249
|
+
"""Rotate pages (1-based). Default rotates all pages by 90 degrees."""
|
|
250
|
+
if self.type != "pdf":
|
|
251
|
+
raise ProcessingError("rotate() is only available for PDF documents.")
|
|
252
|
+
from edocapi.processors.pdf import PDFProcessor
|
|
253
|
+
|
|
254
|
+
proc = PDFProcessor(self.path, temp_storage=self._temp)
|
|
255
|
+
out = proc.rotate(degrees, pages=pages)
|
|
256
|
+
return Document(out, temp_storage=self._temp)
|
|
257
|
+
|
|
258
|
+
# ------------------------------------------------------------------
|
|
259
|
+
# Class-level operations
|
|
260
|
+
# ------------------------------------------------------------------
|
|
261
|
+
|
|
262
|
+
@classmethod
|
|
263
|
+
def merge(cls, files: Sequence[Path | str | "Document"]) -> "Document":
|
|
264
|
+
"""Merge multiple PDF documents into one."""
|
|
265
|
+
from edocapi.processors.pdf import PDFProcessor
|
|
266
|
+
|
|
267
|
+
paths: list[Path] = []
|
|
268
|
+
for f in files:
|
|
269
|
+
if isinstance(f, Document):
|
|
270
|
+
paths.append(f.path)
|
|
271
|
+
else:
|
|
272
|
+
paths.append(Path(f))
|
|
273
|
+
|
|
274
|
+
if not paths:
|
|
275
|
+
raise ProcessingError("No files provided for merge.")
|
|
276
|
+
|
|
277
|
+
temp = TemporaryStorage()
|
|
278
|
+
out = PDFProcessor.merge(paths, temp_storage=temp)
|
|
279
|
+
return cls(out, temp_storage=temp)
|
|
280
|
+
|
|
281
|
+
# ------------------------------------------------------------------
|
|
282
|
+
# Utility
|
|
283
|
+
# ------------------------------------------------------------------
|
|
284
|
+
|
|
285
|
+
def __repr__(self) -> str:
|
|
286
|
+
return f"<Document name={self.name!r} type={self.type!r} size={self.size}>"
|
|
287
|
+
|
|
288
|
+
@staticmethod
|
|
289
|
+
def supported_types() -> list[str]:
|
|
290
|
+
return list_supported_types()
|
edocapi/exceptions.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""eDocAPI exception hierarchy."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class EdocAPIError(Exception):
|
|
7
|
+
"""Base exception for all eDocAPI errors."""
|
|
8
|
+
|
|
9
|
+
def __init__(self, message: str = "An error occurred in eDocAPI.") -> None:
|
|
10
|
+
self.message = message
|
|
11
|
+
super().__init__(self.message)
|
|
12
|
+
|
|
13
|
+
def to_dict(self) -> dict:
|
|
14
|
+
return {
|
|
15
|
+
"error": self.__class__.__name__,
|
|
16
|
+
"message": self.message,
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class UnsupportedFileType(EdocAPIError):
|
|
21
|
+
"""Raised when the uploaded or provided file type is not supported."""
|
|
22
|
+
|
|
23
|
+
def __init__(
|
|
24
|
+
self,
|
|
25
|
+
message: str = "The uploaded file type is not supported.",
|
|
26
|
+
*,
|
|
27
|
+
mime_type: str | None = None,
|
|
28
|
+
extension: str | None = None,
|
|
29
|
+
) -> None:
|
|
30
|
+
if mime_type or extension:
|
|
31
|
+
details = []
|
|
32
|
+
if mime_type:
|
|
33
|
+
details.append(f"MIME type: {mime_type}")
|
|
34
|
+
if extension:
|
|
35
|
+
details.append(f"extension: {extension}")
|
|
36
|
+
message = f"{message} ({', '.join(details)})"
|
|
37
|
+
super().__init__(message)
|
|
38
|
+
self.mime_type = mime_type
|
|
39
|
+
self.extension = extension
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
class InvalidDocument(EdocAPIError):
|
|
43
|
+
"""Raised when a file cannot be parsed as a valid document."""
|
|
44
|
+
|
|
45
|
+
def __init__(self, message: str = "The file is not a valid document.") -> None:
|
|
46
|
+
super().__init__(message)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class FileTooLarge(EdocAPIError):
|
|
50
|
+
"""Raised when an uploaded file exceeds the configured size limit."""
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
message: str = "The uploaded file exceeds the maximum allowed size.",
|
|
55
|
+
*,
|
|
56
|
+
size: int | None = None,
|
|
57
|
+
max_size: int | None = None,
|
|
58
|
+
) -> None:
|
|
59
|
+
if size is not None and max_size is not None:
|
|
60
|
+
message = (
|
|
61
|
+
f"File size {size} bytes exceeds maximum allowed size "
|
|
62
|
+
f"of {max_size} bytes."
|
|
63
|
+
)
|
|
64
|
+
super().__init__(message)
|
|
65
|
+
self.size = size
|
|
66
|
+
self.max_size = max_size
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class FileNotFound(EdocAPIError):
|
|
70
|
+
"""Raised when a required file cannot be found."""
|
|
71
|
+
|
|
72
|
+
def __init__(self, message: str = "The requested file was not found.") -> None:
|
|
73
|
+
super().__init__(message)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class ConversionError(EdocAPIError):
|
|
77
|
+
"""Raised when a document conversion fails."""
|
|
78
|
+
|
|
79
|
+
def __init__(
|
|
80
|
+
self,
|
|
81
|
+
message: str = "Document conversion failed.",
|
|
82
|
+
*,
|
|
83
|
+
source: str | None = None,
|
|
84
|
+
target: str | None = None,
|
|
85
|
+
) -> None:
|
|
86
|
+
if source and target:
|
|
87
|
+
message = f"Failed to convert from {source} to {target}: {message}"
|
|
88
|
+
super().__init__(message)
|
|
89
|
+
self.source = source
|
|
90
|
+
self.target = target
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class ProcessingError(EdocAPIError):
|
|
94
|
+
"""Raised when a document processing operation fails."""
|
|
95
|
+
|
|
96
|
+
def __init__(self, message: str = "Document processing failed.") -> None:
|
|
97
|
+
super().__init__(message)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class ValidationError(EdocAPIError):
|
|
101
|
+
"""Raised when file or input validation fails."""
|
|
102
|
+
|
|
103
|
+
def __init__(self, message: str = "Validation failed.") -> None:
|
|
104
|
+
super().__init__(message)
|
edocapi/files.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""File upload helpers for eDocAPI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
import uuid
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from starlette.datastructures import UploadFile
|
|
11
|
+
from starlette.requests import Request
|
|
12
|
+
|
|
13
|
+
from edocapi.exceptions import FileTooLarge, ValidationError
|
|
14
|
+
from edocapi.storage.temporary import TemporaryStorage
|
|
15
|
+
from edocapi.validation.files import detect_type_from_bytes, validate_extension
|
|
16
|
+
|
|
17
|
+
logger = logging.getLogger("edocapi.files")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
async def extract_uploads(
|
|
21
|
+
request: Request,
|
|
22
|
+
*,
|
|
23
|
+
max_size: int,
|
|
24
|
+
max_files: int = 20,
|
|
25
|
+
temp_storage: TemporaryStorage | None = None,
|
|
26
|
+
) -> list[Path]:
|
|
27
|
+
"""Extract and validate uploaded files from a multipart request.
|
|
28
|
+
|
|
29
|
+
Returns a list of paths to temporary files that contain the uploaded content.
|
|
30
|
+
"""
|
|
31
|
+
content_type = request.headers.get("content-type", "")
|
|
32
|
+
if "multipart/form-data" not in content_type:
|
|
33
|
+
raise ValidationError(
|
|
34
|
+
"Expected multipart/form-data for file uploads. "
|
|
35
|
+
"Did you forget to set enctype='multipart/form-data'?"
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
form = await request.form()
|
|
39
|
+
uploads: list[Path] = []
|
|
40
|
+
|
|
41
|
+
# Collect all UploadFile instances (single 'file' or multiple 'files')
|
|
42
|
+
candidates: list[UploadFile] = []
|
|
43
|
+
for key in form:
|
|
44
|
+
value = form.getlist(key)
|
|
45
|
+
for item in value:
|
|
46
|
+
if isinstance(item, UploadFile):
|
|
47
|
+
candidates.append(item)
|
|
48
|
+
|
|
49
|
+
if not candidates:
|
|
50
|
+
raise ValidationError("No file part found in the request.")
|
|
51
|
+
|
|
52
|
+
if len(candidates) > max_files:
|
|
53
|
+
raise ValidationError(
|
|
54
|
+
f"Too many files uploaded ({len(candidates)}). Maximum is {max_files}."
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
storage = temp_storage or TemporaryStorage()
|
|
58
|
+
|
|
59
|
+
for upload in candidates:
|
|
60
|
+
filename = upload.filename or "unnamed"
|
|
61
|
+
path: Path | None = None
|
|
62
|
+
try:
|
|
63
|
+
safe_name = validate_extension(filename)
|
|
64
|
+
detected = bytearray()
|
|
65
|
+
path = storage.create_file(suffix=Path(safe_name).suffix, prefix="upload_")
|
|
66
|
+
size = 0
|
|
67
|
+
with path.open("wb") as destination:
|
|
68
|
+
while chunk := await upload.read(min(64 * 1024, max_size - size + 1)):
|
|
69
|
+
size += len(chunk)
|
|
70
|
+
if size > max_size:
|
|
71
|
+
raise FileTooLarge(size=size, max_size=max_size)
|
|
72
|
+
destination.write(chunk)
|
|
73
|
+
if len(detected) < 4096:
|
|
74
|
+
detected.extend(chunk[:4096 - len(detected)])
|
|
75
|
+
|
|
76
|
+
kind = detect_type_from_bytes(bytes(detected))
|
|
77
|
+
if kind is None and Path(safe_name).suffix.lower() not in {
|
|
78
|
+
".txt", ".md", ".markdown", ".html", ".htm"
|
|
79
|
+
}:
|
|
80
|
+
raise ValidationError("Unable to determine a supported file type from content.")
|
|
81
|
+
|
|
82
|
+
named_path = path.with_name(f"{uuid.uuid4().hex}_{safe_name}")
|
|
83
|
+
path.rename(named_path)
|
|
84
|
+
storage._files.discard(path)
|
|
85
|
+
path = storage.register(named_path)
|
|
86
|
+
|
|
87
|
+
uploads.append(path)
|
|
88
|
+
logger.info("Document uploaded: %s (%d bytes)", safe_name, size)
|
|
89
|
+
except Exception:
|
|
90
|
+
# A rejected or partially written upload must not remain on disk.
|
|
91
|
+
if path is not None:
|
|
92
|
+
storage.discard(path)
|
|
93
|
+
raise
|
|
94
|
+
finally:
|
|
95
|
+
await upload.close()
|
|
96
|
+
|
|
97
|
+
return uploads
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Document processors import side-effects register the processors."""
|
|
2
|
+
|
|
3
|
+
from edocapi.processors.base import (
|
|
4
|
+
BaseProcessor,
|
|
5
|
+
get_processor_class,
|
|
6
|
+
list_supported_types,
|
|
7
|
+
register_processor,
|
|
8
|
+
)
|
|
9
|
+
from edocapi.processors import pdf, docx, html, markdown, image, txt # noqa: F401
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"BaseProcessor",
|
|
13
|
+
"get_processor_class",
|
|
14
|
+
"list_supported_types",
|
|
15
|
+
"register_processor",
|
|
16
|
+
]
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Base processor and registry for eDocAPI document processors."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from abc import ABC, abstractmethod
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any, Type
|
|
9
|
+
|
|
10
|
+
logger = logging.getLogger("edocapi.processors")
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class BaseProcessor(ABC):
|
|
14
|
+
"""Abstract base class for format-specific document processors."""
|
|
15
|
+
|
|
16
|
+
# Formats this processor can handle as input (e.g. {"pdf"}, {"docx", "doc"})
|
|
17
|
+
supported_input_types: set[str] = set()
|
|
18
|
+
|
|
19
|
+
# Formats this processor can produce as output
|
|
20
|
+
supported_output_types: set[str] = set()
|
|
21
|
+
|
|
22
|
+
def __init__(self, path: Path, *, temp_storage: Any = None) -> None:
|
|
23
|
+
self.path = Path(path)
|
|
24
|
+
self.temp_storage = temp_storage
|
|
25
|
+
|
|
26
|
+
@abstractmethod
|
|
27
|
+
def to_pdf(self) -> Path:
|
|
28
|
+
"""Convert the document to PDF. Returns path to the PDF file."""
|
|
29
|
+
...
|
|
30
|
+
|
|
31
|
+
def to_text(self) -> str:
|
|
32
|
+
"""Extract plain text from the document."""
|
|
33
|
+
raise NotImplementedError(
|
|
34
|
+
f"{self.__class__.__name__} does not support text extraction."
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
def to_html(self) -> str:
|
|
38
|
+
"""Convert the document to HTML string."""
|
|
39
|
+
raise NotImplementedError(
|
|
40
|
+
f"{self.__class__.__name__} does not support HTML conversion."
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
def to_images(self, *, format: str = "png") -> list[Path]:
|
|
44
|
+
"""Convert document pages to images. Returns list of image paths."""
|
|
45
|
+
raise NotImplementedError(
|
|
46
|
+
f"{self.__class__.__name__} does not support image conversion."
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
def info(self) -> dict[str, Any]:
|
|
50
|
+
"""Return basic information about the document."""
|
|
51
|
+
stat = self.path.stat()
|
|
52
|
+
return {
|
|
53
|
+
"filename": self.path.name,
|
|
54
|
+
"type": next(iter(self.supported_input_types), "unknown"),
|
|
55
|
+
"size": stat.st_size,
|
|
56
|
+
"extension": self.path.suffix.lower(),
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
def extract_text(self) -> str:
|
|
60
|
+
"""Alias for to_text() for clearer naming."""
|
|
61
|
+
return self.to_text()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
# ---------------------------------------------------------------------------
|
|
65
|
+
# Processor registry
|
|
66
|
+
# ---------------------------------------------------------------------------
|
|
67
|
+
|
|
68
|
+
_REGISTRY: dict[str, Type[BaseProcessor]] = {}
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def register_processor(
|
|
72
|
+
*types: str,
|
|
73
|
+
) -> callable:
|
|
74
|
+
"""Decorator to register a processor for one or more input types."""
|
|
75
|
+
|
|
76
|
+
def decorator(cls: Type[BaseProcessor]) -> Type[BaseProcessor]:
|
|
77
|
+
for t in types:
|
|
78
|
+
key = t.lower().lstrip(".")
|
|
79
|
+
_REGISTRY[key] = cls
|
|
80
|
+
logger.debug("Registered processor %s for type %s", cls.__name__, key)
|
|
81
|
+
return cls
|
|
82
|
+
|
|
83
|
+
return decorator
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def get_processor_class(doc_type: str) -> Type[BaseProcessor]:
|
|
87
|
+
"""Return the processor class for a given document type."""
|
|
88
|
+
key = doc_type.lower().lstrip(".")
|
|
89
|
+
if key not in _REGISTRY:
|
|
90
|
+
from edocapi.exceptions import UnsupportedFileType
|
|
91
|
+
|
|
92
|
+
raise UnsupportedFileType(
|
|
93
|
+
f"No processor registered for document type '{doc_type}'.",
|
|
94
|
+
extension=key,
|
|
95
|
+
)
|
|
96
|
+
return _REGISTRY[key]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def list_supported_types() -> list[str]:
|
|
100
|
+
"""Return a sorted list of currently supported input types."""
|
|
101
|
+
return sorted(_REGISTRY.keys())
|