edocapi 0.0.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edocapi/__init__.py +49 -0
- edocapi/app.py +255 -0
- edocapi/cli/__init__.py +3 -0
- edocapi/cli/main.py +94 -0
- edocapi/config.py +76 -0
- edocapi/document.py +290 -0
- edocapi/exceptions.py +104 -0
- edocapi/files.py +97 -0
- edocapi/processors/__init__.py +16 -0
- edocapi/processors/base.py +101 -0
- edocapi/processors/docx.py +112 -0
- edocapi/processors/html.py +64 -0
- edocapi/processors/image.py +80 -0
- edocapi/processors/markdown.py +64 -0
- edocapi/processors/pdf.py +231 -0
- edocapi/processors/txt.py +58 -0
- edocapi/responses.py +88 -0
- edocapi/storage/__init__.py +3 -0
- edocapi/storage/temporary.py +132 -0
- edocapi/validation/__init__.py +13 -0
- edocapi/validation/files.py +209 -0
- edocapi-0.0.2.dist-info/METADATA +335 -0
- edocapi-0.0.2.dist-info/RECORD +27 -0
- edocapi-0.0.2.dist-info/WHEEL +5 -0
- edocapi-0.0.2.dist-info/entry_points.txt +2 -0
- edocapi-0.0.2.dist-info/licenses/LICENSE +21 -0
- edocapi-0.0.2.dist-info/top_level.txt +1 -0
edocapi/responses.py
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Response helpers for eDocAPI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import mimetypes
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Any, Mapping
|
|
8
|
+
|
|
9
|
+
from starlette.responses import FileResponse as StarletteFileResponse
|
|
10
|
+
from starlette.responses import JSONResponse as StarletteJSONResponse
|
|
11
|
+
from starlette.responses import Response
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class JSONResponse(StarletteJSONResponse):
|
|
15
|
+
"""JSON response that serializes normal Python dicts / lists."""
|
|
16
|
+
|
|
17
|
+
def __init__(
|
|
18
|
+
self,
|
|
19
|
+
content: Any = None,
|
|
20
|
+
status_code: int = 200,
|
|
21
|
+
headers: Mapping[str, str] | None = None,
|
|
22
|
+
**kwargs: Any,
|
|
23
|
+
) -> None:
|
|
24
|
+
super().__init__(
|
|
25
|
+
content=content if content is not None else {},
|
|
26
|
+
status_code=status_code,
|
|
27
|
+
headers=headers,
|
|
28
|
+
**kwargs,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class FileResponse(StarletteFileResponse):
|
|
33
|
+
"""File response with sensible defaults for document downloads.
|
|
34
|
+
|
|
35
|
+
Automatically sets Content-Type and Content-Disposition when possible.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
def __init__(
|
|
39
|
+
self,
|
|
40
|
+
path: str | Path,
|
|
41
|
+
filename: str | None = None,
|
|
42
|
+
media_type: str | None = None,
|
|
43
|
+
status_code: int = 200,
|
|
44
|
+
headers: Mapping[str, str] | None = None,
|
|
45
|
+
background: Any = None,
|
|
46
|
+
**kwargs: Any,
|
|
47
|
+
) -> None:
|
|
48
|
+
path = Path(path)
|
|
49
|
+
if filename is None:
|
|
50
|
+
filename = path.name
|
|
51
|
+
|
|
52
|
+
if media_type is None:
|
|
53
|
+
guessed, _ = mimetypes.guess_type(filename)
|
|
54
|
+
media_type = guessed or "application/octet-stream"
|
|
55
|
+
|
|
56
|
+
super().__init__(
|
|
57
|
+
path=str(path),
|
|
58
|
+
filename=filename,
|
|
59
|
+
media_type=media_type,
|
|
60
|
+
status_code=status_code,
|
|
61
|
+
headers=headers,
|
|
62
|
+
background=background,
|
|
63
|
+
**kwargs,
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def make_error_response(
|
|
68
|
+
error: Exception,
|
|
69
|
+
*,
|
|
70
|
+
status_code: int = 400,
|
|
71
|
+
debug: bool = False,
|
|
72
|
+
) -> JSONResponse:
|
|
73
|
+
"""Create a consistent JSON error response from an exception."""
|
|
74
|
+
from edocapi.exceptions import EdocAPIError
|
|
75
|
+
|
|
76
|
+
if isinstance(error, EdocAPIError):
|
|
77
|
+
body = error.to_dict()
|
|
78
|
+
if debug:
|
|
79
|
+
body["detail"] = str(error)
|
|
80
|
+
else:
|
|
81
|
+
body = {
|
|
82
|
+
"error": type(error).__name__,
|
|
83
|
+
"message": str(error) if debug else "An unexpected error occurred.",
|
|
84
|
+
}
|
|
85
|
+
if debug:
|
|
86
|
+
body["detail"] = str(error)
|
|
87
|
+
|
|
88
|
+
return JSONResponse(content=body, status_code=status_code)
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""Temporary file management for eDocAPI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import atexit
|
|
6
|
+
import logging
|
|
7
|
+
import os
|
|
8
|
+
import shutil
|
|
9
|
+
import tempfile
|
|
10
|
+
import uuid
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Iterator
|
|
13
|
+
|
|
14
|
+
logger = logging.getLogger("edocapi.storage")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class TemporaryStorage:
|
|
18
|
+
"""Manages a private temporary directory for document processing.
|
|
19
|
+
|
|
20
|
+
Files created here are cleaned up when the storage instance is closed
|
|
21
|
+
or when the process exits (via atexit).
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
def __init__(self, base_dir: Path | str | None = None) -> None:
|
|
25
|
+
if base_dir is not None:
|
|
26
|
+
self._root = Path(base_dir)
|
|
27
|
+
self._root.mkdir(parents=True, exist_ok=True)
|
|
28
|
+
self._owned = False
|
|
29
|
+
else:
|
|
30
|
+
self._root = Path(tempfile.mkdtemp(prefix="edocapi_"))
|
|
31
|
+
self._owned = True
|
|
32
|
+
|
|
33
|
+
self._files: set[Path] = set()
|
|
34
|
+
self._cleanup_callback = self.cleanup
|
|
35
|
+
atexit.register(self._cleanup_callback)
|
|
36
|
+
logger.debug("Temporary storage created at %s", self._root)
|
|
37
|
+
|
|
38
|
+
@property
|
|
39
|
+
def root(self) -> Path:
|
|
40
|
+
return self._root
|
|
41
|
+
|
|
42
|
+
def create_file(
|
|
43
|
+
self,
|
|
44
|
+
suffix: str = "",
|
|
45
|
+
prefix: str = "tmp_",
|
|
46
|
+
content: bytes | None = None,
|
|
47
|
+
) -> Path:
|
|
48
|
+
"""Create a new temporary file path (and optionally write content)."""
|
|
49
|
+
name = f"{prefix}{uuid.uuid4().hex}{suffix}"
|
|
50
|
+
path = self._root / name
|
|
51
|
+
if content is not None:
|
|
52
|
+
path.write_bytes(content)
|
|
53
|
+
else:
|
|
54
|
+
path.touch()
|
|
55
|
+
self._files.add(path)
|
|
56
|
+
return path
|
|
57
|
+
|
|
58
|
+
def create_dir(self, prefix: str = "dir_") -> Path:
|
|
59
|
+
"""Create a subdirectory inside the temporary root."""
|
|
60
|
+
name = f"{prefix}{uuid.uuid4().hex}"
|
|
61
|
+
path = self._root / name
|
|
62
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
63
|
+
self._files.add(path)
|
|
64
|
+
return path
|
|
65
|
+
|
|
66
|
+
def register(self, path: Path | str) -> Path:
|
|
67
|
+
"""Register an existing path for later cleanup."""
|
|
68
|
+
path = Path(path)
|
|
69
|
+
self._files.add(path)
|
|
70
|
+
return path
|
|
71
|
+
|
|
72
|
+
def discard(self, path: Path | str) -> None:
|
|
73
|
+
"""Remove a file from managed storage after its response is sent."""
|
|
74
|
+
path = Path(path)
|
|
75
|
+
self._files.discard(path)
|
|
76
|
+
try:
|
|
77
|
+
path.unlink(missing_ok=True)
|
|
78
|
+
except OSError as exc:
|
|
79
|
+
logger.warning("Failed to remove temporary file %s: %s", path, exc)
|
|
80
|
+
|
|
81
|
+
def cleanup(self) -> None:
|
|
82
|
+
"""Remove all tracked files and, if owned, the root directory."""
|
|
83
|
+
for path in list(self._files):
|
|
84
|
+
try:
|
|
85
|
+
if path.is_file():
|
|
86
|
+
path.unlink(missing_ok=True)
|
|
87
|
+
elif path.is_dir():
|
|
88
|
+
shutil.rmtree(path, ignore_errors=True)
|
|
89
|
+
except OSError as exc:
|
|
90
|
+
logger.warning("Failed to clean up %s: %s", path, exc)
|
|
91
|
+
self._files.clear()
|
|
92
|
+
|
|
93
|
+
if self._owned and self._root.exists():
|
|
94
|
+
try:
|
|
95
|
+
shutil.rmtree(self._root, ignore_errors=True)
|
|
96
|
+
logger.debug("Removed temporary root %s", self._root)
|
|
97
|
+
except OSError as exc:
|
|
98
|
+
logger.warning("Failed to remove temp root %s: %s", self._root, exc)
|
|
99
|
+
|
|
100
|
+
# The instance may be explicitly cleaned up before interpreter exit.
|
|
101
|
+
try:
|
|
102
|
+
atexit.unregister(self._cleanup_callback)
|
|
103
|
+
except Exception:
|
|
104
|
+
pass
|
|
105
|
+
|
|
106
|
+
def __enter__(self) -> "TemporaryStorage":
|
|
107
|
+
return self
|
|
108
|
+
|
|
109
|
+
def __exit__(self, *args: object) -> None:
|
|
110
|
+
self.cleanup()
|
|
111
|
+
|
|
112
|
+
def __del__(self) -> None:
|
|
113
|
+
try:
|
|
114
|
+
self.cleanup()
|
|
115
|
+
except Exception:
|
|
116
|
+
pass
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def safe_filename(name: str) -> str:
|
|
120
|
+
"""Sanitize a filename to prevent path traversal and unsafe characters."""
|
|
121
|
+
# Take only the basename
|
|
122
|
+
name = os.path.basename(name)
|
|
123
|
+
# Remove null bytes and control characters
|
|
124
|
+
name = "".join(c for c in name if c.isprintable() and c not in r'<>:"/\|?*')
|
|
125
|
+
name = name.strip().strip(".")
|
|
126
|
+
if not name:
|
|
127
|
+
name = "unnamed"
|
|
128
|
+
# Limit length
|
|
129
|
+
if len(name) > 200:
|
|
130
|
+
stem, ext = os.path.splitext(name)
|
|
131
|
+
name = stem[:190] + ext
|
|
132
|
+
return name
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""File validation utilities for eDocAPI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import logging
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import BinaryIO
|
|
8
|
+
|
|
9
|
+
from edocapi.exceptions import (
|
|
10
|
+
FileTooLarge,
|
|
11
|
+
InvalidDocument,
|
|
12
|
+
UnsupportedFileType,
|
|
13
|
+
ValidationError,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger("edocapi.validation")
|
|
17
|
+
|
|
18
|
+
# Supported formats for v0.0.2
|
|
19
|
+
SUPPORTED_EXTENSIONS = {
|
|
20
|
+
".pdf",
|
|
21
|
+
".docx",
|
|
22
|
+
".html",
|
|
23
|
+
".htm",
|
|
24
|
+
".md",
|
|
25
|
+
".markdown",
|
|
26
|
+
".jpg",
|
|
27
|
+
".jpeg",
|
|
28
|
+
".png",
|
|
29
|
+
".webp",
|
|
30
|
+
".txt",
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
SUPPORTED_MIME_TYPES = {
|
|
34
|
+
"application/pdf",
|
|
35
|
+
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
36
|
+
"text/html",
|
|
37
|
+
"text/markdown",
|
|
38
|
+
"text/x-markdown",
|
|
39
|
+
"image/jpeg",
|
|
40
|
+
"image/png",
|
|
41
|
+
"image/webp",
|
|
42
|
+
"text/plain",
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
# Magic byte signatures (first few bytes)
|
|
46
|
+
MAGIC_SIGNATURES: dict[bytes, str] = {
|
|
47
|
+
b"%PDF": "pdf",
|
|
48
|
+
b"PK\x03\x04": "zip", # DOCX is a ZIP; further checks needed
|
|
49
|
+
b"\xff\xd8\xff": "jpeg",
|
|
50
|
+
b"\x89PNG\r\n\x1a\n": "png",
|
|
51
|
+
b"RIFF": "webp", # needs further check for WEBP
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def detect_type_from_bytes(data: bytes) -> str | None:
|
|
56
|
+
"""Attempt to detect document type from magic bytes."""
|
|
57
|
+
if not data:
|
|
58
|
+
return None
|
|
59
|
+
|
|
60
|
+
for sig, kind in MAGIC_SIGNATURES.items():
|
|
61
|
+
if data.startswith(sig):
|
|
62
|
+
if kind == "zip":
|
|
63
|
+
# Could be DOCX or other Office format
|
|
64
|
+
if b"word/" in data[:4096] or b"[Content_Types].xml" in data[:4096]:
|
|
65
|
+
return "docx"
|
|
66
|
+
return "zip"
|
|
67
|
+
if kind == "webp":
|
|
68
|
+
if b"WEBP" in data[8:16]:
|
|
69
|
+
return "webp"
|
|
70
|
+
return None
|
|
71
|
+
return kind
|
|
72
|
+
|
|
73
|
+
# HTML / Markdown / TXT heuristics
|
|
74
|
+
sample = data[:1024].decode("utf-8", errors="ignore").strip().lower()
|
|
75
|
+
if sample.startswith("<!doctype html") or sample.startswith("<html"):
|
|
76
|
+
return "html"
|
|
77
|
+
if sample.startswith("# ") or sample.startswith("---\n"):
|
|
78
|
+
return "markdown"
|
|
79
|
+
# Plain text fallback only if mostly printable
|
|
80
|
+
if all(32 <= b < 127 or b in (9, 10, 13) for b in data[:512]):
|
|
81
|
+
return "txt"
|
|
82
|
+
return None
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def validate_extension(filename: str) -> str:
|
|
86
|
+
"""Validate and return the normalized extension (including dot)."""
|
|
87
|
+
from edocapi.storage.temporary import safe_filename
|
|
88
|
+
|
|
89
|
+
name = safe_filename(filename)
|
|
90
|
+
ext = Path(name).suffix.lower()
|
|
91
|
+
if ext not in SUPPORTED_EXTENSIONS:
|
|
92
|
+
raise UnsupportedFileType(
|
|
93
|
+
f"File extension '{ext}' is not supported.",
|
|
94
|
+
extension=ext,
|
|
95
|
+
)
|
|
96
|
+
return ext
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def validate_size(size: int, max_size: int) -> None:
|
|
100
|
+
"""Raise FileTooLarge if size exceeds the limit."""
|
|
101
|
+
if size > max_size:
|
|
102
|
+
raise FileTooLarge(size=size, max_size=max_size)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def validate_file(
|
|
106
|
+
path: Path,
|
|
107
|
+
*,
|
|
108
|
+
max_size: int,
|
|
109
|
+
expected_type: str | None = None,
|
|
110
|
+
) -> str:
|
|
111
|
+
"""Perform basic validation on a file path.
|
|
112
|
+
|
|
113
|
+
Returns the detected document type string (e.g. 'pdf', 'docx').
|
|
114
|
+
"""
|
|
115
|
+
if not path.exists():
|
|
116
|
+
raise ValidationError(f"File does not exist: {path}")
|
|
117
|
+
if not path.is_file():
|
|
118
|
+
raise ValidationError(f"Path is not a regular file: {path}")
|
|
119
|
+
|
|
120
|
+
size = path.stat().st_size
|
|
121
|
+
validate_size(size, max_size)
|
|
122
|
+
|
|
123
|
+
# Read a small header for magic detection
|
|
124
|
+
with path.open("rb") as f:
|
|
125
|
+
header = f.read(4096)
|
|
126
|
+
|
|
127
|
+
detected = detect_type_from_bytes(header)
|
|
128
|
+
ext = path.suffix.lower()
|
|
129
|
+
|
|
130
|
+
# Map extension to type
|
|
131
|
+
ext_to_type = {
|
|
132
|
+
".pdf": "pdf",
|
|
133
|
+
".docx": "docx",
|
|
134
|
+
".html": "html",
|
|
135
|
+
".htm": "html",
|
|
136
|
+
".md": "markdown",
|
|
137
|
+
".markdown": "markdown",
|
|
138
|
+
".jpg": "jpeg",
|
|
139
|
+
".jpeg": "jpeg",
|
|
140
|
+
".png": "png",
|
|
141
|
+
".webp": "webp",
|
|
142
|
+
".txt": "txt",
|
|
143
|
+
}
|
|
144
|
+
ext_type = ext_to_type.get(ext)
|
|
145
|
+
|
|
146
|
+
if detected is None and ext_type is None:
|
|
147
|
+
raise UnsupportedFileType(
|
|
148
|
+
"Unable to determine file type.",
|
|
149
|
+
extension=ext,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
# Prefer detected type; fall back to extension
|
|
153
|
+
doc_type = detected or ext_type
|
|
154
|
+
|
|
155
|
+
# Basic consistency check
|
|
156
|
+
if detected and ext_type and detected != ext_type:
|
|
157
|
+
# Allow some flexibility (e.g. .txt that looks like markdown)
|
|
158
|
+
if not (detected in ("txt", "markdown") and ext_type in ("txt", "markdown")):
|
|
159
|
+
logger.warning(
|
|
160
|
+
"Magic type %s does not match extension type %s for %s",
|
|
161
|
+
detected,
|
|
162
|
+
ext_type,
|
|
163
|
+
path.name,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
if expected_type and doc_type != expected_type:
|
|
167
|
+
raise InvalidDocument(
|
|
168
|
+
f"Expected document type '{expected_type}', got '{doc_type}'."
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
return doc_type # type: ignore[return-value]
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def validate_upload(
|
|
175
|
+
filename: str | None,
|
|
176
|
+
content: bytes | BinaryIO,
|
|
177
|
+
*,
|
|
178
|
+
max_size: int,
|
|
179
|
+
) -> tuple[str, bytes]:
|
|
180
|
+
"""Validate an uploaded file and return (safe_filename, content_bytes)."""
|
|
181
|
+
from edocapi.storage.temporary import safe_filename
|
|
182
|
+
|
|
183
|
+
if not filename:
|
|
184
|
+
raise ValidationError("Uploaded file has no filename.")
|
|
185
|
+
|
|
186
|
+
safe_name = safe_filename(filename)
|
|
187
|
+
validate_extension(safe_name)
|
|
188
|
+
|
|
189
|
+
if hasattr(content, "read"):
|
|
190
|
+
data = content.read()
|
|
191
|
+
else:
|
|
192
|
+
data = content
|
|
193
|
+
|
|
194
|
+
if not isinstance(data, (bytes, bytearray)):
|
|
195
|
+
raise ValidationError("File content must be bytes.")
|
|
196
|
+
|
|
197
|
+
validate_size(len(data), max_size)
|
|
198
|
+
|
|
199
|
+
detected = detect_type_from_bytes(data)
|
|
200
|
+
if detected is None:
|
|
201
|
+
# Still allow based on extension for text formats
|
|
202
|
+
ext = Path(safe_name).suffix.lower()
|
|
203
|
+
if ext not in {".txt", ".md", ".markdown", ".html", ".htm"}:
|
|
204
|
+
raise UnsupportedFileType(
|
|
205
|
+
"Unable to determine a supported file type from content.",
|
|
206
|
+
extension=ext,
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
return safe_name, bytes(data)
|