edocapi 0.0.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
edocapi/responses.py ADDED
@@ -0,0 +1,88 @@
1
+ """Response helpers for eDocAPI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import mimetypes
6
+ from pathlib import Path
7
+ from typing import Any, Mapping
8
+
9
+ from starlette.responses import FileResponse as StarletteFileResponse
10
+ from starlette.responses import JSONResponse as StarletteJSONResponse
11
+ from starlette.responses import Response
12
+
13
+
14
+ class JSONResponse(StarletteJSONResponse):
15
+ """JSON response that serializes normal Python dicts / lists."""
16
+
17
+ def __init__(
18
+ self,
19
+ content: Any = None,
20
+ status_code: int = 200,
21
+ headers: Mapping[str, str] | None = None,
22
+ **kwargs: Any,
23
+ ) -> None:
24
+ super().__init__(
25
+ content=content if content is not None else {},
26
+ status_code=status_code,
27
+ headers=headers,
28
+ **kwargs,
29
+ )
30
+
31
+
32
+ class FileResponse(StarletteFileResponse):
33
+ """File response with sensible defaults for document downloads.
34
+
35
+ Automatically sets Content-Type and Content-Disposition when possible.
36
+ """
37
+
38
+ def __init__(
39
+ self,
40
+ path: str | Path,
41
+ filename: str | None = None,
42
+ media_type: str | None = None,
43
+ status_code: int = 200,
44
+ headers: Mapping[str, str] | None = None,
45
+ background: Any = None,
46
+ **kwargs: Any,
47
+ ) -> None:
48
+ path = Path(path)
49
+ if filename is None:
50
+ filename = path.name
51
+
52
+ if media_type is None:
53
+ guessed, _ = mimetypes.guess_type(filename)
54
+ media_type = guessed or "application/octet-stream"
55
+
56
+ super().__init__(
57
+ path=str(path),
58
+ filename=filename,
59
+ media_type=media_type,
60
+ status_code=status_code,
61
+ headers=headers,
62
+ background=background,
63
+ **kwargs,
64
+ )
65
+
66
+
67
+ def make_error_response(
68
+ error: Exception,
69
+ *,
70
+ status_code: int = 400,
71
+ debug: bool = False,
72
+ ) -> JSONResponse:
73
+ """Create a consistent JSON error response from an exception."""
74
+ from edocapi.exceptions import EdocAPIError
75
+
76
+ if isinstance(error, EdocAPIError):
77
+ body = error.to_dict()
78
+ if debug:
79
+ body["detail"] = str(error)
80
+ else:
81
+ body = {
82
+ "error": type(error).__name__,
83
+ "message": str(error) if debug else "An unexpected error occurred.",
84
+ }
85
+ if debug:
86
+ body["detail"] = str(error)
87
+
88
+ return JSONResponse(content=body, status_code=status_code)
@@ -0,0 +1,3 @@
1
+ from edocapi.storage.temporary import TemporaryStorage, safe_filename
2
+
3
+ __all__ = ["TemporaryStorage", "safe_filename"]
@@ -0,0 +1,132 @@
1
+ """Temporary file management for eDocAPI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import atexit
6
+ import logging
7
+ import os
8
+ import shutil
9
+ import tempfile
10
+ import uuid
11
+ from pathlib import Path
12
+ from typing import Iterator
13
+
14
+ logger = logging.getLogger("edocapi.storage")
15
+
16
+
17
+ class TemporaryStorage:
18
+ """Manages a private temporary directory for document processing.
19
+
20
+ Files created here are cleaned up when the storage instance is closed
21
+ or when the process exits (via atexit).
22
+ """
23
+
24
+ def __init__(self, base_dir: Path | str | None = None) -> None:
25
+ if base_dir is not None:
26
+ self._root = Path(base_dir)
27
+ self._root.mkdir(parents=True, exist_ok=True)
28
+ self._owned = False
29
+ else:
30
+ self._root = Path(tempfile.mkdtemp(prefix="edocapi_"))
31
+ self._owned = True
32
+
33
+ self._files: set[Path] = set()
34
+ self._cleanup_callback = self.cleanup
35
+ atexit.register(self._cleanup_callback)
36
+ logger.debug("Temporary storage created at %s", self._root)
37
+
38
+ @property
39
+ def root(self) -> Path:
40
+ return self._root
41
+
42
+ def create_file(
43
+ self,
44
+ suffix: str = "",
45
+ prefix: str = "tmp_",
46
+ content: bytes | None = None,
47
+ ) -> Path:
48
+ """Create a new temporary file path (and optionally write content)."""
49
+ name = f"{prefix}{uuid.uuid4().hex}{suffix}"
50
+ path = self._root / name
51
+ if content is not None:
52
+ path.write_bytes(content)
53
+ else:
54
+ path.touch()
55
+ self._files.add(path)
56
+ return path
57
+
58
+ def create_dir(self, prefix: str = "dir_") -> Path:
59
+ """Create a subdirectory inside the temporary root."""
60
+ name = f"{prefix}{uuid.uuid4().hex}"
61
+ path = self._root / name
62
+ path.mkdir(parents=True, exist_ok=True)
63
+ self._files.add(path)
64
+ return path
65
+
66
+ def register(self, path: Path | str) -> Path:
67
+ """Register an existing path for later cleanup."""
68
+ path = Path(path)
69
+ self._files.add(path)
70
+ return path
71
+
72
+ def discard(self, path: Path | str) -> None:
73
+ """Remove a file from managed storage after its response is sent."""
74
+ path = Path(path)
75
+ self._files.discard(path)
76
+ try:
77
+ path.unlink(missing_ok=True)
78
+ except OSError as exc:
79
+ logger.warning("Failed to remove temporary file %s: %s", path, exc)
80
+
81
+ def cleanup(self) -> None:
82
+ """Remove all tracked files and, if owned, the root directory."""
83
+ for path in list(self._files):
84
+ try:
85
+ if path.is_file():
86
+ path.unlink(missing_ok=True)
87
+ elif path.is_dir():
88
+ shutil.rmtree(path, ignore_errors=True)
89
+ except OSError as exc:
90
+ logger.warning("Failed to clean up %s: %s", path, exc)
91
+ self._files.clear()
92
+
93
+ if self._owned and self._root.exists():
94
+ try:
95
+ shutil.rmtree(self._root, ignore_errors=True)
96
+ logger.debug("Removed temporary root %s", self._root)
97
+ except OSError as exc:
98
+ logger.warning("Failed to remove temp root %s: %s", self._root, exc)
99
+
100
+ # The instance may be explicitly cleaned up before interpreter exit.
101
+ try:
102
+ atexit.unregister(self._cleanup_callback)
103
+ except Exception:
104
+ pass
105
+
106
+ def __enter__(self) -> "TemporaryStorage":
107
+ return self
108
+
109
+ def __exit__(self, *args: object) -> None:
110
+ self.cleanup()
111
+
112
+ def __del__(self) -> None:
113
+ try:
114
+ self.cleanup()
115
+ except Exception:
116
+ pass
117
+
118
+
119
+ def safe_filename(name: str) -> str:
120
+ """Sanitize a filename to prevent path traversal and unsafe characters."""
121
+ # Take only the basename
122
+ name = os.path.basename(name)
123
+ # Remove null bytes and control characters
124
+ name = "".join(c for c in name if c.isprintable() and c not in r'<>:"/\|?*')
125
+ name = name.strip().strip(".")
126
+ if not name:
127
+ name = "unnamed"
128
+ # Limit length
129
+ if len(name) > 200:
130
+ stem, ext = os.path.splitext(name)
131
+ name = stem[:190] + ext
132
+ return name
@@ -0,0 +1,13 @@
1
+ from edocapi.validation.files import (
2
+ SUPPORTED_EXTENSIONS,
3
+ SUPPORTED_MIME_TYPES,
4
+ validate_file,
5
+ validate_upload,
6
+ )
7
+
8
+ __all__ = [
9
+ "SUPPORTED_EXTENSIONS",
10
+ "SUPPORTED_MIME_TYPES",
11
+ "validate_file",
12
+ "validate_upload",
13
+ ]
@@ -0,0 +1,209 @@
1
+ """File validation utilities for eDocAPI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import logging
6
+ from pathlib import Path
7
+ from typing import BinaryIO
8
+
9
+ from edocapi.exceptions import (
10
+ FileTooLarge,
11
+ InvalidDocument,
12
+ UnsupportedFileType,
13
+ ValidationError,
14
+ )
15
+
16
+ logger = logging.getLogger("edocapi.validation")
17
+
18
+ # Supported formats for v0.0.2
19
+ SUPPORTED_EXTENSIONS = {
20
+ ".pdf",
21
+ ".docx",
22
+ ".html",
23
+ ".htm",
24
+ ".md",
25
+ ".markdown",
26
+ ".jpg",
27
+ ".jpeg",
28
+ ".png",
29
+ ".webp",
30
+ ".txt",
31
+ }
32
+
33
+ SUPPORTED_MIME_TYPES = {
34
+ "application/pdf",
35
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
36
+ "text/html",
37
+ "text/markdown",
38
+ "text/x-markdown",
39
+ "image/jpeg",
40
+ "image/png",
41
+ "image/webp",
42
+ "text/plain",
43
+ }
44
+
45
+ # Magic byte signatures (first few bytes)
46
+ MAGIC_SIGNATURES: dict[bytes, str] = {
47
+ b"%PDF": "pdf",
48
+ b"PK\x03\x04": "zip", # DOCX is a ZIP; further checks needed
49
+ b"\xff\xd8\xff": "jpeg",
50
+ b"\x89PNG\r\n\x1a\n": "png",
51
+ b"RIFF": "webp", # needs further check for WEBP
52
+ }
53
+
54
+
55
+ def detect_type_from_bytes(data: bytes) -> str | None:
56
+ """Attempt to detect document type from magic bytes."""
57
+ if not data:
58
+ return None
59
+
60
+ for sig, kind in MAGIC_SIGNATURES.items():
61
+ if data.startswith(sig):
62
+ if kind == "zip":
63
+ # Could be DOCX or other Office format
64
+ if b"word/" in data[:4096] or b"[Content_Types].xml" in data[:4096]:
65
+ return "docx"
66
+ return "zip"
67
+ if kind == "webp":
68
+ if b"WEBP" in data[8:16]:
69
+ return "webp"
70
+ return None
71
+ return kind
72
+
73
+ # HTML / Markdown / TXT heuristics
74
+ sample = data[:1024].decode("utf-8", errors="ignore").strip().lower()
75
+ if sample.startswith("<!doctype html") or sample.startswith("<html"):
76
+ return "html"
77
+ if sample.startswith("# ") or sample.startswith("---\n"):
78
+ return "markdown"
79
+ # Plain text fallback only if mostly printable
80
+ if all(32 <= b < 127 or b in (9, 10, 13) for b in data[:512]):
81
+ return "txt"
82
+ return None
83
+
84
+
85
+ def validate_extension(filename: str) -> str:
86
+ """Validate and return the normalized extension (including dot)."""
87
+ from edocapi.storage.temporary import safe_filename
88
+
89
+ name = safe_filename(filename)
90
+ ext = Path(name).suffix.lower()
91
+ if ext not in SUPPORTED_EXTENSIONS:
92
+ raise UnsupportedFileType(
93
+ f"File extension '{ext}' is not supported.",
94
+ extension=ext,
95
+ )
96
+ return ext
97
+
98
+
99
+ def validate_size(size: int, max_size: int) -> None:
100
+ """Raise FileTooLarge if size exceeds the limit."""
101
+ if size > max_size:
102
+ raise FileTooLarge(size=size, max_size=max_size)
103
+
104
+
105
+ def validate_file(
106
+ path: Path,
107
+ *,
108
+ max_size: int,
109
+ expected_type: str | None = None,
110
+ ) -> str:
111
+ """Perform basic validation on a file path.
112
+
113
+ Returns the detected document type string (e.g. 'pdf', 'docx').
114
+ """
115
+ if not path.exists():
116
+ raise ValidationError(f"File does not exist: {path}")
117
+ if not path.is_file():
118
+ raise ValidationError(f"Path is not a regular file: {path}")
119
+
120
+ size = path.stat().st_size
121
+ validate_size(size, max_size)
122
+
123
+ # Read a small header for magic detection
124
+ with path.open("rb") as f:
125
+ header = f.read(4096)
126
+
127
+ detected = detect_type_from_bytes(header)
128
+ ext = path.suffix.lower()
129
+
130
+ # Map extension to type
131
+ ext_to_type = {
132
+ ".pdf": "pdf",
133
+ ".docx": "docx",
134
+ ".html": "html",
135
+ ".htm": "html",
136
+ ".md": "markdown",
137
+ ".markdown": "markdown",
138
+ ".jpg": "jpeg",
139
+ ".jpeg": "jpeg",
140
+ ".png": "png",
141
+ ".webp": "webp",
142
+ ".txt": "txt",
143
+ }
144
+ ext_type = ext_to_type.get(ext)
145
+
146
+ if detected is None and ext_type is None:
147
+ raise UnsupportedFileType(
148
+ "Unable to determine file type.",
149
+ extension=ext,
150
+ )
151
+
152
+ # Prefer detected type; fall back to extension
153
+ doc_type = detected or ext_type
154
+
155
+ # Basic consistency check
156
+ if detected and ext_type and detected != ext_type:
157
+ # Allow some flexibility (e.g. .txt that looks like markdown)
158
+ if not (detected in ("txt", "markdown") and ext_type in ("txt", "markdown")):
159
+ logger.warning(
160
+ "Magic type %s does not match extension type %s for %s",
161
+ detected,
162
+ ext_type,
163
+ path.name,
164
+ )
165
+
166
+ if expected_type and doc_type != expected_type:
167
+ raise InvalidDocument(
168
+ f"Expected document type '{expected_type}', got '{doc_type}'."
169
+ )
170
+
171
+ return doc_type # type: ignore[return-value]
172
+
173
+
174
+ def validate_upload(
175
+ filename: str | None,
176
+ content: bytes | BinaryIO,
177
+ *,
178
+ max_size: int,
179
+ ) -> tuple[str, bytes]:
180
+ """Validate an uploaded file and return (safe_filename, content_bytes)."""
181
+ from edocapi.storage.temporary import safe_filename
182
+
183
+ if not filename:
184
+ raise ValidationError("Uploaded file has no filename.")
185
+
186
+ safe_name = safe_filename(filename)
187
+ validate_extension(safe_name)
188
+
189
+ if hasattr(content, "read"):
190
+ data = content.read()
191
+ else:
192
+ data = content
193
+
194
+ if not isinstance(data, (bytes, bytearray)):
195
+ raise ValidationError("File content must be bytes.")
196
+
197
+ validate_size(len(data), max_size)
198
+
199
+ detected = detect_type_from_bytes(data)
200
+ if detected is None:
201
+ # Still allow based on extension for text formats
202
+ ext = Path(safe_name).suffix.lower()
203
+ if ext not in {".txt", ".md", ".markdown", ".html", ".htm"}:
204
+ raise UnsupportedFileType(
205
+ "Unable to determine a supported file type from content.",
206
+ extension=ext,
207
+ )
208
+
209
+ return safe_name, bytes(data)