pdf-to-markdown-cli 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docs_to_md/__init__.py +0 -0
- docs_to_md/__main__.py +5 -0
- docs_to_md/api/__init__.py +0 -0
- docs_to_md/api/client.py +218 -0
- docs_to_md/api/models.py +66 -0
- docs_to_md/config/__init__.py +0 -0
- docs_to_md/config/cli.py +86 -0
- docs_to_md/config/settings.py +108 -0
- docs_to_md/core/__init__.py +0 -0
- docs_to_md/core/processor.py +229 -0
- docs_to_md/core/result_handler.py +496 -0
- docs_to_md/main.py +55 -0
- docs_to_md/pdf/__init__.py +0 -0
- docs_to_md/pdf/splitter.py +199 -0
- docs_to_md/storage/__init__.py +0 -0
- docs_to_md/storage/cache.py +125 -0
- docs_to_md/storage/models.py +85 -0
- docs_to_md/utils/__init__.py +0 -0
- docs_to_md/utils/exceptions.py +33 -0
- docs_to_md/utils/file_utils.py +189 -0
- docs_to_md/utils/logging.py +134 -0
- pdf_to_markdown_cli-0.2.0.dist-info/METADATA +179 -0
- pdf_to_markdown_cli-0.2.0.dist-info/RECORD +27 -0
- pdf_to_markdown_cli-0.2.0.dist-info/WHEEL +5 -0
- pdf_to_markdown_cli-0.2.0.dist-info/entry_points.txt +2 -0
- pdf_to_markdown_cli-0.2.0.dist-info/licenses/LICENSE +21 -0
- pdf_to_markdown_cli-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
import uuid
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import List, Optional
|
|
6
|
+
|
|
7
|
+
import pikepdf
|
|
8
|
+
from pydantic import BaseModel
|
|
9
|
+
from tqdm import tqdm
|
|
10
|
+
|
|
11
|
+
from docs_to_md.utils.exceptions import PDFProcessingError
|
|
12
|
+
from docs_to_md.utils.file_utils import ensure_directory
|
|
13
|
+
from docs_to_md.utils.logging import ProgressTracker
|
|
14
|
+
|
|
15
|
+
logger = logging.getLogger(__name__)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _sanitize_path(path: Path) -> Path:
|
|
19
|
+
"""
|
|
20
|
+
Sanitize path to prevent path traversal attacks.
|
|
21
|
+
|
|
22
|
+
Args:
|
|
23
|
+
path: Path to sanitize
|
|
24
|
+
|
|
25
|
+
Returns:
|
|
26
|
+
Sanitized path
|
|
27
|
+
|
|
28
|
+
Raises:
|
|
29
|
+
PDFProcessingError: If path contains suspicious patterns
|
|
30
|
+
"""
|
|
31
|
+
try:
|
|
32
|
+
# Convert to absolute to resolve any .. or other path manipulations
|
|
33
|
+
path = path.absolute().resolve()
|
|
34
|
+
|
|
35
|
+
# Get the filename only, discarding any directory components
|
|
36
|
+
safe_name = path.name
|
|
37
|
+
|
|
38
|
+
# Check for suspicious patterns
|
|
39
|
+
if safe_name != path.name or ".." in str(path):
|
|
40
|
+
raise PDFProcessingError(f"Potentially unsafe path: {path}")
|
|
41
|
+
|
|
42
|
+
return path
|
|
43
|
+
except Exception as e:
|
|
44
|
+
if isinstance(e, PDFProcessingError):
|
|
45
|
+
raise
|
|
46
|
+
raise PDFProcessingError(f"Failed to sanitize path {path}: {e}")
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class PDFChunkInfo(BaseModel):
|
|
50
|
+
"""Information about a single PDF chunk."""
|
|
51
|
+
path: str
|
|
52
|
+
index: int
|
|
53
|
+
start_page: int
|
|
54
|
+
end_page: int
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class PDFChunks(BaseModel):
|
|
58
|
+
"""Collection of PDF chunks."""
|
|
59
|
+
chunks: List[PDFChunkInfo]
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _create_chunk(pdf: pikepdf.Pdf, chunks_dir: Path, chunk_num: int, num_chunks: int, start: int, end: int) -> str:
|
|
63
|
+
"""
|
|
64
|
+
Create a single PDF chunk.
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
pdf: Source PDF
|
|
68
|
+
chunks_dir: Directory to save chunks
|
|
69
|
+
chunk_num: Index of this chunk
|
|
70
|
+
num_chunks: Total number of chunks
|
|
71
|
+
start: Start page index (inclusive)
|
|
72
|
+
end: End page index (exclusive)
|
|
73
|
+
|
|
74
|
+
Returns:
|
|
75
|
+
Path to created chunk file
|
|
76
|
+
|
|
77
|
+
Raises:
|
|
78
|
+
PDFProcessingError: If chunk creation fails
|
|
79
|
+
"""
|
|
80
|
+
chunk_pdf = pikepdf.Pdf.new()
|
|
81
|
+
try:
|
|
82
|
+
for i in range(start, end):
|
|
83
|
+
chunk_pdf.pages.append(pdf.pages[i])
|
|
84
|
+
|
|
85
|
+
# Ensure safe path creation
|
|
86
|
+
chunk_filename = f"{chunk_num+1:03d}of{num_chunks:03d}.pdf"
|
|
87
|
+
chunk_path = str(chunks_dir / chunk_filename)
|
|
88
|
+
|
|
89
|
+
chunk_pdf.save(
|
|
90
|
+
chunk_path,
|
|
91
|
+
compress_streams=True,
|
|
92
|
+
object_stream_mode=pikepdf.ObjectStreamMode.generate
|
|
93
|
+
)
|
|
94
|
+
return chunk_path
|
|
95
|
+
except Exception as e:
|
|
96
|
+
raise PDFProcessingError(f"Failed to create PDF chunk {chunk_num+1}: {e}")
|
|
97
|
+
finally:
|
|
98
|
+
chunk_pdf.close()
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _create_chunks(pdf: pikepdf.Pdf, path: Path, pages_per_chunk: int, tmp_dir: Path) -> PDFChunks:
|
|
102
|
+
"""
|
|
103
|
+
Create multiple chunks from a PDF.
|
|
104
|
+
|
|
105
|
+
Args:
|
|
106
|
+
pdf: Source PDF
|
|
107
|
+
path: Original PDF path (for naming)
|
|
108
|
+
pages_per_chunk: Number of pages per chunk
|
|
109
|
+
tmp_dir: Directory to save chunks
|
|
110
|
+
|
|
111
|
+
Returns:
|
|
112
|
+
PDFChunks with information about created chunks
|
|
113
|
+
|
|
114
|
+
Raises:
|
|
115
|
+
PDFProcessingError: If chunking fails
|
|
116
|
+
"""
|
|
117
|
+
# Use provided temp directory
|
|
118
|
+
ensure_directory(tmp_dir)
|
|
119
|
+
|
|
120
|
+
num_chunks = (len(pdf.pages) + pages_per_chunk - 1) // pages_per_chunk
|
|
121
|
+
chunks: List[PDFChunkInfo] = []
|
|
122
|
+
|
|
123
|
+
progress = ProgressTracker(num_chunks, "Chunking PDF", "chunk")
|
|
124
|
+
|
|
125
|
+
try:
|
|
126
|
+
for chunk_num in range(num_chunks):
|
|
127
|
+
start = chunk_num * pages_per_chunk
|
|
128
|
+
end = min(start + pages_per_chunk, len(pdf.pages))
|
|
129
|
+
chunk_path = _create_chunk(pdf, tmp_dir, chunk_num, num_chunks, start, end)
|
|
130
|
+
chunks.append(PDFChunkInfo(
|
|
131
|
+
path=chunk_path,
|
|
132
|
+
index=chunk_num,
|
|
133
|
+
start_page=start,
|
|
134
|
+
end_page=end-1
|
|
135
|
+
))
|
|
136
|
+
progress.update()
|
|
137
|
+
|
|
138
|
+
if not chunks:
|
|
139
|
+
raise PDFProcessingError(f"Failed to create any chunks for {path}")
|
|
140
|
+
|
|
141
|
+
return PDFChunks(chunks=chunks)
|
|
142
|
+
|
|
143
|
+
finally:
|
|
144
|
+
progress.close()
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def chunk_pdf_to_temp(pdf_path: str, pages_per_chunk: int = 10, tmp_dir: Optional[Path] = None) -> Optional[PDFChunks]:
|
|
148
|
+
"""
|
|
149
|
+
Split a PDF into chunks of specified size and save to temp directory.
|
|
150
|
+
|
|
151
|
+
Args:
|
|
152
|
+
pdf_path: Path to the PDF file
|
|
153
|
+
pages_per_chunk: Number of pages per chunk (default: 10)
|
|
154
|
+
tmp_dir: Directory to save chunks in (default: creates one)
|
|
155
|
+
|
|
156
|
+
Returns:
|
|
157
|
+
PDFChunks containing information about each chunk, or None if no chunking needed
|
|
158
|
+
|
|
159
|
+
Raises:
|
|
160
|
+
PDFProcessingError: If the PDF is invalid or cannot be processed
|
|
161
|
+
ValueError: If pages_per_chunk < 1
|
|
162
|
+
"""
|
|
163
|
+
if pages_per_chunk < 1:
|
|
164
|
+
raise ValueError("pages_per_chunk must be at least 1")
|
|
165
|
+
|
|
166
|
+
# Sanitize and validate path
|
|
167
|
+
path = _sanitize_path(Path(pdf_path))
|
|
168
|
+
|
|
169
|
+
if not path.exists():
|
|
170
|
+
raise PDFProcessingError(f"File does not exist: {pdf_path}")
|
|
171
|
+
|
|
172
|
+
if path.suffix.lower() != '.pdf':
|
|
173
|
+
raise PDFProcessingError(f"File is not a PDF: {pdf_path}")
|
|
174
|
+
|
|
175
|
+
pdf = None
|
|
176
|
+
try:
|
|
177
|
+
pdf = pikepdf.Pdf.open(pdf_path)
|
|
178
|
+
if len(pdf.pages) == 0:
|
|
179
|
+
raise PDFProcessingError(f"PDF has no pages: {pdf_path}")
|
|
180
|
+
|
|
181
|
+
if len(pdf.pages) <= pages_per_chunk:
|
|
182
|
+
return None # No chunking needed
|
|
183
|
+
|
|
184
|
+
# Create temp directory if not provided
|
|
185
|
+
if tmp_dir is None:
|
|
186
|
+
tmp_dir = Path("chunks") / f"{path.stem}_{uuid.uuid4().hex[:8]}"
|
|
187
|
+
ensure_directory(tmp_dir)
|
|
188
|
+
|
|
189
|
+
return _create_chunks(pdf, path, pages_per_chunk, tmp_dir)
|
|
190
|
+
|
|
191
|
+
except pikepdf.PdfError as e:
|
|
192
|
+
raise PDFProcessingError(f"Invalid PDF {pdf_path}: {e}")
|
|
193
|
+
except Exception as e:
|
|
194
|
+
if isinstance(e, PDFProcessingError):
|
|
195
|
+
raise
|
|
196
|
+
raise PDFProcessingError(f"Error processing PDF {pdf_path}: {e}")
|
|
197
|
+
finally:
|
|
198
|
+
if pdf is not None:
|
|
199
|
+
pdf.close()
|
|
File without changes
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import threading
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import List, Optional
|
|
5
|
+
|
|
6
|
+
from diskcache import Cache
|
|
7
|
+
|
|
8
|
+
from docs_to_md.storage.models import ConversionRequest
|
|
9
|
+
from docs_to_md.utils.exceptions import CacheError
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class CacheManager:
|
|
15
|
+
"""Handles persistence of conversion requests."""
|
|
16
|
+
|
|
17
|
+
def __init__(self, cache_dir: str):
|
|
18
|
+
"""
|
|
19
|
+
Initialize cache manager with given directory.
|
|
20
|
+
|
|
21
|
+
Args:
|
|
22
|
+
cache_dir: Directory for cache storage
|
|
23
|
+
|
|
24
|
+
Raises:
|
|
25
|
+
CacheError: If cache initialization fails
|
|
26
|
+
"""
|
|
27
|
+
try:
|
|
28
|
+
# Initialize with thread safety enabled
|
|
29
|
+
self.cache: Cache = Cache(cache_dir, statistics=True, timeout=60)
|
|
30
|
+
# Lock for thread-safe operations
|
|
31
|
+
self._lock = threading.RLock()
|
|
32
|
+
except Exception as e:
|
|
33
|
+
logger.error(f"Failed to initialize cache in {cache_dir}: {e}")
|
|
34
|
+
raise CacheError(f"Failed to initialize cache in {cache_dir}: {e}")
|
|
35
|
+
|
|
36
|
+
def save(self, request: ConversionRequest) -> None:
|
|
37
|
+
"""
|
|
38
|
+
Save request to cache.
|
|
39
|
+
|
|
40
|
+
Args:
|
|
41
|
+
request: Request to save
|
|
42
|
+
|
|
43
|
+
Raises:
|
|
44
|
+
CacheError: If save operation fails
|
|
45
|
+
"""
|
|
46
|
+
with self._lock:
|
|
47
|
+
try:
|
|
48
|
+
self.cache.set(request.request_id, request.model_dump())
|
|
49
|
+
except Exception as e:
|
|
50
|
+
logger.error(f"Failed to save request {request.request_id}: {e}")
|
|
51
|
+
raise CacheError(f"Failed to save request {request.request_id}: {e}")
|
|
52
|
+
|
|
53
|
+
def get(self, request_id: str) -> Optional[ConversionRequest]:
|
|
54
|
+
"""
|
|
55
|
+
Get request from cache.
|
|
56
|
+
|
|
57
|
+
Args:
|
|
58
|
+
request_id: ID of request to retrieve
|
|
59
|
+
|
|
60
|
+
Returns:
|
|
61
|
+
ConversionRequest if found, None otherwise
|
|
62
|
+
|
|
63
|
+
Raises:
|
|
64
|
+
CacheError: If retrieval fails
|
|
65
|
+
"""
|
|
66
|
+
with self._lock:
|
|
67
|
+
try:
|
|
68
|
+
data = self.cache.get(request_id)
|
|
69
|
+
if data:
|
|
70
|
+
return ConversionRequest.model_validate(data)
|
|
71
|
+
return None
|
|
72
|
+
except Exception as e:
|
|
73
|
+
logger.error(f"Failed to get request {request_id}: {e}")
|
|
74
|
+
return None
|
|
75
|
+
|
|
76
|
+
def delete(self, request_id: str) -> bool:
|
|
77
|
+
"""
|
|
78
|
+
Delete request from cache.
|
|
79
|
+
|
|
80
|
+
Args:
|
|
81
|
+
request_id: ID of request to delete
|
|
82
|
+
|
|
83
|
+
Returns:
|
|
84
|
+
True if successful, False otherwise
|
|
85
|
+
"""
|
|
86
|
+
with self._lock:
|
|
87
|
+
try:
|
|
88
|
+
return bool(self.cache.delete(request_id))
|
|
89
|
+
except Exception as e:
|
|
90
|
+
logger.error(f"Failed to delete request {request_id}: {e}")
|
|
91
|
+
return False
|
|
92
|
+
|
|
93
|
+
def get_all(self) -> List[ConversionRequest]:
|
|
94
|
+
"""
|
|
95
|
+
Get all requests from cache.
|
|
96
|
+
|
|
97
|
+
Returns:
|
|
98
|
+
List of all conversion requests
|
|
99
|
+
"""
|
|
100
|
+
results = []
|
|
101
|
+
# Use a consistent view of the cache to avoid inconsistencies
|
|
102
|
+
# during iteration
|
|
103
|
+
with self._lock:
|
|
104
|
+
keys = list(self.cache.iterkeys())
|
|
105
|
+
|
|
106
|
+
for key in keys:
|
|
107
|
+
if request := self.get(str(key)):
|
|
108
|
+
results.append(request)
|
|
109
|
+
return results
|
|
110
|
+
|
|
111
|
+
def close(self) -> None:
|
|
112
|
+
"""Close cache connection and free resources."""
|
|
113
|
+
with self._lock:
|
|
114
|
+
try:
|
|
115
|
+
self.cache.close()
|
|
116
|
+
except Exception as e:
|
|
117
|
+
logger.error(f"Error closing cache: {e}")
|
|
118
|
+
|
|
119
|
+
def __enter__(self):
|
|
120
|
+
"""Support for context manager."""
|
|
121
|
+
return self
|
|
122
|
+
|
|
123
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
124
|
+
"""Clean up resources when exiting context."""
|
|
125
|
+
self.close()
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
from enum import Enum
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import List, Optional
|
|
4
|
+
|
|
5
|
+
from pydantic import BaseModel
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class Status(str, Enum):
|
|
9
|
+
"""Status of a request or chunk."""
|
|
10
|
+
PENDING = "pending" # Waiting to be processed by the API
|
|
11
|
+
PROCESSING = "processing" # Currently being processed by the API
|
|
12
|
+
COMPLETE = "complete" # Successfully processed by the API
|
|
13
|
+
FAILED = "failed" # Failed to process by the API
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class ChunkInfo(BaseModel):
|
|
17
|
+
"""Information about a chunk or original file being processed."""
|
|
18
|
+
path: Path
|
|
19
|
+
index: int
|
|
20
|
+
request_id: Optional[str] = None
|
|
21
|
+
status: Status = Status.PENDING
|
|
22
|
+
error: Optional[str] = None
|
|
23
|
+
|
|
24
|
+
def mark_processing(self, request_id: str) -> None:
|
|
25
|
+
"""Mark chunk as processing with given request ID."""
|
|
26
|
+
self.request_id = request_id
|
|
27
|
+
self.status = Status.PROCESSING
|
|
28
|
+
|
|
29
|
+
def mark_failed(self, error: str) -> None:
|
|
30
|
+
"""Mark chunk as failed with error message."""
|
|
31
|
+
self.status = Status.FAILED
|
|
32
|
+
self.error = error
|
|
33
|
+
|
|
34
|
+
def mark_complete(self) -> None:
|
|
35
|
+
"""Mark chunk as complete."""
|
|
36
|
+
self.status = Status.COMPLETE
|
|
37
|
+
|
|
38
|
+
def get_result_path(self, tmp_dir: Path) -> Path:
|
|
39
|
+
"""Get the path where the result should be stored."""
|
|
40
|
+
return tmp_dir / f"{Path(self.path).name}.out"
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class ConversionRequest(BaseModel):
|
|
44
|
+
"""Tracks a conversion request and its state."""
|
|
45
|
+
request_id: str
|
|
46
|
+
original_file: Path
|
|
47
|
+
target_file: Path
|
|
48
|
+
output_format: str = "markdown"
|
|
49
|
+
status: Status = Status.PENDING
|
|
50
|
+
error: Optional[str] = None
|
|
51
|
+
chunks: List[ChunkInfo] = []
|
|
52
|
+
chunk_size: int
|
|
53
|
+
tmp_dir: Optional[Path] = None # Directory for temporary files for this conversion
|
|
54
|
+
|
|
55
|
+
def set_status(self, status: Status, error: Optional[str] = None) -> None:
|
|
56
|
+
"""Set status and optional error message."""
|
|
57
|
+
self.status = status
|
|
58
|
+
if error:
|
|
59
|
+
self.error = error
|
|
60
|
+
|
|
61
|
+
def add_chunk(self, path: Path, index: int) -> ChunkInfo:
|
|
62
|
+
"""Add a new chunk and return it."""
|
|
63
|
+
chunk = ChunkInfo(path=path, index=index)
|
|
64
|
+
self.chunks.append(chunk)
|
|
65
|
+
return chunk
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def pending_chunks(self) -> List[ChunkInfo]:
|
|
69
|
+
"""Get all pending or processing chunks."""
|
|
70
|
+
return [c for c in self.chunks if c.status in (Status.PENDING, Status.PROCESSING)]
|
|
71
|
+
|
|
72
|
+
@property
|
|
73
|
+
def ordered_chunks(self) -> List[ChunkInfo]:
|
|
74
|
+
"""Get chunks ordered by index."""
|
|
75
|
+
return sorted(self.chunks, key=lambda x: x.index)
|
|
76
|
+
|
|
77
|
+
@property
|
|
78
|
+
def has_failed(self) -> bool:
|
|
79
|
+
"""Check if any chunks have failed."""
|
|
80
|
+
return any(c.status == Status.FAILED for c in self.chunks)
|
|
81
|
+
|
|
82
|
+
@property
|
|
83
|
+
def all_complete(self) -> bool:
|
|
84
|
+
"""Check if all chunks are complete."""
|
|
85
|
+
return all(c.status == Status.COMPLETE for c in self.chunks)
|
|
File without changes
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
class DocsToMdError(Exception):
|
|
2
|
+
"""Base exception for all docs-to-md errors."""
|
|
3
|
+
pass
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class ConfigurationError(DocsToMdError):
|
|
7
|
+
"""Error related to configuration loading or validation."""
|
|
8
|
+
pass
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class FileError(DocsToMdError):
|
|
12
|
+
"""Error related to file operations (reading, writing, discovery)."""
|
|
13
|
+
pass
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class APIError(DocsToMdError):
|
|
17
|
+
"""Error related to the external conversion API communication."""
|
|
18
|
+
pass
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class PDFProcessingError(DocsToMdError):
|
|
22
|
+
"""Error related to PDF manipulation (splitting, etc.)."""
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class CacheError(DocsToMdError):
|
|
27
|
+
"""Error related to cache operations."""
|
|
28
|
+
pass
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class ResultProcessingError(DocsToMdError):
|
|
32
|
+
"""Error related to processing or combining results."""
|
|
33
|
+
pass
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
import hashlib
|
|
2
|
+
import logging
|
|
3
|
+
import os
|
|
4
|
+
import shutil
|
|
5
|
+
import uuid
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import List, Optional, Set
|
|
8
|
+
|
|
9
|
+
import filetype
|
|
10
|
+
|
|
11
|
+
from docs_to_md.utils.exceptions import FileError
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def ensure_directory(path: Path) -> None:
|
|
17
|
+
"""Create directory if it doesn't exist."""
|
|
18
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def safe_delete(path: Path) -> None:
|
|
22
|
+
"""Safely delete a file or directory."""
|
|
23
|
+
try:
|
|
24
|
+
path = Path(path) # Convert to Path if string
|
|
25
|
+
if not path.exists():
|
|
26
|
+
return
|
|
27
|
+
|
|
28
|
+
if path.is_file():
|
|
29
|
+
path.unlink()
|
|
30
|
+
elif path.is_dir():
|
|
31
|
+
# Remove all files in directory recursively
|
|
32
|
+
shutil.rmtree(path, ignore_errors=True)
|
|
33
|
+
except Exception as e:
|
|
34
|
+
logger.error(f"Failed to delete {path}: {e}")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def get_file_size(path: Path) -> int:
|
|
38
|
+
"""Get file size in bytes."""
|
|
39
|
+
try:
|
|
40
|
+
return path.stat().st_size
|
|
41
|
+
except Exception as e:
|
|
42
|
+
logger.error(f"Error getting file size for {path}: {e}")
|
|
43
|
+
return 0
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def get_env_var(name: str, required: bool = True) -> Optional[str]:
|
|
47
|
+
"""Get environment variable with optional requirement."""
|
|
48
|
+
value = os.getenv(name)
|
|
49
|
+
if required and not value:
|
|
50
|
+
raise FileError(f"Required environment variable {name} is not set")
|
|
51
|
+
return value
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class FileDiscovery:
|
|
55
|
+
"""Handles finding and filtering files based on various criteria."""
|
|
56
|
+
|
|
57
|
+
@staticmethod
|
|
58
|
+
def find_processable_files(input_path: Path, supported_types: Set[str],
|
|
59
|
+
supported_extensions: Optional[List[str]] = None) -> List[Path]:
|
|
60
|
+
"""
|
|
61
|
+
Find all processable files from an input path.
|
|
62
|
+
|
|
63
|
+
Args:
|
|
64
|
+
input_path: Directory or file path to search
|
|
65
|
+
supported_types: Set of supported MIME types
|
|
66
|
+
supported_extensions: List of supported file extensions (without dot)
|
|
67
|
+
|
|
68
|
+
Returns:
|
|
69
|
+
List of processable file paths
|
|
70
|
+
"""
|
|
71
|
+
if supported_extensions is None:
|
|
72
|
+
supported_extensions = ['.pdf', '.docx', '.doc', '.pptx', '.ppt',
|
|
73
|
+
'.jpg', '.jpeg', '.png', '.gif', '.webp', '.tiff']
|
|
74
|
+
|
|
75
|
+
files_to_process = []
|
|
76
|
+
|
|
77
|
+
# Helper function to check file type without loading entire file
|
|
78
|
+
def check_file_type(file_path: Path) -> Optional[filetype.Type]:
|
|
79
|
+
try:
|
|
80
|
+
# Only read first 8KB for type detection rather than the whole file
|
|
81
|
+
with open(file_path, 'rb') as f:
|
|
82
|
+
header = f.read(8192) # First 8KB is enough for type detection
|
|
83
|
+
return filetype.guess(header)
|
|
84
|
+
except Exception as e:
|
|
85
|
+
logger.warning(f"Error reading file header for {file_path}: {e}")
|
|
86
|
+
return None
|
|
87
|
+
|
|
88
|
+
# If input is a file, validate and return it
|
|
89
|
+
if input_path.is_file():
|
|
90
|
+
try:
|
|
91
|
+
kind = check_file_type(input_path)
|
|
92
|
+
if (kind and kind.mime in supported_types) or \
|
|
93
|
+
(input_path.suffix.lower() in supported_extensions):
|
|
94
|
+
files_to_process.append(input_path)
|
|
95
|
+
else:
|
|
96
|
+
logger.warning(f"Unsupported file type: {input_path}")
|
|
97
|
+
except Exception as e:
|
|
98
|
+
logger.warning(f"Error checking file {input_path}: {e}")
|
|
99
|
+
|
|
100
|
+
return files_to_process
|
|
101
|
+
|
|
102
|
+
# If input is a directory, find all matching files
|
|
103
|
+
if not input_path.exists():
|
|
104
|
+
raise FileError(f"Input path does not exist: {input_path}")
|
|
105
|
+
|
|
106
|
+
# Process directory
|
|
107
|
+
for p in input_path.glob("**/*"):
|
|
108
|
+
if p.is_file():
|
|
109
|
+
try:
|
|
110
|
+
kind = check_file_type(p)
|
|
111
|
+
if (kind and kind.mime in supported_types):
|
|
112
|
+
files_to_process.append(p)
|
|
113
|
+
elif p.suffix.lower() in supported_extensions:
|
|
114
|
+
# Some file types might not be detected correctly by filetype
|
|
115
|
+
files_to_process.append(p)
|
|
116
|
+
except Exception as e:
|
|
117
|
+
logger.warning(f"Error checking file type for {p}: {e}")
|
|
118
|
+
|
|
119
|
+
return files_to_process
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
class TemporaryDirectory:
|
|
123
|
+
"""Context manager for temporary directories."""
|
|
124
|
+
|
|
125
|
+
def __init__(self, base_path: Path, prefix: str):
|
|
126
|
+
"""
|
|
127
|
+
Initialize a temporary directory.
|
|
128
|
+
|
|
129
|
+
Args:
|
|
130
|
+
base_path: Base path where to create the temporary directory
|
|
131
|
+
prefix: Prefix for the directory name
|
|
132
|
+
"""
|
|
133
|
+
self.path = base_path / f"{prefix}_{uuid.uuid4().hex[:8]}"
|
|
134
|
+
ensure_directory(self.path)
|
|
135
|
+
|
|
136
|
+
def __enter__(self) -> Path:
|
|
137
|
+
"""Enter the context and return the path to the temporary directory."""
|
|
138
|
+
return self.path
|
|
139
|
+
|
|
140
|
+
def __exit__(self, exc_type, exc_val, exc_tb) -> None:
|
|
141
|
+
"""Clean up the temporary directory when exiting the context."""
|
|
142
|
+
safe_delete(self.path)
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
class FileIO:
|
|
146
|
+
"""Utility class for file operations."""
|
|
147
|
+
|
|
148
|
+
@staticmethod
|
|
149
|
+
def read_file(path: Path) -> bytes:
|
|
150
|
+
"""Read a file's content as bytes."""
|
|
151
|
+
try:
|
|
152
|
+
return path.read_bytes()
|
|
153
|
+
except Exception as e:
|
|
154
|
+
raise FileError(f"Failed to read file {path}: {e}")
|
|
155
|
+
|
|
156
|
+
@staticmethod
|
|
157
|
+
def read_text(path: Path, encoding: str = 'utf-8') -> str:
|
|
158
|
+
"""Read a file's content as text."""
|
|
159
|
+
try:
|
|
160
|
+
return path.read_text(encoding=encoding)
|
|
161
|
+
except Exception as e:
|
|
162
|
+
raise FileError(f"Failed to read file {path}: {e}")
|
|
163
|
+
|
|
164
|
+
@staticmethod
|
|
165
|
+
def write_file(path: Path, content: str, encoding: str = 'utf-8') -> None:
|
|
166
|
+
"""Write text content to a file."""
|
|
167
|
+
try:
|
|
168
|
+
ensure_directory(path.parent)
|
|
169
|
+
path.write_text(content, encoding=encoding)
|
|
170
|
+
except Exception as e:
|
|
171
|
+
raise FileError(f"Failed to write to file {path}: {e}")
|
|
172
|
+
|
|
173
|
+
@staticmethod
|
|
174
|
+
def write_binary(path: Path, content: bytes) -> None:
|
|
175
|
+
"""Write binary content to a file."""
|
|
176
|
+
try:
|
|
177
|
+
ensure_directory(path.parent)
|
|
178
|
+
path.write_bytes(content)
|
|
179
|
+
except Exception as e:
|
|
180
|
+
raise FileError(f"Failed to write binary data to {path}: {e}")
|
|
181
|
+
|
|
182
|
+
@staticmethod
|
|
183
|
+
def copy_file(src: Path, dst: Path) -> None:
|
|
184
|
+
"""Copy a file from source to destination."""
|
|
185
|
+
try:
|
|
186
|
+
ensure_directory(dst.parent)
|
|
187
|
+
shutil.copy2(src, dst)
|
|
188
|
+
except Exception as e:
|
|
189
|
+
raise FileError(f"Failed to copy file from {src} to {dst}: {e}")
|