pdf-to-markdown-cli 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,199 @@
1
+ import logging
2
+ import os
3
+ import uuid
4
+ from pathlib import Path
5
+ from typing import List, Optional
6
+
7
+ import pikepdf
8
+ from pydantic import BaseModel
9
+ from tqdm import tqdm
10
+
11
+ from docs_to_md.utils.exceptions import PDFProcessingError
12
+ from docs_to_md.utils.file_utils import ensure_directory
13
+ from docs_to_md.utils.logging import ProgressTracker
14
+
15
+ logger = logging.getLogger(__name__)
16
+
17
+
18
+ def _sanitize_path(path: Path) -> Path:
19
+ """
20
+ Sanitize path to prevent path traversal attacks.
21
+
22
+ Args:
23
+ path: Path to sanitize
24
+
25
+ Returns:
26
+ Sanitized path
27
+
28
+ Raises:
29
+ PDFProcessingError: If path contains suspicious patterns
30
+ """
31
+ try:
32
+ # Convert to absolute to resolve any .. or other path manipulations
33
+ path = path.absolute().resolve()
34
+
35
+ # Get the filename only, discarding any directory components
36
+ safe_name = path.name
37
+
38
+ # Check for suspicious patterns
39
+ if safe_name != path.name or ".." in str(path):
40
+ raise PDFProcessingError(f"Potentially unsafe path: {path}")
41
+
42
+ return path
43
+ except Exception as e:
44
+ if isinstance(e, PDFProcessingError):
45
+ raise
46
+ raise PDFProcessingError(f"Failed to sanitize path {path}: {e}")
47
+
48
+
49
+ class PDFChunkInfo(BaseModel):
50
+ """Information about a single PDF chunk."""
51
+ path: str
52
+ index: int
53
+ start_page: int
54
+ end_page: int
55
+
56
+
57
+ class PDFChunks(BaseModel):
58
+ """Collection of PDF chunks."""
59
+ chunks: List[PDFChunkInfo]
60
+
61
+
62
+ def _create_chunk(pdf: pikepdf.Pdf, chunks_dir: Path, chunk_num: int, num_chunks: int, start: int, end: int) -> str:
63
+ """
64
+ Create a single PDF chunk.
65
+
66
+ Args:
67
+ pdf: Source PDF
68
+ chunks_dir: Directory to save chunks
69
+ chunk_num: Index of this chunk
70
+ num_chunks: Total number of chunks
71
+ start: Start page index (inclusive)
72
+ end: End page index (exclusive)
73
+
74
+ Returns:
75
+ Path to created chunk file
76
+
77
+ Raises:
78
+ PDFProcessingError: If chunk creation fails
79
+ """
80
+ chunk_pdf = pikepdf.Pdf.new()
81
+ try:
82
+ for i in range(start, end):
83
+ chunk_pdf.pages.append(pdf.pages[i])
84
+
85
+ # Ensure safe path creation
86
+ chunk_filename = f"{chunk_num+1:03d}of{num_chunks:03d}.pdf"
87
+ chunk_path = str(chunks_dir / chunk_filename)
88
+
89
+ chunk_pdf.save(
90
+ chunk_path,
91
+ compress_streams=True,
92
+ object_stream_mode=pikepdf.ObjectStreamMode.generate
93
+ )
94
+ return chunk_path
95
+ except Exception as e:
96
+ raise PDFProcessingError(f"Failed to create PDF chunk {chunk_num+1}: {e}")
97
+ finally:
98
+ chunk_pdf.close()
99
+
100
+
101
+ def _create_chunks(pdf: pikepdf.Pdf, path: Path, pages_per_chunk: int, tmp_dir: Path) -> PDFChunks:
102
+ """
103
+ Create multiple chunks from a PDF.
104
+
105
+ Args:
106
+ pdf: Source PDF
107
+ path: Original PDF path (for naming)
108
+ pages_per_chunk: Number of pages per chunk
109
+ tmp_dir: Directory to save chunks
110
+
111
+ Returns:
112
+ PDFChunks with information about created chunks
113
+
114
+ Raises:
115
+ PDFProcessingError: If chunking fails
116
+ """
117
+ # Use provided temp directory
118
+ ensure_directory(tmp_dir)
119
+
120
+ num_chunks = (len(pdf.pages) + pages_per_chunk - 1) // pages_per_chunk
121
+ chunks: List[PDFChunkInfo] = []
122
+
123
+ progress = ProgressTracker(num_chunks, "Chunking PDF", "chunk")
124
+
125
+ try:
126
+ for chunk_num in range(num_chunks):
127
+ start = chunk_num * pages_per_chunk
128
+ end = min(start + pages_per_chunk, len(pdf.pages))
129
+ chunk_path = _create_chunk(pdf, tmp_dir, chunk_num, num_chunks, start, end)
130
+ chunks.append(PDFChunkInfo(
131
+ path=chunk_path,
132
+ index=chunk_num,
133
+ start_page=start,
134
+ end_page=end-1
135
+ ))
136
+ progress.update()
137
+
138
+ if not chunks:
139
+ raise PDFProcessingError(f"Failed to create any chunks for {path}")
140
+
141
+ return PDFChunks(chunks=chunks)
142
+
143
+ finally:
144
+ progress.close()
145
+
146
+
147
+ def chunk_pdf_to_temp(pdf_path: str, pages_per_chunk: int = 10, tmp_dir: Optional[Path] = None) -> Optional[PDFChunks]:
148
+ """
149
+ Split a PDF into chunks of specified size and save to temp directory.
150
+
151
+ Args:
152
+ pdf_path: Path to the PDF file
153
+ pages_per_chunk: Number of pages per chunk (default: 10)
154
+ tmp_dir: Directory to save chunks in (default: creates one)
155
+
156
+ Returns:
157
+ PDFChunks containing information about each chunk, or None if no chunking needed
158
+
159
+ Raises:
160
+ PDFProcessingError: If the PDF is invalid or cannot be processed
161
+ ValueError: If pages_per_chunk < 1
162
+ """
163
+ if pages_per_chunk < 1:
164
+ raise ValueError("pages_per_chunk must be at least 1")
165
+
166
+ # Sanitize and validate path
167
+ path = _sanitize_path(Path(pdf_path))
168
+
169
+ if not path.exists():
170
+ raise PDFProcessingError(f"File does not exist: {pdf_path}")
171
+
172
+ if path.suffix.lower() != '.pdf':
173
+ raise PDFProcessingError(f"File is not a PDF: {pdf_path}")
174
+
175
+ pdf = None
176
+ try:
177
+ pdf = pikepdf.Pdf.open(pdf_path)
178
+ if len(pdf.pages) == 0:
179
+ raise PDFProcessingError(f"PDF has no pages: {pdf_path}")
180
+
181
+ if len(pdf.pages) <= pages_per_chunk:
182
+ return None # No chunking needed
183
+
184
+ # Create temp directory if not provided
185
+ if tmp_dir is None:
186
+ tmp_dir = Path("chunks") / f"{path.stem}_{uuid.uuid4().hex[:8]}"
187
+ ensure_directory(tmp_dir)
188
+
189
+ return _create_chunks(pdf, path, pages_per_chunk, tmp_dir)
190
+
191
+ except pikepdf.PdfError as e:
192
+ raise PDFProcessingError(f"Invalid PDF {pdf_path}: {e}")
193
+ except Exception as e:
194
+ if isinstance(e, PDFProcessingError):
195
+ raise
196
+ raise PDFProcessingError(f"Error processing PDF {pdf_path}: {e}")
197
+ finally:
198
+ if pdf is not None:
199
+ pdf.close()
File without changes
@@ -0,0 +1,125 @@
1
+ import logging
2
+ import threading
3
+ from pathlib import Path
4
+ from typing import List, Optional
5
+
6
+ from diskcache import Cache
7
+
8
+ from docs_to_md.storage.models import ConversionRequest
9
+ from docs_to_md.utils.exceptions import CacheError
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+
14
+ class CacheManager:
15
+ """Handles persistence of conversion requests."""
16
+
17
+ def __init__(self, cache_dir: str):
18
+ """
19
+ Initialize cache manager with given directory.
20
+
21
+ Args:
22
+ cache_dir: Directory for cache storage
23
+
24
+ Raises:
25
+ CacheError: If cache initialization fails
26
+ """
27
+ try:
28
+ # Initialize with thread safety enabled
29
+ self.cache: Cache = Cache(cache_dir, statistics=True, timeout=60)
30
+ # Lock for thread-safe operations
31
+ self._lock = threading.RLock()
32
+ except Exception as e:
33
+ logger.error(f"Failed to initialize cache in {cache_dir}: {e}")
34
+ raise CacheError(f"Failed to initialize cache in {cache_dir}: {e}")
35
+
36
+ def save(self, request: ConversionRequest) -> None:
37
+ """
38
+ Save request to cache.
39
+
40
+ Args:
41
+ request: Request to save
42
+
43
+ Raises:
44
+ CacheError: If save operation fails
45
+ """
46
+ with self._lock:
47
+ try:
48
+ self.cache.set(request.request_id, request.model_dump())
49
+ except Exception as e:
50
+ logger.error(f"Failed to save request {request.request_id}: {e}")
51
+ raise CacheError(f"Failed to save request {request.request_id}: {e}")
52
+
53
+ def get(self, request_id: str) -> Optional[ConversionRequest]:
54
+ """
55
+ Get request from cache.
56
+
57
+ Args:
58
+ request_id: ID of request to retrieve
59
+
60
+ Returns:
61
+ ConversionRequest if found, None otherwise
62
+
63
+ Raises:
64
+ CacheError: If retrieval fails
65
+ """
66
+ with self._lock:
67
+ try:
68
+ data = self.cache.get(request_id)
69
+ if data:
70
+ return ConversionRequest.model_validate(data)
71
+ return None
72
+ except Exception as e:
73
+ logger.error(f"Failed to get request {request_id}: {e}")
74
+ return None
75
+
76
+ def delete(self, request_id: str) -> bool:
77
+ """
78
+ Delete request from cache.
79
+
80
+ Args:
81
+ request_id: ID of request to delete
82
+
83
+ Returns:
84
+ True if successful, False otherwise
85
+ """
86
+ with self._lock:
87
+ try:
88
+ return bool(self.cache.delete(request_id))
89
+ except Exception as e:
90
+ logger.error(f"Failed to delete request {request_id}: {e}")
91
+ return False
92
+
93
+ def get_all(self) -> List[ConversionRequest]:
94
+ """
95
+ Get all requests from cache.
96
+
97
+ Returns:
98
+ List of all conversion requests
99
+ """
100
+ results = []
101
+ # Use a consistent view of the cache to avoid inconsistencies
102
+ # during iteration
103
+ with self._lock:
104
+ keys = list(self.cache.iterkeys())
105
+
106
+ for key in keys:
107
+ if request := self.get(str(key)):
108
+ results.append(request)
109
+ return results
110
+
111
+ def close(self) -> None:
112
+ """Close cache connection and free resources."""
113
+ with self._lock:
114
+ try:
115
+ self.cache.close()
116
+ except Exception as e:
117
+ logger.error(f"Error closing cache: {e}")
118
+
119
+ def __enter__(self):
120
+ """Support for context manager."""
121
+ return self
122
+
123
+ def __exit__(self, exc_type, exc_val, exc_tb):
124
+ """Clean up resources when exiting context."""
125
+ self.close()
@@ -0,0 +1,85 @@
1
+ from enum import Enum
2
+ from pathlib import Path
3
+ from typing import List, Optional
4
+
5
+ from pydantic import BaseModel
6
+
7
+
8
+ class Status(str, Enum):
9
+ """Status of a request or chunk."""
10
+ PENDING = "pending" # Waiting to be processed by the API
11
+ PROCESSING = "processing" # Currently being processed by the API
12
+ COMPLETE = "complete" # Successfully processed by the API
13
+ FAILED = "failed" # Failed to process by the API
14
+
15
+
16
+ class ChunkInfo(BaseModel):
17
+ """Information about a chunk or original file being processed."""
18
+ path: Path
19
+ index: int
20
+ request_id: Optional[str] = None
21
+ status: Status = Status.PENDING
22
+ error: Optional[str] = None
23
+
24
+ def mark_processing(self, request_id: str) -> None:
25
+ """Mark chunk as processing with given request ID."""
26
+ self.request_id = request_id
27
+ self.status = Status.PROCESSING
28
+
29
+ def mark_failed(self, error: str) -> None:
30
+ """Mark chunk as failed with error message."""
31
+ self.status = Status.FAILED
32
+ self.error = error
33
+
34
+ def mark_complete(self) -> None:
35
+ """Mark chunk as complete."""
36
+ self.status = Status.COMPLETE
37
+
38
+ def get_result_path(self, tmp_dir: Path) -> Path:
39
+ """Get the path where the result should be stored."""
40
+ return tmp_dir / f"{Path(self.path).name}.out"
41
+
42
+
43
+ class ConversionRequest(BaseModel):
44
+ """Tracks a conversion request and its state."""
45
+ request_id: str
46
+ original_file: Path
47
+ target_file: Path
48
+ output_format: str = "markdown"
49
+ status: Status = Status.PENDING
50
+ error: Optional[str] = None
51
+ chunks: List[ChunkInfo] = []
52
+ chunk_size: int
53
+ tmp_dir: Optional[Path] = None # Directory for temporary files for this conversion
54
+
55
+ def set_status(self, status: Status, error: Optional[str] = None) -> None:
56
+ """Set status and optional error message."""
57
+ self.status = status
58
+ if error:
59
+ self.error = error
60
+
61
+ def add_chunk(self, path: Path, index: int) -> ChunkInfo:
62
+ """Add a new chunk and return it."""
63
+ chunk = ChunkInfo(path=path, index=index)
64
+ self.chunks.append(chunk)
65
+ return chunk
66
+
67
+ @property
68
+ def pending_chunks(self) -> List[ChunkInfo]:
69
+ """Get all pending or processing chunks."""
70
+ return [c for c in self.chunks if c.status in (Status.PENDING, Status.PROCESSING)]
71
+
72
+ @property
73
+ def ordered_chunks(self) -> List[ChunkInfo]:
74
+ """Get chunks ordered by index."""
75
+ return sorted(self.chunks, key=lambda x: x.index)
76
+
77
+ @property
78
+ def has_failed(self) -> bool:
79
+ """Check if any chunks have failed."""
80
+ return any(c.status == Status.FAILED for c in self.chunks)
81
+
82
+ @property
83
+ def all_complete(self) -> bool:
84
+ """Check if all chunks are complete."""
85
+ return all(c.status == Status.COMPLETE for c in self.chunks)
File without changes
@@ -0,0 +1,33 @@
1
+ class DocsToMdError(Exception):
2
+ """Base exception for all docs-to-md errors."""
3
+ pass
4
+
5
+
6
+ class ConfigurationError(DocsToMdError):
7
+ """Error related to configuration loading or validation."""
8
+ pass
9
+
10
+
11
+ class FileError(DocsToMdError):
12
+ """Error related to file operations (reading, writing, discovery)."""
13
+ pass
14
+
15
+
16
+ class APIError(DocsToMdError):
17
+ """Error related to the external conversion API communication."""
18
+ pass
19
+
20
+
21
+ class PDFProcessingError(DocsToMdError):
22
+ """Error related to PDF manipulation (splitting, etc.)."""
23
+ pass
24
+
25
+
26
+ class CacheError(DocsToMdError):
27
+ """Error related to cache operations."""
28
+ pass
29
+
30
+
31
+ class ResultProcessingError(DocsToMdError):
32
+ """Error related to processing or combining results."""
33
+ pass
@@ -0,0 +1,189 @@
1
+ import hashlib
2
+ import logging
3
+ import os
4
+ import shutil
5
+ import uuid
6
+ from pathlib import Path
7
+ from typing import List, Optional, Set
8
+
9
+ import filetype
10
+
11
+ from docs_to_md.utils.exceptions import FileError
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+
16
+ def ensure_directory(path: Path) -> None:
17
+ """Create directory if it doesn't exist."""
18
+ path.mkdir(parents=True, exist_ok=True)
19
+
20
+
21
+ def safe_delete(path: Path) -> None:
22
+ """Safely delete a file or directory."""
23
+ try:
24
+ path = Path(path) # Convert to Path if string
25
+ if not path.exists():
26
+ return
27
+
28
+ if path.is_file():
29
+ path.unlink()
30
+ elif path.is_dir():
31
+ # Remove all files in directory recursively
32
+ shutil.rmtree(path, ignore_errors=True)
33
+ except Exception as e:
34
+ logger.error(f"Failed to delete {path}: {e}")
35
+
36
+
37
+ def get_file_size(path: Path) -> int:
38
+ """Get file size in bytes."""
39
+ try:
40
+ return path.stat().st_size
41
+ except Exception as e:
42
+ logger.error(f"Error getting file size for {path}: {e}")
43
+ return 0
44
+
45
+
46
+ def get_env_var(name: str, required: bool = True) -> Optional[str]:
47
+ """Get environment variable with optional requirement."""
48
+ value = os.getenv(name)
49
+ if required and not value:
50
+ raise FileError(f"Required environment variable {name} is not set")
51
+ return value
52
+
53
+
54
+ class FileDiscovery:
55
+ """Handles finding and filtering files based on various criteria."""
56
+
57
+ @staticmethod
58
+ def find_processable_files(input_path: Path, supported_types: Set[str],
59
+ supported_extensions: Optional[List[str]] = None) -> List[Path]:
60
+ """
61
+ Find all processable files from an input path.
62
+
63
+ Args:
64
+ input_path: Directory or file path to search
65
+ supported_types: Set of supported MIME types
66
+ supported_extensions: List of supported file extensions (without dot)
67
+
68
+ Returns:
69
+ List of processable file paths
70
+ """
71
+ if supported_extensions is None:
72
+ supported_extensions = ['.pdf', '.docx', '.doc', '.pptx', '.ppt',
73
+ '.jpg', '.jpeg', '.png', '.gif', '.webp', '.tiff']
74
+
75
+ files_to_process = []
76
+
77
+ # Helper function to check file type without loading entire file
78
+ def check_file_type(file_path: Path) -> Optional[filetype.Type]:
79
+ try:
80
+ # Only read first 8KB for type detection rather than the whole file
81
+ with open(file_path, 'rb') as f:
82
+ header = f.read(8192) # First 8KB is enough for type detection
83
+ return filetype.guess(header)
84
+ except Exception as e:
85
+ logger.warning(f"Error reading file header for {file_path}: {e}")
86
+ return None
87
+
88
+ # If input is a file, validate and return it
89
+ if input_path.is_file():
90
+ try:
91
+ kind = check_file_type(input_path)
92
+ if (kind and kind.mime in supported_types) or \
93
+ (input_path.suffix.lower() in supported_extensions):
94
+ files_to_process.append(input_path)
95
+ else:
96
+ logger.warning(f"Unsupported file type: {input_path}")
97
+ except Exception as e:
98
+ logger.warning(f"Error checking file {input_path}: {e}")
99
+
100
+ return files_to_process
101
+
102
+ # If input is a directory, find all matching files
103
+ if not input_path.exists():
104
+ raise FileError(f"Input path does not exist: {input_path}")
105
+
106
+ # Process directory
107
+ for p in input_path.glob("**/*"):
108
+ if p.is_file():
109
+ try:
110
+ kind = check_file_type(p)
111
+ if (kind and kind.mime in supported_types):
112
+ files_to_process.append(p)
113
+ elif p.suffix.lower() in supported_extensions:
114
+ # Some file types might not be detected correctly by filetype
115
+ files_to_process.append(p)
116
+ except Exception as e:
117
+ logger.warning(f"Error checking file type for {p}: {e}")
118
+
119
+ return files_to_process
120
+
121
+
122
+ class TemporaryDirectory:
123
+ """Context manager for temporary directories."""
124
+
125
+ def __init__(self, base_path: Path, prefix: str):
126
+ """
127
+ Initialize a temporary directory.
128
+
129
+ Args:
130
+ base_path: Base path where to create the temporary directory
131
+ prefix: Prefix for the directory name
132
+ """
133
+ self.path = base_path / f"{prefix}_{uuid.uuid4().hex[:8]}"
134
+ ensure_directory(self.path)
135
+
136
+ def __enter__(self) -> Path:
137
+ """Enter the context and return the path to the temporary directory."""
138
+ return self.path
139
+
140
+ def __exit__(self, exc_type, exc_val, exc_tb) -> None:
141
+ """Clean up the temporary directory when exiting the context."""
142
+ safe_delete(self.path)
143
+
144
+
145
+ class FileIO:
146
+ """Utility class for file operations."""
147
+
148
+ @staticmethod
149
+ def read_file(path: Path) -> bytes:
150
+ """Read a file's content as bytes."""
151
+ try:
152
+ return path.read_bytes()
153
+ except Exception as e:
154
+ raise FileError(f"Failed to read file {path}: {e}")
155
+
156
+ @staticmethod
157
+ def read_text(path: Path, encoding: str = 'utf-8') -> str:
158
+ """Read a file's content as text."""
159
+ try:
160
+ return path.read_text(encoding=encoding)
161
+ except Exception as e:
162
+ raise FileError(f"Failed to read file {path}: {e}")
163
+
164
+ @staticmethod
165
+ def write_file(path: Path, content: str, encoding: str = 'utf-8') -> None:
166
+ """Write text content to a file."""
167
+ try:
168
+ ensure_directory(path.parent)
169
+ path.write_text(content, encoding=encoding)
170
+ except Exception as e:
171
+ raise FileError(f"Failed to write to file {path}: {e}")
172
+
173
+ @staticmethod
174
+ def write_binary(path: Path, content: bytes) -> None:
175
+ """Write binary content to a file."""
176
+ try:
177
+ ensure_directory(path.parent)
178
+ path.write_bytes(content)
179
+ except Exception as e:
180
+ raise FileError(f"Failed to write binary data to {path}: {e}")
181
+
182
+ @staticmethod
183
+ def copy_file(src: Path, dst: Path) -> None:
184
+ """Copy a file from source to destination."""
185
+ try:
186
+ ensure_directory(dst.parent)
187
+ shutil.copy2(src, dst)
188
+ except Exception as e:
189
+ raise FileError(f"Failed to copy file from {src} to {dst}: {e}")