pdf-to-markdown-cli 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,229 @@
1
+ import logging
2
+ import uuid
3
+ from pathlib import Path
4
+
5
+ from docs_to_md.api.client import MarkerClient
6
+ from docs_to_md.api.models import SUPPORTED_MIME_TYPES, ApiParams
7
+ from docs_to_md.config.settings import Config
8
+ from docs_to_md.pdf.splitter import chunk_pdf_to_temp
9
+ from docs_to_md.storage.cache import CacheManager
10
+ from docs_to_md.storage.models import ConversionRequest, Status
11
+ from docs_to_md.utils.exceptions import FileError, PDFProcessingError
12
+ from docs_to_md.utils.file_utils import FileDiscovery, TemporaryDirectory, ensure_directory
13
+ from docs_to_md.utils.logging import ProgressTracker
14
+ from docs_to_md.core.result_handler import ResultHandler
15
+
16
+ logger = logging.getLogger(__name__)
17
+
18
+
19
+ class BatchProcessor:
20
+ """Handles processing of files, including chunking if needed."""
21
+
22
+ def __init__(self, api_key: str, cache_dir: Path, chunk_size: int = 25):
23
+ """
24
+ Initialize the batch processor.
25
+
26
+ Args:
27
+ api_key: API key for authentication
28
+ cache_dir: Directory for caching
29
+ chunk_size: Pages per chunk for PDFs
30
+ """
31
+ self.client = MarkerClient(api_key)
32
+ self.cache = CacheManager(str(cache_dir))
33
+ self.chunk_size = chunk_size # pages per chunk for PDFs
34
+
35
+ def should_chunk(self, file_path: Path) -> bool:
36
+ """
37
+ Determine if a file should be chunked.
38
+
39
+ Args:
40
+ file_path: Path to check
41
+
42
+ Returns:
43
+ True if file should be chunked, False otherwise
44
+ """
45
+ return file_path.suffix.lower() == '.pdf'
46
+
47
+ def process_file(
48
+ self,
49
+ file_path: Path,
50
+ output_dir: Path,
51
+ api_params: ApiParams
52
+ ) -> str:
53
+ """
54
+ Process a single file.
55
+
56
+ Args:
57
+ file_path: Path to file
58
+ output_dir: Output directory
59
+ api_params: Parameters for the API call
60
+
61
+ Returns:
62
+ Request ID for tracking
63
+
64
+ Raises:
65
+ FileError: If file processing fails
66
+ """
67
+ # Create temp directory for this conversion
68
+ with TemporaryDirectory(Path("chunks"), file_path.stem) as tmp_dir:
69
+ ensure_directory(output_dir)
70
+
71
+ # Initialize request
72
+ request = ConversionRequest(
73
+ request_id=str(uuid.uuid4()),
74
+ original_file=file_path,
75
+ target_file=output_dir / file_path.name,
76
+ output_format=api_params.output_format,
77
+ status=Status.PENDING,
78
+ tmp_dir=tmp_dir,
79
+ chunk_size=self.chunk_size
80
+ )
81
+
82
+ # Save request early for tracking
83
+ self.cache.save(request)
84
+
85
+ try:
86
+ # Handle PDF chunking
87
+ if self.should_chunk(file_path):
88
+ try:
89
+ chunk_result = chunk_pdf_to_temp(str(file_path), self.chunk_size, tmp_dir)
90
+ if chunk_result:
91
+ # Add additional chunks
92
+ for chunk_info in chunk_result.chunks:
93
+ request.add_chunk(Path(chunk_info.path), chunk_info.index)
94
+ except (PDFProcessingError, Exception) as e:
95
+ logger.error(f"Error chunking PDF {file_path}: {e}")
96
+ request.set_status(Status.FAILED, f"Error chunking PDF: {str(e)}")
97
+ self.cache.save(request)
98
+ return request.request_id
99
+
100
+ if len(request.chunks) == 0: # If no chunks created, add original file
101
+ request.add_chunk(file_path, 0)
102
+
103
+ # Submit all chunks to API
104
+ logger.info(f"Submitting {len(request.chunks)} chunk(s) to API...")
105
+
106
+ progress = ProgressTracker(len(request.chunks), "Submitting to API", "chunk")
107
+
108
+ for chunk in request.ordered_chunks:
109
+ chunk_request_id = self.client.submit_file(
110
+ chunk.path,
111
+ output_format=api_params.output_format,
112
+ langs=api_params.langs,
113
+ use_llm=api_params.use_llm,
114
+ strip_existing_ocr=api_params.strip_existing_ocr,
115
+ disable_image_extraction=api_params.disable_image_extraction,
116
+ force_ocr=api_params.force_ocr,
117
+ paginate=api_params.paginate,
118
+ max_pages=api_params.max_pages
119
+ )
120
+ if chunk_request_id:
121
+ chunk.mark_processing(chunk_request_id)
122
+ else:
123
+ chunk.mark_failed(f"Failed to submit file {chunk.path}")
124
+ break
125
+
126
+ progress.update()
127
+
128
+ progress.close()
129
+
130
+ # Update request status
131
+ if request.has_failed:
132
+ request.set_status(Status.FAILED)
133
+ else:
134
+ request.status = Status.PROCESSING
135
+
136
+ self.cache.save(request)
137
+ return request.request_id
138
+
139
+ except Exception as e:
140
+ request.status = Status.FAILED
141
+ request.error = str(e)
142
+ self.cache.save(request)
143
+ logger.error(f"Error processing file {file_path}: {e}")
144
+ return request.request_id
145
+
146
+ def close(self) -> None:
147
+ """Clean up resources."""
148
+ self.cache.close()
149
+
150
+ def __enter__(self):
151
+ """Support for context manager."""
152
+ return self
153
+
154
+ def __exit__(self, exc_type, exc_val, exc_tb):
155
+ """Clean up resources when exiting context."""
156
+ self.close()
157
+
158
+
159
+ class MarkerProcessor:
160
+ """Handles the core business logic for processing PDFs."""
161
+
162
+ def __init__(self, config: Config):
163
+ """
164
+ Initialize the processor with configuration.
165
+
166
+ Args:
167
+ config: Application configuration
168
+ """
169
+ self.config = config
170
+ self.config.ensure_directories()
171
+
172
+ def process(self) -> None:
173
+ """
174
+ Process files according to configuration.
175
+
176
+ Raises:
177
+ FileError: If file processing fails
178
+ """
179
+ # Initialize processors
180
+ with BatchProcessor(self.config.api_key, self.config.cache_dir, chunk_size=self.config.chunk_size) as batch_processor, \
181
+ ResultHandler(self.config.api_key, self.config.cache_dir) as result_handler:
182
+
183
+ try:
184
+ # Process input path
185
+ input_path = Path(self.config.input_path)
186
+
187
+ # Find processable files
188
+ supported_extensions = ['.pdf', '.docx', '.doc', '.pptx', '.ppt',
189
+ '.jpg', '.jpeg', '.png', '.gif', '.webp', '.tiff']
190
+
191
+ files_to_process = FileDiscovery.find_processable_files(
192
+ input_path,
193
+ SUPPORTED_MIME_TYPES,
194
+ supported_extensions
195
+ )
196
+
197
+ if not files_to_process:
198
+ logger.warning(f"No processable files found in {input_path}")
199
+ return
200
+
201
+ # Create ApiParams from config
202
+ api_params = ApiParams(
203
+ output_format=self.config.output_format,
204
+ langs=self.config.langs,
205
+ use_llm=self.config.use_llm,
206
+ strip_existing_ocr=self.config.strip_existing_ocr,
207
+ disable_image_extraction=self.config.disable_image_extraction,
208
+ force_ocr=self.config.force_ocr,
209
+ paginate=self.config.paginate,
210
+ max_pages=self.config.max_pages
211
+ )
212
+
213
+ # Process each file
214
+ for file_path in files_to_process:
215
+ logger.info(f"Processing {file_path}")
216
+ batch_processor.process_file(
217
+ file_path=file_path,
218
+ output_dir=self.config.output_dir,
219
+ api_params=api_params
220
+ )
221
+
222
+ # Process results
223
+ result_handler.process_cache_items()
224
+
225
+ logger.info("All processing completed successfully.")
226
+
227
+ except Exception as e:
228
+ logger.error(f"Error during processing: {e}")
229
+ raise FileError(f"Processing failed: {e}")