pdf-to-markdown-cli 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docs_to_md/__init__.py +0 -0
- docs_to_md/__main__.py +5 -0
- docs_to_md/api/__init__.py +0 -0
- docs_to_md/api/client.py +218 -0
- docs_to_md/api/models.py +66 -0
- docs_to_md/config/__init__.py +0 -0
- docs_to_md/config/cli.py +86 -0
- docs_to_md/config/settings.py +108 -0
- docs_to_md/core/__init__.py +0 -0
- docs_to_md/core/processor.py +229 -0
- docs_to_md/core/result_handler.py +496 -0
- docs_to_md/main.py +55 -0
- docs_to_md/pdf/__init__.py +0 -0
- docs_to_md/pdf/splitter.py +199 -0
- docs_to_md/storage/__init__.py +0 -0
- docs_to_md/storage/cache.py +125 -0
- docs_to_md/storage/models.py +85 -0
- docs_to_md/utils/__init__.py +0 -0
- docs_to_md/utils/exceptions.py +33 -0
- docs_to_md/utils/file_utils.py +189 -0
- docs_to_md/utils/logging.py +134 -0
- pdf_to_markdown_cli-0.2.0.dist-info/METADATA +179 -0
- pdf_to_markdown_cli-0.2.0.dist-info/RECORD +27 -0
- pdf_to_markdown_cli-0.2.0.dist-info/WHEEL +5 -0
- pdf_to_markdown_cli-0.2.0.dist-info/entry_points.txt +2 -0
- pdf_to_markdown_cli-0.2.0.dist-info/licenses/LICENSE +21 -0
- pdf_to_markdown_cli-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,496 @@
|
|
|
1
|
+
import base64
|
|
2
|
+
import json
|
|
3
|
+
import logging
|
|
4
|
+
import re
|
|
5
|
+
import shutil
|
|
6
|
+
import time
|
|
7
|
+
import uuid
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Dict, List, Optional, Tuple
|
|
11
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
12
|
+
|
|
13
|
+
from docs_to_md.api.client import MarkerClient
|
|
14
|
+
from docs_to_md.api.models import MarkerStatus, StatusEnum
|
|
15
|
+
from docs_to_md.storage.cache import CacheManager
|
|
16
|
+
from docs_to_md.storage.models import ChunkInfo, ConversionRequest, Status
|
|
17
|
+
from docs_to_md.utils.exceptions import ResultProcessingError
|
|
18
|
+
from docs_to_md.utils.file_utils import FileIO, ensure_directory, safe_delete
|
|
19
|
+
from docs_to_md.utils.logging import ProgressTracker
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger(__name__)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class ImageProcessor:
|
|
25
|
+
"""Handles processing and transformation of images extracted from documents."""
|
|
26
|
+
|
|
27
|
+
def transform_image_name(self, original_name: str, chunk: ChunkInfo,
|
|
28
|
+
chunk_size: int) -> Tuple[str, str]:
|
|
29
|
+
"""
|
|
30
|
+
Transform image name and return new names for file and markdown.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
original_name: Original image name from API
|
|
34
|
+
chunk: Chunk info for context
|
|
35
|
+
chunk_size: Number of pages per chunk
|
|
36
|
+
|
|
37
|
+
Returns:
|
|
38
|
+
Tuple of (new_filename, markdown_reference_name)
|
|
39
|
+
"""
|
|
40
|
+
base_page_num = (chunk.index * chunk_size) + 1
|
|
41
|
+
|
|
42
|
+
# Extract file extension properly
|
|
43
|
+
parts = original_name.split('.')
|
|
44
|
+
extension = ""
|
|
45
|
+
if len(parts) > 1:
|
|
46
|
+
extension = f".{parts[-1].lower()}"
|
|
47
|
+
# Validate extension is a common image type
|
|
48
|
+
if extension not in ['.jpg', '.jpeg', '.png', '.gif', '.webp', '.tiff']:
|
|
49
|
+
extension = ".jpg" # Default to jpg if unknown
|
|
50
|
+
else:
|
|
51
|
+
extension = ".jpg" # Default extension
|
|
52
|
+
|
|
53
|
+
# Try to extract page and figure numbers with more robust patterns
|
|
54
|
+
page_match = re.search(r'(?:_|-)page(?:_|-)?(\d+)', original_name, re.IGNORECASE)
|
|
55
|
+
figure_match = re.search(r'(?:_|-)(?:figure|fig)(?:_|-)?(\d+)', original_name, re.IGNORECASE)
|
|
56
|
+
|
|
57
|
+
if page_match and figure_match:
|
|
58
|
+
# Calculate new page number based on chunk position
|
|
59
|
+
page_num = int(page_match.group(1))
|
|
60
|
+
corrected_page_num = base_page_num + page_num - 1 # -1 because pages often start at 1
|
|
61
|
+
figure_num = int(figure_match.group(1))
|
|
62
|
+
|
|
63
|
+
# Generate new filename with corrected page number
|
|
64
|
+
new_name = f"page_{corrected_page_num}_figure_{figure_num}{extension}"
|
|
65
|
+
return new_name, new_name
|
|
66
|
+
else:
|
|
67
|
+
# Fallback: use chunk number, timestamp, and random identifier for uniqueness
|
|
68
|
+
timestamp = datetime.now().strftime("%H%M%S")
|
|
69
|
+
random_suffix = uuid.uuid4().hex[:6]
|
|
70
|
+
fallback_name = f"chunk_{chunk.index}_img_{timestamp}_{random_suffix}{extension}"
|
|
71
|
+
return fallback_name, fallback_name
|
|
72
|
+
|
|
73
|
+
def process_images(self, images: Dict[str, str], chunk: ChunkInfo,
|
|
74
|
+
tmp_dir: Path, chunk_size: int) -> Dict[str, str]:
|
|
75
|
+
"""
|
|
76
|
+
Process images from API response.
|
|
77
|
+
|
|
78
|
+
Args:
|
|
79
|
+
images: Dictionary of image names to base64 content
|
|
80
|
+
chunk: Chunk info for context
|
|
81
|
+
tmp_dir: Temporary directory for storage
|
|
82
|
+
chunk_size: Number of pages per chunk
|
|
83
|
+
|
|
84
|
+
Returns:
|
|
85
|
+
Dictionary mapping original image names to new image references
|
|
86
|
+
"""
|
|
87
|
+
if not images:
|
|
88
|
+
return {}
|
|
89
|
+
|
|
90
|
+
# Prepare images directory
|
|
91
|
+
images_dir = tmp_dir / "images"
|
|
92
|
+
ensure_directory(images_dir)
|
|
93
|
+
|
|
94
|
+
# Process each image
|
|
95
|
+
image_map = {}
|
|
96
|
+
for original_name, b64_content in images.items():
|
|
97
|
+
# Transform name
|
|
98
|
+
new_name, markdown_name = self.transform_image_name(original_name, chunk, chunk_size)
|
|
99
|
+
|
|
100
|
+
# Save image
|
|
101
|
+
try:
|
|
102
|
+
image_data = base64.b64decode(b64_content)
|
|
103
|
+
(images_dir / new_name).write_bytes(image_data)
|
|
104
|
+
|
|
105
|
+
# Map original name to new markdown reference
|
|
106
|
+
image_map[original_name] = f"images/{markdown_name}"
|
|
107
|
+
except Exception as e:
|
|
108
|
+
logger.error(f"Failed to save image {original_name} to {new_name}: {e}")
|
|
109
|
+
|
|
110
|
+
return image_map
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
class ResultSaver:
|
|
114
|
+
"""Handles saving and combining results."""
|
|
115
|
+
|
|
116
|
+
def __init__(self):
|
|
117
|
+
"""Initialize the result saver."""
|
|
118
|
+
self.format_extensions = {
|
|
119
|
+
"markdown": ".md",
|
|
120
|
+
"json": ".json",
|
|
121
|
+
"html": ".html",
|
|
122
|
+
"txt": ".txt" # default
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
def save_content(self, content: str, path: Path) -> None:
|
|
126
|
+
"""
|
|
127
|
+
Save content to file, ensuring parent directory exists.
|
|
128
|
+
|
|
129
|
+
Args:
|
|
130
|
+
content: Content to save
|
|
131
|
+
path: Path to save to
|
|
132
|
+
|
|
133
|
+
Raises:
|
|
134
|
+
ResultProcessingError: If save fails
|
|
135
|
+
"""
|
|
136
|
+
try:
|
|
137
|
+
FileIO.write_file(path, content)
|
|
138
|
+
except Exception as e:
|
|
139
|
+
raise ResultProcessingError(f"Failed to save content to {path}: {e}")
|
|
140
|
+
|
|
141
|
+
def get_output_directory(self, req: ConversionRequest) -> Path:
|
|
142
|
+
"""
|
|
143
|
+
Create and return output directory with timestamp.
|
|
144
|
+
|
|
145
|
+
Args:
|
|
146
|
+
req: Conversion request
|
|
147
|
+
|
|
148
|
+
Returns:
|
|
149
|
+
Path to output directory
|
|
150
|
+
"""
|
|
151
|
+
timestamp = datetime.now().strftime("%y-%m-%d_%H-%M")
|
|
152
|
+
base_dir = Path("converted")
|
|
153
|
+
output_dir = base_dir / req.original_file.stem / timestamp
|
|
154
|
+
ensure_directory(output_dir)
|
|
155
|
+
return output_dir
|
|
156
|
+
|
|
157
|
+
def combine_results(self, req: ConversionRequest) -> Tuple[Path, int]:
|
|
158
|
+
"""
|
|
159
|
+
Combine chunk results into a single output file.
|
|
160
|
+
|
|
161
|
+
Args:
|
|
162
|
+
req: Conversion request with chunks
|
|
163
|
+
|
|
164
|
+
Returns:
|
|
165
|
+
Tuple of (output_file_path, total_size)
|
|
166
|
+
|
|
167
|
+
Raises:
|
|
168
|
+
ResultProcessingError: If combination fails
|
|
169
|
+
"""
|
|
170
|
+
# Create output directory
|
|
171
|
+
output_dir = self.get_output_directory(req)
|
|
172
|
+
output_file = output_dir / f"{req.original_file.name}{self.format_extensions.get(req.output_format, '.txt')}"
|
|
173
|
+
|
|
174
|
+
# Ensure output directory exists
|
|
175
|
+
ensure_directory(output_file.parent)
|
|
176
|
+
|
|
177
|
+
try:
|
|
178
|
+
# Write chunks to output file one at a time to minimize memory usage
|
|
179
|
+
total_size = 0
|
|
180
|
+
with open(output_file, 'w', encoding='utf-8') as outf:
|
|
181
|
+
for i, chunk in enumerate(req.ordered_chunks):
|
|
182
|
+
result_path = chunk.get_result_path(req.tmp_dir)
|
|
183
|
+
if not result_path.exists():
|
|
184
|
+
raise ResultProcessingError(f"Result file does not exist: {result_path}")
|
|
185
|
+
|
|
186
|
+
# Read and write chunk content in blocks
|
|
187
|
+
with open(result_path, 'r', encoding='utf-8') as infile:
|
|
188
|
+
# Copy content in blocks
|
|
189
|
+
while True:
|
|
190
|
+
block = infile.read(65536) # Read 64KB at a time
|
|
191
|
+
if not block:
|
|
192
|
+
break
|
|
193
|
+
outf.write(block)
|
|
194
|
+
total_size += len(block)
|
|
195
|
+
|
|
196
|
+
# Add separator between chunks, but not after the last one
|
|
197
|
+
if i < len(req.ordered_chunks) - 1:
|
|
198
|
+
outf.write("\n\n")
|
|
199
|
+
total_size += 2
|
|
200
|
+
|
|
201
|
+
# Verify file was written successfully
|
|
202
|
+
if total_size == 0:
|
|
203
|
+
safe_delete(output_file) # Clean up empty file
|
|
204
|
+
raise ResultProcessingError("Combined content is empty")
|
|
205
|
+
|
|
206
|
+
return output_file, total_size
|
|
207
|
+
|
|
208
|
+
except Exception as e:
|
|
209
|
+
# Clean up partial file on error
|
|
210
|
+
if output_file.exists():
|
|
211
|
+
safe_delete(output_file)
|
|
212
|
+
|
|
213
|
+
if isinstance(e, ResultProcessingError):
|
|
214
|
+
raise
|
|
215
|
+
raise ResultProcessingError(f"Failed to combine results: {str(e)}")
|
|
216
|
+
|
|
217
|
+
def move_images(self, source_dir: Path, target_dir: Path) -> None:
|
|
218
|
+
"""
|
|
219
|
+
Move images from source to target directory.
|
|
220
|
+
|
|
221
|
+
Args:
|
|
222
|
+
source_dir: Source images directory
|
|
223
|
+
target_dir: Target images directory
|
|
224
|
+
|
|
225
|
+
Raises:
|
|
226
|
+
ResultProcessingError: If image move fails
|
|
227
|
+
"""
|
|
228
|
+
if not source_dir.exists():
|
|
229
|
+
return
|
|
230
|
+
|
|
231
|
+
try:
|
|
232
|
+
# Ensure target directory exists
|
|
233
|
+
target_images_dir = target_dir / "images"
|
|
234
|
+
if target_images_dir.exists():
|
|
235
|
+
safe_delete(target_images_dir)
|
|
236
|
+
|
|
237
|
+
# Copy images directory
|
|
238
|
+
try:
|
|
239
|
+
shutil.copytree(source_dir, target_images_dir)
|
|
240
|
+
safe_delete(source_dir) # Clean up after successful copy
|
|
241
|
+
logger.info(f"Moved images to {target_dir}/images/")
|
|
242
|
+
except Exception as e:
|
|
243
|
+
logger.warning(f"Could not copy images directory: {e}. Trying to copy files...")
|
|
244
|
+
ensure_directory(target_images_dir)
|
|
245
|
+
for img_file in source_dir.glob("*"):
|
|
246
|
+
try:
|
|
247
|
+
shutil.copy2(img_file, target_images_dir / img_file.name)
|
|
248
|
+
except Exception as copy_e:
|
|
249
|
+
logger.error(f"Failed to copy image {img_file}: {copy_e}")
|
|
250
|
+
except Exception as e:
|
|
251
|
+
raise ResultProcessingError(f"Failed to move images: {e}")
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
class ResultHandler:
|
|
255
|
+
"""Handles processing of conversion results."""
|
|
256
|
+
|
|
257
|
+
def __init__(self, api_key: str, cache_dir: Path, check_interval: int = 15):
|
|
258
|
+
"""
|
|
259
|
+
Initialize the result handler.
|
|
260
|
+
|
|
261
|
+
Args:
|
|
262
|
+
api_key: API key for authentication
|
|
263
|
+
cache_dir: Directory for cache
|
|
264
|
+
check_interval: Interval between API status checks
|
|
265
|
+
"""
|
|
266
|
+
self.client = MarkerClient(api_key)
|
|
267
|
+
self.cache = CacheManager(str(cache_dir))
|
|
268
|
+
self.check_interval = check_interval
|
|
269
|
+
self.saver = ResultSaver()
|
|
270
|
+
self.image_processor = ImageProcessor()
|
|
271
|
+
|
|
272
|
+
def process_cache_items(self) -> None:
|
|
273
|
+
"""
|
|
274
|
+
Process all pending items in the cache.
|
|
275
|
+
|
|
276
|
+
This is the main method to process results from the API.
|
|
277
|
+
"""
|
|
278
|
+
reqs = self.cache.get_all()
|
|
279
|
+
if not reqs:
|
|
280
|
+
return
|
|
281
|
+
|
|
282
|
+
progress = ProgressTracker(len(reqs), "Processing requests")
|
|
283
|
+
|
|
284
|
+
for req in reqs:
|
|
285
|
+
try:
|
|
286
|
+
# Skip already completed or failed requests
|
|
287
|
+
if req.status in (Status.FAILED, Status.COMPLETE):
|
|
288
|
+
self._cleanup_request(req)
|
|
289
|
+
progress.update()
|
|
290
|
+
continue
|
|
291
|
+
|
|
292
|
+
# Process any pending chunks
|
|
293
|
+
if pending := req.pending_chunks:
|
|
294
|
+
self._process_pending_chunks(req, pending)
|
|
295
|
+
|
|
296
|
+
# Try to combine results and cleanup if complete
|
|
297
|
+
if req.all_complete:
|
|
298
|
+
self._combine_and_save_result(req)
|
|
299
|
+
self._cleanup_request(req)
|
|
300
|
+
elif req.has_failed:
|
|
301
|
+
self._cleanup_request(req)
|
|
302
|
+
|
|
303
|
+
progress.update()
|
|
304
|
+
|
|
305
|
+
except Exception as e:
|
|
306
|
+
logger.error(f"Error processing request {req.request_id}: {e}")
|
|
307
|
+
req.set_status(Status.FAILED, str(e))
|
|
308
|
+
self._cleanup_request(req)
|
|
309
|
+
|
|
310
|
+
progress.close()
|
|
311
|
+
|
|
312
|
+
def _process_pending_chunks(self, req: ConversionRequest, chunks: List[ChunkInfo]) -> None:
|
|
313
|
+
"""
|
|
314
|
+
Process pending chunks.
|
|
315
|
+
|
|
316
|
+
Args:
|
|
317
|
+
req: Conversion request
|
|
318
|
+
chunks: List of pending chunks
|
|
319
|
+
"""
|
|
320
|
+
logger.info(f"Processing {len(chunks)} chunks for {req.original_file.name}")
|
|
321
|
+
|
|
322
|
+
progress = ProgressTracker(len(chunks), "Processing chunks")
|
|
323
|
+
|
|
324
|
+
for chunk in chunks:
|
|
325
|
+
if self._process_chunk(chunk, req):
|
|
326
|
+
req.set_status(Status.FAILED, chunk.error)
|
|
327
|
+
break
|
|
328
|
+
|
|
329
|
+
progress.update()
|
|
330
|
+
|
|
331
|
+
progress.close()
|
|
332
|
+
|
|
333
|
+
# Save updated request
|
|
334
|
+
self.cache.save(req)
|
|
335
|
+
|
|
336
|
+
def _process_chunk(self, chunk: ChunkInfo, req: ConversionRequest) -> bool:
|
|
337
|
+
"""
|
|
338
|
+
Process a single chunk.
|
|
339
|
+
|
|
340
|
+
Args:
|
|
341
|
+
chunk: Chunk to process
|
|
342
|
+
req: Parent conversion request
|
|
343
|
+
|
|
344
|
+
Returns:
|
|
345
|
+
True if processing should stop (failure), False otherwise
|
|
346
|
+
"""
|
|
347
|
+
if not req.tmp_dir:
|
|
348
|
+
chunk.mark_failed("No temporary directory set for request")
|
|
349
|
+
return True
|
|
350
|
+
|
|
351
|
+
max_retries = 20 # Maximum number of retries (5 minutes with 15 seconds interval)
|
|
352
|
+
retry_count = 0
|
|
353
|
+
|
|
354
|
+
while retry_count < max_retries:
|
|
355
|
+
status = self.client.check_status(chunk.request_id)
|
|
356
|
+
|
|
357
|
+
# Add None check before accessing status attributes
|
|
358
|
+
if status is None:
|
|
359
|
+
retry_count += 1
|
|
360
|
+
if retry_count >= max_retries:
|
|
361
|
+
chunk.mark_failed("Failed to retrieve status from API after multiple attempts")
|
|
362
|
+
return True
|
|
363
|
+
logger.warning(f"Received None status for chunk {chunk.request_id}, retrying...")
|
|
364
|
+
time.sleep(self.check_interval)
|
|
365
|
+
continue
|
|
366
|
+
|
|
367
|
+
match status.status:
|
|
368
|
+
case StatusEnum.FAILED:
|
|
369
|
+
chunk.mark_failed(status.error or "Unknown API error")
|
|
370
|
+
return True
|
|
371
|
+
|
|
372
|
+
case StatusEnum.COMPLETE:
|
|
373
|
+
try:
|
|
374
|
+
self._save_chunk_result(chunk, status, req)
|
|
375
|
+
return False
|
|
376
|
+
except Exception as e:
|
|
377
|
+
chunk.mark_failed(str(e))
|
|
378
|
+
return True
|
|
379
|
+
|
|
380
|
+
case _:
|
|
381
|
+
retry_count += 1
|
|
382
|
+
if retry_count >= max_retries:
|
|
383
|
+
chunk.mark_failed(f"API processing timed out after {max_retries * self.check_interval} seconds")
|
|
384
|
+
return True
|
|
385
|
+
time.sleep(self.check_interval)
|
|
386
|
+
|
|
387
|
+
# This line should never be reached due to the return inside the loop
|
|
388
|
+
chunk.mark_failed("Unexpected error processing chunk")
|
|
389
|
+
return True
|
|
390
|
+
|
|
391
|
+
def _save_chunk_result(self, chunk: ChunkInfo, status: MarkerStatus, req: ConversionRequest) -> None:
|
|
392
|
+
"""
|
|
393
|
+
Save chunk result to temporary storage.
|
|
394
|
+
|
|
395
|
+
Args:
|
|
396
|
+
chunk: Chunk being processed
|
|
397
|
+
status: API status response
|
|
398
|
+
req: Parent conversion request
|
|
399
|
+
|
|
400
|
+
Raises:
|
|
401
|
+
ResultProcessingError: If saving fails
|
|
402
|
+
"""
|
|
403
|
+
content = None
|
|
404
|
+
|
|
405
|
+
# Get content from appropriate field based on output format
|
|
406
|
+
if status.markdown is not None:
|
|
407
|
+
content = status.markdown
|
|
408
|
+
elif status.json_data is not None:
|
|
409
|
+
content = json.dumps(status.json_data)
|
|
410
|
+
|
|
411
|
+
if not content:
|
|
412
|
+
logger.error(f"No content found in result for chunk {chunk.path}")
|
|
413
|
+
raise ResultProcessingError("No content in result")
|
|
414
|
+
|
|
415
|
+
# Save content to temp file in request's tmp_dir
|
|
416
|
+
temp_file = chunk.get_result_path(req.tmp_dir)
|
|
417
|
+
|
|
418
|
+
# Process images if present
|
|
419
|
+
image_map = {}
|
|
420
|
+
if status.images:
|
|
421
|
+
image_map = self.image_processor.process_images(
|
|
422
|
+
status.images, chunk, req.tmp_dir, req.chunk_size
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
# Update image references in content
|
|
426
|
+
for original_name, new_ref in image_map.items():
|
|
427
|
+
content = content.replace(f"]({original_name})", f"]({new_ref})")
|
|
428
|
+
|
|
429
|
+
logger.info(f"Saving chunk result to {temp_file}")
|
|
430
|
+
self.saver.save_content(content, temp_file)
|
|
431
|
+
chunk.mark_complete()
|
|
432
|
+
|
|
433
|
+
def _combine_and_save_result(self, req: ConversionRequest) -> None:
|
|
434
|
+
"""
|
|
435
|
+
Combine results from all chunks and save final output.
|
|
436
|
+
|
|
437
|
+
Args:
|
|
438
|
+
req: Conversion request
|
|
439
|
+
|
|
440
|
+
Raises:
|
|
441
|
+
ResultProcessingError: If combination fails
|
|
442
|
+
"""
|
|
443
|
+
try:
|
|
444
|
+
# Combine chunks and get output file
|
|
445
|
+
output_file, total_size = self.saver.combine_results(req)
|
|
446
|
+
logger.info(f"Successfully saved output to {output_file} ({total_size} bytes)")
|
|
447
|
+
|
|
448
|
+
# Move images directory if it exists
|
|
449
|
+
images_dir = req.tmp_dir / "images"
|
|
450
|
+
if images_dir.exists():
|
|
451
|
+
self.saver.move_images(images_dir, output_file.parent)
|
|
452
|
+
|
|
453
|
+
req.set_status(Status.COMPLETE)
|
|
454
|
+
|
|
455
|
+
# Save final status to cache
|
|
456
|
+
self.cache.save(req)
|
|
457
|
+
|
|
458
|
+
except Exception as e:
|
|
459
|
+
error_msg = f"Failed to combine results: {str(e)}"
|
|
460
|
+
logger.error(error_msg)
|
|
461
|
+
req.set_status(Status.FAILED, error_msg)
|
|
462
|
+
self.cache.save(req)
|
|
463
|
+
raise
|
|
464
|
+
|
|
465
|
+
def _cleanup_request(self, req: ConversionRequest) -> None:
|
|
466
|
+
"""
|
|
467
|
+
Clean up resources for a request.
|
|
468
|
+
|
|
469
|
+
Args:
|
|
470
|
+
req: Request to clean up
|
|
471
|
+
"""
|
|
472
|
+
try:
|
|
473
|
+
# Clean up temp directory if it exists
|
|
474
|
+
if req.tmp_dir and Path(req.tmp_dir).exists():
|
|
475
|
+
safe_delete(req.tmp_dir)
|
|
476
|
+
|
|
477
|
+
# Remove from cache
|
|
478
|
+
self.cache.delete(req.request_id)
|
|
479
|
+
|
|
480
|
+
except Exception as e:
|
|
481
|
+
logger.error(f"Error cleaning up request {req.request_id}: {e}")
|
|
482
|
+
|
|
483
|
+
def close(self) -> None:
|
|
484
|
+
"""Close connections and free resources."""
|
|
485
|
+
try:
|
|
486
|
+
self.cache.close()
|
|
487
|
+
except Exception as e:
|
|
488
|
+
logger.error(f"Error closing cache: {e}")
|
|
489
|
+
|
|
490
|
+
def __enter__(self):
|
|
491
|
+
"""Support for context manager."""
|
|
492
|
+
return self
|
|
493
|
+
|
|
494
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
495
|
+
"""Clean up resources when exiting context."""
|
|
496
|
+
self.close()
|
docs_to_md/main.py
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Marker PDF to Markdown Converter
|
|
4
|
+
|
|
5
|
+
This script converts PDF files to markdown using the Marker API.
|
|
6
|
+
"""
|
|
7
|
+
import logging
|
|
8
|
+
import os
|
|
9
|
+
import sys
|
|
10
|
+
|
|
11
|
+
from docs_to_md.config.cli import create_config_from_args
|
|
12
|
+
from docs_to_md.core.processor import MarkerProcessor
|
|
13
|
+
from docs_to_md.utils.exceptions import ConfigurationError, DocsToMdError
|
|
14
|
+
from docs_to_md.utils.logging import setup_logging
|
|
15
|
+
|
|
16
|
+
# Setup logging
|
|
17
|
+
logger = logging.getLogger(__name__)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def main() -> int:
|
|
21
|
+
"""
|
|
22
|
+
Main entry point for the application.
|
|
23
|
+
|
|
24
|
+
Returns:
|
|
25
|
+
Exit code (0 for success, non-zero for failure)
|
|
26
|
+
"""
|
|
27
|
+
# Initialize logging configuration first
|
|
28
|
+
setup_logging()
|
|
29
|
+
|
|
30
|
+
try:
|
|
31
|
+
# Create configuration from command-line arguments
|
|
32
|
+
config = create_config_from_args()
|
|
33
|
+
|
|
34
|
+
# Create processor and run
|
|
35
|
+
processor = MarkerProcessor(config)
|
|
36
|
+
processor.process()
|
|
37
|
+
|
|
38
|
+
logger.info("Conversion completed successfully.")
|
|
39
|
+
return 0
|
|
40
|
+
|
|
41
|
+
except ConfigurationError as e:
|
|
42
|
+
logger.error(f"Configuration error: {e}")
|
|
43
|
+
return 1
|
|
44
|
+
|
|
45
|
+
except DocsToMdError as e:
|
|
46
|
+
logger.error(f"Processing error: {e}")
|
|
47
|
+
return 2
|
|
48
|
+
|
|
49
|
+
except Exception as e:
|
|
50
|
+
logger.error(f"Unexpected error: {e}")
|
|
51
|
+
return 3
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
if __name__ == "__main__":
|
|
55
|
+
sys.exit(main())
|
|
File without changes
|