pdf-to-markdown-cli 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,496 @@
1
+ import base64
2
+ import json
3
+ import logging
4
+ import re
5
+ import shutil
6
+ import time
7
+ import uuid
8
+ from datetime import datetime
9
+ from pathlib import Path
10
+ from typing import Dict, List, Optional, Tuple
11
+ from concurrent.futures import ThreadPoolExecutor, as_completed
12
+
13
+ from docs_to_md.api.client import MarkerClient
14
+ from docs_to_md.api.models import MarkerStatus, StatusEnum
15
+ from docs_to_md.storage.cache import CacheManager
16
+ from docs_to_md.storage.models import ChunkInfo, ConversionRequest, Status
17
+ from docs_to_md.utils.exceptions import ResultProcessingError
18
+ from docs_to_md.utils.file_utils import FileIO, ensure_directory, safe_delete
19
+ from docs_to_md.utils.logging import ProgressTracker
20
+
21
+ logger = logging.getLogger(__name__)
22
+
23
+
24
+ class ImageProcessor:
25
+ """Handles processing and transformation of images extracted from documents."""
26
+
27
+ def transform_image_name(self, original_name: str, chunk: ChunkInfo,
28
+ chunk_size: int) -> Tuple[str, str]:
29
+ """
30
+ Transform image name and return new names for file and markdown.
31
+
32
+ Args:
33
+ original_name: Original image name from API
34
+ chunk: Chunk info for context
35
+ chunk_size: Number of pages per chunk
36
+
37
+ Returns:
38
+ Tuple of (new_filename, markdown_reference_name)
39
+ """
40
+ base_page_num = (chunk.index * chunk_size) + 1
41
+
42
+ # Extract file extension properly
43
+ parts = original_name.split('.')
44
+ extension = ""
45
+ if len(parts) > 1:
46
+ extension = f".{parts[-1].lower()}"
47
+ # Validate extension is a common image type
48
+ if extension not in ['.jpg', '.jpeg', '.png', '.gif', '.webp', '.tiff']:
49
+ extension = ".jpg" # Default to jpg if unknown
50
+ else:
51
+ extension = ".jpg" # Default extension
52
+
53
+ # Try to extract page and figure numbers with more robust patterns
54
+ page_match = re.search(r'(?:_|-)page(?:_|-)?(\d+)', original_name, re.IGNORECASE)
55
+ figure_match = re.search(r'(?:_|-)(?:figure|fig)(?:_|-)?(\d+)', original_name, re.IGNORECASE)
56
+
57
+ if page_match and figure_match:
58
+ # Calculate new page number based on chunk position
59
+ page_num = int(page_match.group(1))
60
+ corrected_page_num = base_page_num + page_num - 1 # -1 because pages often start at 1
61
+ figure_num = int(figure_match.group(1))
62
+
63
+ # Generate new filename with corrected page number
64
+ new_name = f"page_{corrected_page_num}_figure_{figure_num}{extension}"
65
+ return new_name, new_name
66
+ else:
67
+ # Fallback: use chunk number, timestamp, and random identifier for uniqueness
68
+ timestamp = datetime.now().strftime("%H%M%S")
69
+ random_suffix = uuid.uuid4().hex[:6]
70
+ fallback_name = f"chunk_{chunk.index}_img_{timestamp}_{random_suffix}{extension}"
71
+ return fallback_name, fallback_name
72
+
73
+ def process_images(self, images: Dict[str, str], chunk: ChunkInfo,
74
+ tmp_dir: Path, chunk_size: int) -> Dict[str, str]:
75
+ """
76
+ Process images from API response.
77
+
78
+ Args:
79
+ images: Dictionary of image names to base64 content
80
+ chunk: Chunk info for context
81
+ tmp_dir: Temporary directory for storage
82
+ chunk_size: Number of pages per chunk
83
+
84
+ Returns:
85
+ Dictionary mapping original image names to new image references
86
+ """
87
+ if not images:
88
+ return {}
89
+
90
+ # Prepare images directory
91
+ images_dir = tmp_dir / "images"
92
+ ensure_directory(images_dir)
93
+
94
+ # Process each image
95
+ image_map = {}
96
+ for original_name, b64_content in images.items():
97
+ # Transform name
98
+ new_name, markdown_name = self.transform_image_name(original_name, chunk, chunk_size)
99
+
100
+ # Save image
101
+ try:
102
+ image_data = base64.b64decode(b64_content)
103
+ (images_dir / new_name).write_bytes(image_data)
104
+
105
+ # Map original name to new markdown reference
106
+ image_map[original_name] = f"images/{markdown_name}"
107
+ except Exception as e:
108
+ logger.error(f"Failed to save image {original_name} to {new_name}: {e}")
109
+
110
+ return image_map
111
+
112
+
113
+ class ResultSaver:
114
+ """Handles saving and combining results."""
115
+
116
+ def __init__(self):
117
+ """Initialize the result saver."""
118
+ self.format_extensions = {
119
+ "markdown": ".md",
120
+ "json": ".json",
121
+ "html": ".html",
122
+ "txt": ".txt" # default
123
+ }
124
+
125
+ def save_content(self, content: str, path: Path) -> None:
126
+ """
127
+ Save content to file, ensuring parent directory exists.
128
+
129
+ Args:
130
+ content: Content to save
131
+ path: Path to save to
132
+
133
+ Raises:
134
+ ResultProcessingError: If save fails
135
+ """
136
+ try:
137
+ FileIO.write_file(path, content)
138
+ except Exception as e:
139
+ raise ResultProcessingError(f"Failed to save content to {path}: {e}")
140
+
141
+ def get_output_directory(self, req: ConversionRequest) -> Path:
142
+ """
143
+ Create and return output directory with timestamp.
144
+
145
+ Args:
146
+ req: Conversion request
147
+
148
+ Returns:
149
+ Path to output directory
150
+ """
151
+ timestamp = datetime.now().strftime("%y-%m-%d_%H-%M")
152
+ base_dir = Path("converted")
153
+ output_dir = base_dir / req.original_file.stem / timestamp
154
+ ensure_directory(output_dir)
155
+ return output_dir
156
+
157
+ def combine_results(self, req: ConversionRequest) -> Tuple[Path, int]:
158
+ """
159
+ Combine chunk results into a single output file.
160
+
161
+ Args:
162
+ req: Conversion request with chunks
163
+
164
+ Returns:
165
+ Tuple of (output_file_path, total_size)
166
+
167
+ Raises:
168
+ ResultProcessingError: If combination fails
169
+ """
170
+ # Create output directory
171
+ output_dir = self.get_output_directory(req)
172
+ output_file = output_dir / f"{req.original_file.name}{self.format_extensions.get(req.output_format, '.txt')}"
173
+
174
+ # Ensure output directory exists
175
+ ensure_directory(output_file.parent)
176
+
177
+ try:
178
+ # Write chunks to output file one at a time to minimize memory usage
179
+ total_size = 0
180
+ with open(output_file, 'w', encoding='utf-8') as outf:
181
+ for i, chunk in enumerate(req.ordered_chunks):
182
+ result_path = chunk.get_result_path(req.tmp_dir)
183
+ if not result_path.exists():
184
+ raise ResultProcessingError(f"Result file does not exist: {result_path}")
185
+
186
+ # Read and write chunk content in blocks
187
+ with open(result_path, 'r', encoding='utf-8') as infile:
188
+ # Copy content in blocks
189
+ while True:
190
+ block = infile.read(65536) # Read 64KB at a time
191
+ if not block:
192
+ break
193
+ outf.write(block)
194
+ total_size += len(block)
195
+
196
+ # Add separator between chunks, but not after the last one
197
+ if i < len(req.ordered_chunks) - 1:
198
+ outf.write("\n\n")
199
+ total_size += 2
200
+
201
+ # Verify file was written successfully
202
+ if total_size == 0:
203
+ safe_delete(output_file) # Clean up empty file
204
+ raise ResultProcessingError("Combined content is empty")
205
+
206
+ return output_file, total_size
207
+
208
+ except Exception as e:
209
+ # Clean up partial file on error
210
+ if output_file.exists():
211
+ safe_delete(output_file)
212
+
213
+ if isinstance(e, ResultProcessingError):
214
+ raise
215
+ raise ResultProcessingError(f"Failed to combine results: {str(e)}")
216
+
217
+ def move_images(self, source_dir: Path, target_dir: Path) -> None:
218
+ """
219
+ Move images from source to target directory.
220
+
221
+ Args:
222
+ source_dir: Source images directory
223
+ target_dir: Target images directory
224
+
225
+ Raises:
226
+ ResultProcessingError: If image move fails
227
+ """
228
+ if not source_dir.exists():
229
+ return
230
+
231
+ try:
232
+ # Ensure target directory exists
233
+ target_images_dir = target_dir / "images"
234
+ if target_images_dir.exists():
235
+ safe_delete(target_images_dir)
236
+
237
+ # Copy images directory
238
+ try:
239
+ shutil.copytree(source_dir, target_images_dir)
240
+ safe_delete(source_dir) # Clean up after successful copy
241
+ logger.info(f"Moved images to {target_dir}/images/")
242
+ except Exception as e:
243
+ logger.warning(f"Could not copy images directory: {e}. Trying to copy files...")
244
+ ensure_directory(target_images_dir)
245
+ for img_file in source_dir.glob("*"):
246
+ try:
247
+ shutil.copy2(img_file, target_images_dir / img_file.name)
248
+ except Exception as copy_e:
249
+ logger.error(f"Failed to copy image {img_file}: {copy_e}")
250
+ except Exception as e:
251
+ raise ResultProcessingError(f"Failed to move images: {e}")
252
+
253
+
254
+ class ResultHandler:
255
+ """Handles processing of conversion results."""
256
+
257
+ def __init__(self, api_key: str, cache_dir: Path, check_interval: int = 15):
258
+ """
259
+ Initialize the result handler.
260
+
261
+ Args:
262
+ api_key: API key for authentication
263
+ cache_dir: Directory for cache
264
+ check_interval: Interval between API status checks
265
+ """
266
+ self.client = MarkerClient(api_key)
267
+ self.cache = CacheManager(str(cache_dir))
268
+ self.check_interval = check_interval
269
+ self.saver = ResultSaver()
270
+ self.image_processor = ImageProcessor()
271
+
272
+ def process_cache_items(self) -> None:
273
+ """
274
+ Process all pending items in the cache.
275
+
276
+ This is the main method to process results from the API.
277
+ """
278
+ reqs = self.cache.get_all()
279
+ if not reqs:
280
+ return
281
+
282
+ progress = ProgressTracker(len(reqs), "Processing requests")
283
+
284
+ for req in reqs:
285
+ try:
286
+ # Skip already completed or failed requests
287
+ if req.status in (Status.FAILED, Status.COMPLETE):
288
+ self._cleanup_request(req)
289
+ progress.update()
290
+ continue
291
+
292
+ # Process any pending chunks
293
+ if pending := req.pending_chunks:
294
+ self._process_pending_chunks(req, pending)
295
+
296
+ # Try to combine results and cleanup if complete
297
+ if req.all_complete:
298
+ self._combine_and_save_result(req)
299
+ self._cleanup_request(req)
300
+ elif req.has_failed:
301
+ self._cleanup_request(req)
302
+
303
+ progress.update()
304
+
305
+ except Exception as e:
306
+ logger.error(f"Error processing request {req.request_id}: {e}")
307
+ req.set_status(Status.FAILED, str(e))
308
+ self._cleanup_request(req)
309
+
310
+ progress.close()
311
+
312
+ def _process_pending_chunks(self, req: ConversionRequest, chunks: List[ChunkInfo]) -> None:
313
+ """
314
+ Process pending chunks.
315
+
316
+ Args:
317
+ req: Conversion request
318
+ chunks: List of pending chunks
319
+ """
320
+ logger.info(f"Processing {len(chunks)} chunks for {req.original_file.name}")
321
+
322
+ progress = ProgressTracker(len(chunks), "Processing chunks")
323
+
324
+ for chunk in chunks:
325
+ if self._process_chunk(chunk, req):
326
+ req.set_status(Status.FAILED, chunk.error)
327
+ break
328
+
329
+ progress.update()
330
+
331
+ progress.close()
332
+
333
+ # Save updated request
334
+ self.cache.save(req)
335
+
336
+ def _process_chunk(self, chunk: ChunkInfo, req: ConversionRequest) -> bool:
337
+ """
338
+ Process a single chunk.
339
+
340
+ Args:
341
+ chunk: Chunk to process
342
+ req: Parent conversion request
343
+
344
+ Returns:
345
+ True if processing should stop (failure), False otherwise
346
+ """
347
+ if not req.tmp_dir:
348
+ chunk.mark_failed("No temporary directory set for request")
349
+ return True
350
+
351
+ max_retries = 20 # Maximum number of retries (5 minutes with 15 seconds interval)
352
+ retry_count = 0
353
+
354
+ while retry_count < max_retries:
355
+ status = self.client.check_status(chunk.request_id)
356
+
357
+ # Add None check before accessing status attributes
358
+ if status is None:
359
+ retry_count += 1
360
+ if retry_count >= max_retries:
361
+ chunk.mark_failed("Failed to retrieve status from API after multiple attempts")
362
+ return True
363
+ logger.warning(f"Received None status for chunk {chunk.request_id}, retrying...")
364
+ time.sleep(self.check_interval)
365
+ continue
366
+
367
+ match status.status:
368
+ case StatusEnum.FAILED:
369
+ chunk.mark_failed(status.error or "Unknown API error")
370
+ return True
371
+
372
+ case StatusEnum.COMPLETE:
373
+ try:
374
+ self._save_chunk_result(chunk, status, req)
375
+ return False
376
+ except Exception as e:
377
+ chunk.mark_failed(str(e))
378
+ return True
379
+
380
+ case _:
381
+ retry_count += 1
382
+ if retry_count >= max_retries:
383
+ chunk.mark_failed(f"API processing timed out after {max_retries * self.check_interval} seconds")
384
+ return True
385
+ time.sleep(self.check_interval)
386
+
387
+ # This line should never be reached due to the return inside the loop
388
+ chunk.mark_failed("Unexpected error processing chunk")
389
+ return True
390
+
391
+ def _save_chunk_result(self, chunk: ChunkInfo, status: MarkerStatus, req: ConversionRequest) -> None:
392
+ """
393
+ Save chunk result to temporary storage.
394
+
395
+ Args:
396
+ chunk: Chunk being processed
397
+ status: API status response
398
+ req: Parent conversion request
399
+
400
+ Raises:
401
+ ResultProcessingError: If saving fails
402
+ """
403
+ content = None
404
+
405
+ # Get content from appropriate field based on output format
406
+ if status.markdown is not None:
407
+ content = status.markdown
408
+ elif status.json_data is not None:
409
+ content = json.dumps(status.json_data)
410
+
411
+ if not content:
412
+ logger.error(f"No content found in result for chunk {chunk.path}")
413
+ raise ResultProcessingError("No content in result")
414
+
415
+ # Save content to temp file in request's tmp_dir
416
+ temp_file = chunk.get_result_path(req.tmp_dir)
417
+
418
+ # Process images if present
419
+ image_map = {}
420
+ if status.images:
421
+ image_map = self.image_processor.process_images(
422
+ status.images, chunk, req.tmp_dir, req.chunk_size
423
+ )
424
+
425
+ # Update image references in content
426
+ for original_name, new_ref in image_map.items():
427
+ content = content.replace(f"]({original_name})", f"]({new_ref})")
428
+
429
+ logger.info(f"Saving chunk result to {temp_file}")
430
+ self.saver.save_content(content, temp_file)
431
+ chunk.mark_complete()
432
+
433
+ def _combine_and_save_result(self, req: ConversionRequest) -> None:
434
+ """
435
+ Combine results from all chunks and save final output.
436
+
437
+ Args:
438
+ req: Conversion request
439
+
440
+ Raises:
441
+ ResultProcessingError: If combination fails
442
+ """
443
+ try:
444
+ # Combine chunks and get output file
445
+ output_file, total_size = self.saver.combine_results(req)
446
+ logger.info(f"Successfully saved output to {output_file} ({total_size} bytes)")
447
+
448
+ # Move images directory if it exists
449
+ images_dir = req.tmp_dir / "images"
450
+ if images_dir.exists():
451
+ self.saver.move_images(images_dir, output_file.parent)
452
+
453
+ req.set_status(Status.COMPLETE)
454
+
455
+ # Save final status to cache
456
+ self.cache.save(req)
457
+
458
+ except Exception as e:
459
+ error_msg = f"Failed to combine results: {str(e)}"
460
+ logger.error(error_msg)
461
+ req.set_status(Status.FAILED, error_msg)
462
+ self.cache.save(req)
463
+ raise
464
+
465
+ def _cleanup_request(self, req: ConversionRequest) -> None:
466
+ """
467
+ Clean up resources for a request.
468
+
469
+ Args:
470
+ req: Request to clean up
471
+ """
472
+ try:
473
+ # Clean up temp directory if it exists
474
+ if req.tmp_dir and Path(req.tmp_dir).exists():
475
+ safe_delete(req.tmp_dir)
476
+
477
+ # Remove from cache
478
+ self.cache.delete(req.request_id)
479
+
480
+ except Exception as e:
481
+ logger.error(f"Error cleaning up request {req.request_id}: {e}")
482
+
483
+ def close(self) -> None:
484
+ """Close connections and free resources."""
485
+ try:
486
+ self.cache.close()
487
+ except Exception as e:
488
+ logger.error(f"Error closing cache: {e}")
489
+
490
+ def __enter__(self):
491
+ """Support for context manager."""
492
+ return self
493
+
494
+ def __exit__(self, exc_type, exc_val, exc_tb):
495
+ """Clean up resources when exiting context."""
496
+ self.close()
docs_to_md/main.py ADDED
@@ -0,0 +1,55 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Marker PDF to Markdown Converter
4
+
5
+ This script converts PDF files to markdown using the Marker API.
6
+ """
7
+ import logging
8
+ import os
9
+ import sys
10
+
11
+ from docs_to_md.config.cli import create_config_from_args
12
+ from docs_to_md.core.processor import MarkerProcessor
13
+ from docs_to_md.utils.exceptions import ConfigurationError, DocsToMdError
14
+ from docs_to_md.utils.logging import setup_logging
15
+
16
+ # Setup logging
17
+ logger = logging.getLogger(__name__)
18
+
19
+
20
+ def main() -> int:
21
+ """
22
+ Main entry point for the application.
23
+
24
+ Returns:
25
+ Exit code (0 for success, non-zero for failure)
26
+ """
27
+ # Initialize logging configuration first
28
+ setup_logging()
29
+
30
+ try:
31
+ # Create configuration from command-line arguments
32
+ config = create_config_from_args()
33
+
34
+ # Create processor and run
35
+ processor = MarkerProcessor(config)
36
+ processor.process()
37
+
38
+ logger.info("Conversion completed successfully.")
39
+ return 0
40
+
41
+ except ConfigurationError as e:
42
+ logger.error(f"Configuration error: {e}")
43
+ return 1
44
+
45
+ except DocsToMdError as e:
46
+ logger.error(f"Processing error: {e}")
47
+ return 2
48
+
49
+ except Exception as e:
50
+ logger.error(f"Unexpected error: {e}")
51
+ return 3
52
+
53
+
54
+ if __name__ == "__main__":
55
+ sys.exit(main())
File without changes