pdf-to-markdown-cli 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docs_to_md/__init__.py +0 -0
- docs_to_md/__main__.py +5 -0
- docs_to_md/api/__init__.py +0 -0
- docs_to_md/api/client.py +218 -0
- docs_to_md/api/models.py +66 -0
- docs_to_md/config/__init__.py +0 -0
- docs_to_md/config/cli.py +86 -0
- docs_to_md/config/settings.py +108 -0
- docs_to_md/core/__init__.py +0 -0
- docs_to_md/core/processor.py +229 -0
- docs_to_md/core/result_handler.py +496 -0
- docs_to_md/main.py +55 -0
- docs_to_md/pdf/__init__.py +0 -0
- docs_to_md/pdf/splitter.py +199 -0
- docs_to_md/storage/__init__.py +0 -0
- docs_to_md/storage/cache.py +125 -0
- docs_to_md/storage/models.py +85 -0
- docs_to_md/utils/__init__.py +0 -0
- docs_to_md/utils/exceptions.py +33 -0
- docs_to_md/utils/file_utils.py +189 -0
- docs_to_md/utils/logging.py +134 -0
- pdf_to_markdown_cli-0.2.0.dist-info/METADATA +179 -0
- pdf_to_markdown_cli-0.2.0.dist-info/RECORD +27 -0
- pdf_to_markdown_cli-0.2.0.dist-info/WHEEL +5 -0
- pdf_to_markdown_cli-0.2.0.dist-info/entry_points.txt +2 -0
- pdf_to_markdown_cli-0.2.0.dist-info/licenses/LICENSE +21 -0
- pdf_to_markdown_cli-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import uuid
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from docs_to_md.api.client import MarkerClient
|
|
6
|
+
from docs_to_md.api.models import SUPPORTED_MIME_TYPES, ApiParams
|
|
7
|
+
from docs_to_md.config.settings import Config
|
|
8
|
+
from docs_to_md.pdf.splitter import chunk_pdf_to_temp
|
|
9
|
+
from docs_to_md.storage.cache import CacheManager
|
|
10
|
+
from docs_to_md.storage.models import ConversionRequest, Status
|
|
11
|
+
from docs_to_md.utils.exceptions import FileError, PDFProcessingError
|
|
12
|
+
from docs_to_md.utils.file_utils import FileDiscovery, TemporaryDirectory, ensure_directory
|
|
13
|
+
from docs_to_md.utils.logging import ProgressTracker
|
|
14
|
+
from docs_to_md.core.result_handler import ResultHandler
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class BatchProcessor:
|
|
20
|
+
"""Handles processing of files, including chunking if needed."""
|
|
21
|
+
|
|
22
|
+
def __init__(self, api_key: str, cache_dir: Path, chunk_size: int = 25):
|
|
23
|
+
"""
|
|
24
|
+
Initialize the batch processor.
|
|
25
|
+
|
|
26
|
+
Args:
|
|
27
|
+
api_key: API key for authentication
|
|
28
|
+
cache_dir: Directory for caching
|
|
29
|
+
chunk_size: Pages per chunk for PDFs
|
|
30
|
+
"""
|
|
31
|
+
self.client = MarkerClient(api_key)
|
|
32
|
+
self.cache = CacheManager(str(cache_dir))
|
|
33
|
+
self.chunk_size = chunk_size # pages per chunk for PDFs
|
|
34
|
+
|
|
35
|
+
def should_chunk(self, file_path: Path) -> bool:
|
|
36
|
+
"""
|
|
37
|
+
Determine if a file should be chunked.
|
|
38
|
+
|
|
39
|
+
Args:
|
|
40
|
+
file_path: Path to check
|
|
41
|
+
|
|
42
|
+
Returns:
|
|
43
|
+
True if file should be chunked, False otherwise
|
|
44
|
+
"""
|
|
45
|
+
return file_path.suffix.lower() == '.pdf'
|
|
46
|
+
|
|
47
|
+
def process_file(
|
|
48
|
+
self,
|
|
49
|
+
file_path: Path,
|
|
50
|
+
output_dir: Path,
|
|
51
|
+
api_params: ApiParams
|
|
52
|
+
) -> str:
|
|
53
|
+
"""
|
|
54
|
+
Process a single file.
|
|
55
|
+
|
|
56
|
+
Args:
|
|
57
|
+
file_path: Path to file
|
|
58
|
+
output_dir: Output directory
|
|
59
|
+
api_params: Parameters for the API call
|
|
60
|
+
|
|
61
|
+
Returns:
|
|
62
|
+
Request ID for tracking
|
|
63
|
+
|
|
64
|
+
Raises:
|
|
65
|
+
FileError: If file processing fails
|
|
66
|
+
"""
|
|
67
|
+
# Create temp directory for this conversion
|
|
68
|
+
with TemporaryDirectory(Path("chunks"), file_path.stem) as tmp_dir:
|
|
69
|
+
ensure_directory(output_dir)
|
|
70
|
+
|
|
71
|
+
# Initialize request
|
|
72
|
+
request = ConversionRequest(
|
|
73
|
+
request_id=str(uuid.uuid4()),
|
|
74
|
+
original_file=file_path,
|
|
75
|
+
target_file=output_dir / file_path.name,
|
|
76
|
+
output_format=api_params.output_format,
|
|
77
|
+
status=Status.PENDING,
|
|
78
|
+
tmp_dir=tmp_dir,
|
|
79
|
+
chunk_size=self.chunk_size
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
# Save request early for tracking
|
|
83
|
+
self.cache.save(request)
|
|
84
|
+
|
|
85
|
+
try:
|
|
86
|
+
# Handle PDF chunking
|
|
87
|
+
if self.should_chunk(file_path):
|
|
88
|
+
try:
|
|
89
|
+
chunk_result = chunk_pdf_to_temp(str(file_path), self.chunk_size, tmp_dir)
|
|
90
|
+
if chunk_result:
|
|
91
|
+
# Add additional chunks
|
|
92
|
+
for chunk_info in chunk_result.chunks:
|
|
93
|
+
request.add_chunk(Path(chunk_info.path), chunk_info.index)
|
|
94
|
+
except (PDFProcessingError, Exception) as e:
|
|
95
|
+
logger.error(f"Error chunking PDF {file_path}: {e}")
|
|
96
|
+
request.set_status(Status.FAILED, f"Error chunking PDF: {str(e)}")
|
|
97
|
+
self.cache.save(request)
|
|
98
|
+
return request.request_id
|
|
99
|
+
|
|
100
|
+
if len(request.chunks) == 0: # If no chunks created, add original file
|
|
101
|
+
request.add_chunk(file_path, 0)
|
|
102
|
+
|
|
103
|
+
# Submit all chunks to API
|
|
104
|
+
logger.info(f"Submitting {len(request.chunks)} chunk(s) to API...")
|
|
105
|
+
|
|
106
|
+
progress = ProgressTracker(len(request.chunks), "Submitting to API", "chunk")
|
|
107
|
+
|
|
108
|
+
for chunk in request.ordered_chunks:
|
|
109
|
+
chunk_request_id = self.client.submit_file(
|
|
110
|
+
chunk.path,
|
|
111
|
+
output_format=api_params.output_format,
|
|
112
|
+
langs=api_params.langs,
|
|
113
|
+
use_llm=api_params.use_llm,
|
|
114
|
+
strip_existing_ocr=api_params.strip_existing_ocr,
|
|
115
|
+
disable_image_extraction=api_params.disable_image_extraction,
|
|
116
|
+
force_ocr=api_params.force_ocr,
|
|
117
|
+
paginate=api_params.paginate,
|
|
118
|
+
max_pages=api_params.max_pages
|
|
119
|
+
)
|
|
120
|
+
if chunk_request_id:
|
|
121
|
+
chunk.mark_processing(chunk_request_id)
|
|
122
|
+
else:
|
|
123
|
+
chunk.mark_failed(f"Failed to submit file {chunk.path}")
|
|
124
|
+
break
|
|
125
|
+
|
|
126
|
+
progress.update()
|
|
127
|
+
|
|
128
|
+
progress.close()
|
|
129
|
+
|
|
130
|
+
# Update request status
|
|
131
|
+
if request.has_failed:
|
|
132
|
+
request.set_status(Status.FAILED)
|
|
133
|
+
else:
|
|
134
|
+
request.status = Status.PROCESSING
|
|
135
|
+
|
|
136
|
+
self.cache.save(request)
|
|
137
|
+
return request.request_id
|
|
138
|
+
|
|
139
|
+
except Exception as e:
|
|
140
|
+
request.status = Status.FAILED
|
|
141
|
+
request.error = str(e)
|
|
142
|
+
self.cache.save(request)
|
|
143
|
+
logger.error(f"Error processing file {file_path}: {e}")
|
|
144
|
+
return request.request_id
|
|
145
|
+
|
|
146
|
+
def close(self) -> None:
|
|
147
|
+
"""Clean up resources."""
|
|
148
|
+
self.cache.close()
|
|
149
|
+
|
|
150
|
+
def __enter__(self):
|
|
151
|
+
"""Support for context manager."""
|
|
152
|
+
return self
|
|
153
|
+
|
|
154
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
155
|
+
"""Clean up resources when exiting context."""
|
|
156
|
+
self.close()
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
class MarkerProcessor:
|
|
160
|
+
"""Handles the core business logic for processing PDFs."""
|
|
161
|
+
|
|
162
|
+
def __init__(self, config: Config):
|
|
163
|
+
"""
|
|
164
|
+
Initialize the processor with configuration.
|
|
165
|
+
|
|
166
|
+
Args:
|
|
167
|
+
config: Application configuration
|
|
168
|
+
"""
|
|
169
|
+
self.config = config
|
|
170
|
+
self.config.ensure_directories()
|
|
171
|
+
|
|
172
|
+
def process(self) -> None:
|
|
173
|
+
"""
|
|
174
|
+
Process files according to configuration.
|
|
175
|
+
|
|
176
|
+
Raises:
|
|
177
|
+
FileError: If file processing fails
|
|
178
|
+
"""
|
|
179
|
+
# Initialize processors
|
|
180
|
+
with BatchProcessor(self.config.api_key, self.config.cache_dir, chunk_size=self.config.chunk_size) as batch_processor, \
|
|
181
|
+
ResultHandler(self.config.api_key, self.config.cache_dir) as result_handler:
|
|
182
|
+
|
|
183
|
+
try:
|
|
184
|
+
# Process input path
|
|
185
|
+
input_path = Path(self.config.input_path)
|
|
186
|
+
|
|
187
|
+
# Find processable files
|
|
188
|
+
supported_extensions = ['.pdf', '.docx', '.doc', '.pptx', '.ppt',
|
|
189
|
+
'.jpg', '.jpeg', '.png', '.gif', '.webp', '.tiff']
|
|
190
|
+
|
|
191
|
+
files_to_process = FileDiscovery.find_processable_files(
|
|
192
|
+
input_path,
|
|
193
|
+
SUPPORTED_MIME_TYPES,
|
|
194
|
+
supported_extensions
|
|
195
|
+
)
|
|
196
|
+
|
|
197
|
+
if not files_to_process:
|
|
198
|
+
logger.warning(f"No processable files found in {input_path}")
|
|
199
|
+
return
|
|
200
|
+
|
|
201
|
+
# Create ApiParams from config
|
|
202
|
+
api_params = ApiParams(
|
|
203
|
+
output_format=self.config.output_format,
|
|
204
|
+
langs=self.config.langs,
|
|
205
|
+
use_llm=self.config.use_llm,
|
|
206
|
+
strip_existing_ocr=self.config.strip_existing_ocr,
|
|
207
|
+
disable_image_extraction=self.config.disable_image_extraction,
|
|
208
|
+
force_ocr=self.config.force_ocr,
|
|
209
|
+
paginate=self.config.paginate,
|
|
210
|
+
max_pages=self.config.max_pages
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
# Process each file
|
|
214
|
+
for file_path in files_to_process:
|
|
215
|
+
logger.info(f"Processing {file_path}")
|
|
216
|
+
batch_processor.process_file(
|
|
217
|
+
file_path=file_path,
|
|
218
|
+
output_dir=self.config.output_dir,
|
|
219
|
+
api_params=api_params
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
# Process results
|
|
223
|
+
result_handler.process_cache_items()
|
|
224
|
+
|
|
225
|
+
logger.info("All processing completed successfully.")
|
|
226
|
+
|
|
227
|
+
except Exception as e:
|
|
228
|
+
logger.error(f"Error during processing: {e}")
|
|
229
|
+
raise FileError(f"Processing failed: {e}")
|