pdf-to-markdown-cli 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docs_to_md/__init__.py +0 -0
- docs_to_md/__main__.py +5 -0
- docs_to_md/api/__init__.py +0 -0
- docs_to_md/api/client.py +218 -0
- docs_to_md/api/models.py +66 -0
- docs_to_md/config/__init__.py +0 -0
- docs_to_md/config/cli.py +86 -0
- docs_to_md/config/settings.py +108 -0
- docs_to_md/core/__init__.py +0 -0
- docs_to_md/core/processor.py +229 -0
- docs_to_md/core/result_handler.py +496 -0
- docs_to_md/main.py +55 -0
- docs_to_md/pdf/__init__.py +0 -0
- docs_to_md/pdf/splitter.py +199 -0
- docs_to_md/storage/__init__.py +0 -0
- docs_to_md/storage/cache.py +125 -0
- docs_to_md/storage/models.py +85 -0
- docs_to_md/utils/__init__.py +0 -0
- docs_to_md/utils/exceptions.py +33 -0
- docs_to_md/utils/file_utils.py +189 -0
- docs_to_md/utils/logging.py +134 -0
- pdf_to_markdown_cli-0.2.0.dist-info/METADATA +179 -0
- pdf_to_markdown_cli-0.2.0.dist-info/RECORD +27 -0
- pdf_to_markdown_cli-0.2.0.dist-info/WHEEL +5 -0
- pdf_to_markdown_cli-0.2.0.dist-info/entry_points.txt +2 -0
- pdf_to_markdown_cli-0.2.0.dist-info/licenses/LICENSE +21 -0
- pdf_to_markdown_cli-0.2.0.dist-info/top_level.txt +1 -0
docs_to_md/__init__.py
ADDED
|
File without changes
|
docs_to_md/__main__.py
ADDED
|
File without changes
|
docs_to_md/api/client.py
ADDED
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import logging
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Dict, Optional, Any
|
|
5
|
+
|
|
6
|
+
import backoff
|
|
7
|
+
import filetype
|
|
8
|
+
import requests
|
|
9
|
+
from ratelimit import limits, sleep_and_retry
|
|
10
|
+
|
|
11
|
+
from docs_to_md.api.models import MarkerStatus, StatusEnum, SubmitResponse, SUPPORTED_MIME_TYPES
|
|
12
|
+
from docs_to_md.utils.exceptions import APIError
|
|
13
|
+
from docs_to_md.utils.file_utils import FileIO
|
|
14
|
+
|
|
15
|
+
# Constants
|
|
16
|
+
MAX_REQUESTS_PER_MINUTE = 150
|
|
17
|
+
REQUEST_TIMEOUT = 30 # seconds
|
|
18
|
+
MAX_RETRIES = 3
|
|
19
|
+
|
|
20
|
+
logger = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class MarkerClient:
|
|
24
|
+
"""Client for interacting with the Marker API."""
|
|
25
|
+
|
|
26
|
+
BASE_URL = "https://www.datalab.to/api/v1/marker"
|
|
27
|
+
|
|
28
|
+
def __init__(self, api_key: str):
|
|
29
|
+
"""
|
|
30
|
+
Initialize the API client.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
api_key: API key for authentication
|
|
34
|
+
|
|
35
|
+
Raises:
|
|
36
|
+
APIError: If API key is not provided or invalid
|
|
37
|
+
"""
|
|
38
|
+
# Check for empty API key
|
|
39
|
+
if not api_key:
|
|
40
|
+
raise APIError("API key is required")
|
|
41
|
+
|
|
42
|
+
# Basic validation - API keys should typically be alphanumeric
|
|
43
|
+
# and have a reasonable length. The exact format depends on Marker's specs.
|
|
44
|
+
api_key = api_key.strip() # Remove accidental whitespace
|
|
45
|
+
if len(api_key) < 8: # Assuming a minimum sensible length
|
|
46
|
+
raise APIError("API key appears to be too short")
|
|
47
|
+
|
|
48
|
+
self.headers = {"X-Api-Key": api_key}
|
|
49
|
+
|
|
50
|
+
@sleep_and_retry
|
|
51
|
+
@limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
|
|
52
|
+
@backoff.on_exception(
|
|
53
|
+
backoff.expo,
|
|
54
|
+
(requests.exceptions.RequestException, json.JSONDecodeError),
|
|
55
|
+
max_tries=MAX_RETRIES
|
|
56
|
+
)
|
|
57
|
+
def submit_file(
|
|
58
|
+
self,
|
|
59
|
+
file_path: Path,
|
|
60
|
+
output_format: str = "markdown",
|
|
61
|
+
langs: str = "English",
|
|
62
|
+
use_llm: bool = False,
|
|
63
|
+
strip_existing_ocr: bool = False,
|
|
64
|
+
disable_image_extraction: bool = False,
|
|
65
|
+
force_ocr: bool = False,
|
|
66
|
+
paginate: bool = False,
|
|
67
|
+
max_pages: Optional[int] = None,
|
|
68
|
+
) -> Optional[str]:
|
|
69
|
+
"""
|
|
70
|
+
Submit a file for conversion.
|
|
71
|
+
|
|
72
|
+
Args:
|
|
73
|
+
file_path: Path to the file to submit
|
|
74
|
+
output_format: Desired output format (markdown, json, html)
|
|
75
|
+
langs: Comma-separated OCR languages
|
|
76
|
+
use_llm: Whether to use LLM for enhanced processing
|
|
77
|
+
strip_existing_ocr: Whether to redo OCR processing
|
|
78
|
+
disable_image_extraction: Whether to disable image extraction
|
|
79
|
+
force_ocr: Whether to force OCR on all pages
|
|
80
|
+
paginate: Whether to add page delimiters
|
|
81
|
+
max_pages: Maximum pages to process
|
|
82
|
+
|
|
83
|
+
Returns:
|
|
84
|
+
Request ID if successful, None otherwise
|
|
85
|
+
|
|
86
|
+
Raises:
|
|
87
|
+
APIError: If file is invalid or API request fails
|
|
88
|
+
"""
|
|
89
|
+
try:
|
|
90
|
+
# Validate file exists
|
|
91
|
+
if not file_path.exists():
|
|
92
|
+
raise APIError(f"File not found: {file_path}")
|
|
93
|
+
|
|
94
|
+
# Read file and check type
|
|
95
|
+
file_data = FileIO.read_file(file_path)
|
|
96
|
+
kind = filetype.guess(file_data)
|
|
97
|
+
if not kind or kind.mime not in SUPPORTED_MIME_TYPES:
|
|
98
|
+
raise APIError(f"Unsupported file type: {kind.mime if kind else 'unknown'}")
|
|
99
|
+
|
|
100
|
+
# Build form data
|
|
101
|
+
files = {
|
|
102
|
+
'file': (file_path.name, file_data, kind.mime),
|
|
103
|
+
'langs': (None, langs),
|
|
104
|
+
'force_ocr': (None, force_ocr),
|
|
105
|
+
'paginate': (None, paginate),
|
|
106
|
+
'strip_existing_ocr': (None, strip_existing_ocr),
|
|
107
|
+
'disable_image_extraction': (None, disable_image_extraction),
|
|
108
|
+
'use_llm': (None, use_llm),
|
|
109
|
+
'output_format': (None, output_format),
|
|
110
|
+
}
|
|
111
|
+
# Add max_pages only if it's provided
|
|
112
|
+
if max_pages is not None:
|
|
113
|
+
files['max_pages'] = (None, max_pages)
|
|
114
|
+
|
|
115
|
+
# Send request
|
|
116
|
+
response = requests.post(
|
|
117
|
+
self.BASE_URL,
|
|
118
|
+
files=files,
|
|
119
|
+
headers=self.headers,
|
|
120
|
+
timeout=REQUEST_TIMEOUT
|
|
121
|
+
)
|
|
122
|
+
response.raise_for_status()
|
|
123
|
+
|
|
124
|
+
# Parse response
|
|
125
|
+
data = response.json()
|
|
126
|
+
submit_response = SubmitResponse.model_validate(data)
|
|
127
|
+
|
|
128
|
+
if not submit_response.success:
|
|
129
|
+
logger.error(f"API request failed: {submit_response.error or 'Unknown error'}")
|
|
130
|
+
return None
|
|
131
|
+
|
|
132
|
+
logger.info(f"Successfully submitted file. Request ID: {submit_response.request_id}")
|
|
133
|
+
return submit_response.request_id
|
|
134
|
+
|
|
135
|
+
except Exception as e:
|
|
136
|
+
logger.error(f"Error submitting file {file_path}: {e}")
|
|
137
|
+
return None
|
|
138
|
+
|
|
139
|
+
@sleep_and_retry
|
|
140
|
+
@limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
|
|
141
|
+
@backoff.on_exception(
|
|
142
|
+
backoff.expo,
|
|
143
|
+
(requests.exceptions.RequestException, json.JSONDecodeError),
|
|
144
|
+
max_tries=MAX_RETRIES
|
|
145
|
+
)
|
|
146
|
+
def check_status(self, request_id: str) -> Optional[MarkerStatus]:
|
|
147
|
+
"""
|
|
148
|
+
Check the status of a conversion request.
|
|
149
|
+
|
|
150
|
+
Args:
|
|
151
|
+
request_id: Request ID to check
|
|
152
|
+
|
|
153
|
+
Returns:
|
|
154
|
+
MarkerStatus object with current status, or None if request fails
|
|
155
|
+
"""
|
|
156
|
+
if not request_id:
|
|
157
|
+
logger.error("Cannot check status: empty request_id provided")
|
|
158
|
+
return None
|
|
159
|
+
|
|
160
|
+
try:
|
|
161
|
+
# Make API request with timeout
|
|
162
|
+
response = requests.get(
|
|
163
|
+
f"{self.BASE_URL}/{request_id}",
|
|
164
|
+
headers=self.headers,
|
|
165
|
+
timeout=REQUEST_TIMEOUT
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
# Handle non-200 responses properly
|
|
169
|
+
if response.status_code != 200:
|
|
170
|
+
logger.error(f"API returned status code {response.status_code} for request {request_id}")
|
|
171
|
+
if response.status_code == 404:
|
|
172
|
+
return MarkerStatus(status=StatusEnum.PROCESSING, error="Request not found")
|
|
173
|
+
elif response.status_code == 401:
|
|
174
|
+
return MarkerStatus(status=StatusEnum.FAILED, error="Authentication failed")
|
|
175
|
+
elif response.status_code == 429:
|
|
176
|
+
return MarkerStatus(status=StatusEnum.PROCESSING, error="Rate limit exceeded")
|
|
177
|
+
return None
|
|
178
|
+
|
|
179
|
+
# Parse response JSON
|
|
180
|
+
try:
|
|
181
|
+
data = response.json()
|
|
182
|
+
except json.JSONDecodeError as e:
|
|
183
|
+
logger.error(f"Invalid JSON response for request {request_id}: {e}")
|
|
184
|
+
return None
|
|
185
|
+
|
|
186
|
+
# Handle empty response
|
|
187
|
+
if not data:
|
|
188
|
+
logger.error(f"Empty response for request {request_id}")
|
|
189
|
+
return None
|
|
190
|
+
|
|
191
|
+
# Validate and create status object
|
|
192
|
+
try:
|
|
193
|
+
status = MarkerStatus.model_validate(data)
|
|
194
|
+
return status
|
|
195
|
+
except Exception as e:
|
|
196
|
+
logger.error(f"Failed to parse status response for {request_id}: {e}")
|
|
197
|
+
return None
|
|
198
|
+
|
|
199
|
+
except requests.exceptions.Timeout:
|
|
200
|
+
logger.error(f"Timeout checking status for {request_id}")
|
|
201
|
+
return None
|
|
202
|
+
except requests.exceptions.ConnectionError:
|
|
203
|
+
logger.error(f"Connection error checking status for {request_id}")
|
|
204
|
+
return None
|
|
205
|
+
except requests.exceptions.RequestException as e:
|
|
206
|
+
logger.error(f"Request error checking status for {request_id}: {e}")
|
|
207
|
+
return None
|
|
208
|
+
except Exception as e:
|
|
209
|
+
logger.error(f"Unexpected error checking status for {request_id}: {e}")
|
|
210
|
+
return None
|
|
211
|
+
|
|
212
|
+
def __enter__(self):
|
|
213
|
+
"""Support for context manager."""
|
|
214
|
+
return self
|
|
215
|
+
|
|
216
|
+
def __exit__(self, exc_type, exc_val, exc_tb):
|
|
217
|
+
"""Clean up resources if needed."""
|
|
218
|
+
pass # No cleanup needed for API client
|
docs_to_md/api/models.py
ADDED
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
from enum import Enum
|
|
2
|
+
from typing import Any, Dict, Optional, Set
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
|
|
5
|
+
from pydantic import BaseModel
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class StatusEnum(str, Enum):
|
|
9
|
+
"""Enum for API status values."""
|
|
10
|
+
COMPLETE = "complete"
|
|
11
|
+
PROCESSING = "processing"
|
|
12
|
+
FAILED = "failed"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class MarkerStatus(BaseModel):
|
|
16
|
+
"""Model for API status response."""
|
|
17
|
+
status: StatusEnum # Indicates the status of the request (`complete`, or `processing`).
|
|
18
|
+
output_format: Optional[str] = None # The requested output format, `json`, `html`, or `markdown`.
|
|
19
|
+
success: Optional[bool] = None # Indicates if the request completed successfully. `True` or `False`.
|
|
20
|
+
error: Optional[str] = None # If there was an error, this contains the error message.
|
|
21
|
+
markdown: Optional[str] = None # The output from the file if `output_format` is `markdown`.
|
|
22
|
+
json_data: Optional[Dict[str, Any]] = None # The output from the file if `output_format` is `json`.
|
|
23
|
+
images: Optional[Dict[str, str]] = None # Dictionary of image filenames (keys) and base64 encoded images (values).
|
|
24
|
+
meta: Optional[Dict[str, Any]] = None # Metadata about the markdown conversion.
|
|
25
|
+
page_count: Optional[int] = None # Number of pages that were converted.
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class SubmitResponse(BaseModel):
|
|
29
|
+
"""Model for API submit response."""
|
|
30
|
+
success: bool
|
|
31
|
+
error: Optional[str] = None
|
|
32
|
+
request_id: str
|
|
33
|
+
request_check_url: Optional[str] = None
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class ApiParams:
|
|
38
|
+
"""Parameters for the Marker API submit call."""
|
|
39
|
+
output_format: str = "markdown"
|
|
40
|
+
langs: str = "English"
|
|
41
|
+
use_llm: bool = False
|
|
42
|
+
strip_existing_ocr: bool = False
|
|
43
|
+
disable_image_extraction: bool = False
|
|
44
|
+
force_ocr: bool = False
|
|
45
|
+
paginate: bool = False
|
|
46
|
+
max_pages: Optional[int] = None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# Supported mime types according to API docs
|
|
50
|
+
SUPPORTED_MIME_TYPES: Set[str] = {
|
|
51
|
+
# PDF
|
|
52
|
+
'application/pdf',
|
|
53
|
+
# Word documents
|
|
54
|
+
'application/msword',
|
|
55
|
+
'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
|
|
56
|
+
# Powerpoint
|
|
57
|
+
'application/vnd.ms-powerpoint',
|
|
58
|
+
'application/vnd.openxmlformats-officedocument.presentationml.presentation',
|
|
59
|
+
# Images
|
|
60
|
+
'image/png',
|
|
61
|
+
'image/jpeg',
|
|
62
|
+
'image/webp',
|
|
63
|
+
'image/gif',
|
|
64
|
+
'image/tiff',
|
|
65
|
+
'image/jpg'
|
|
66
|
+
}
|
|
File without changes
|
docs_to_md/config/cli.py
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from docs_to_md.config.settings import Config
|
|
5
|
+
from docs_to_md.utils.exceptions import ConfigurationError
|
|
6
|
+
from docs_to_md.utils.file_utils import get_env_var
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def parse_args() -> argparse.Namespace:
|
|
10
|
+
"""
|
|
11
|
+
Parse command line arguments.
|
|
12
|
+
|
|
13
|
+
Returns:
|
|
14
|
+
Parsed arguments namespace
|
|
15
|
+
"""
|
|
16
|
+
parser = argparse.ArgumentParser(
|
|
17
|
+
description="Process PDF files using Marker API.",
|
|
18
|
+
formatter_class=argparse.ArgumentDefaultsHelpFormatter
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
# Required arguments
|
|
22
|
+
parser.add_argument("input", help="Input file or directory path")
|
|
23
|
+
|
|
24
|
+
# Output format
|
|
25
|
+
parser.add_argument("--json", action="store_true", help="Output in JSON format")
|
|
26
|
+
|
|
27
|
+
# OCR settings
|
|
28
|
+
parser.add_argument("--langs", default="English", help="Comma-separated OCR languages")
|
|
29
|
+
parser.add_argument("--llm", action="store_true", help="Use LLM for enhanced processing")
|
|
30
|
+
parser.add_argument("--strip", action="store_true", help="Redo OCR processing")
|
|
31
|
+
parser.add_argument("--noimg", action="store_true", help="Disable image extraction")
|
|
32
|
+
parser.add_argument("--force", action="store_true", help="Force OCR on all pages")
|
|
33
|
+
parser.add_argument("--pages", action="store_true", help="Add page delimiters")
|
|
34
|
+
parser.add_argument("--max-pages", type=int, help="Maximum number of pages to process from the start of the file")
|
|
35
|
+
|
|
36
|
+
# Advanced settings
|
|
37
|
+
parser.add_argument("--max", action="store_true", help="Enable all OCR enhancements (LLM, strip OCR, force OCR)")
|
|
38
|
+
parser.add_argument("--no-chunk", action="store_true", help="Disable PDF chunking (sets chunk size to 1 million)")
|
|
39
|
+
parser.add_argument("-cs", "--chunk-size", type=int, help="Set PDF chunk size in pages", default=25)
|
|
40
|
+
parser.add_argument("--output-dir", help="Output directory", default="converted")
|
|
41
|
+
parser.add_argument("--cache-dir", help="Cache directory", default=".docs_to_md_cache")
|
|
42
|
+
|
|
43
|
+
return parser.parse_args()
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def create_config_from_args() -> Config:
|
|
47
|
+
"""
|
|
48
|
+
Create configuration from command line arguments.
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
Config object with settings from command line
|
|
52
|
+
|
|
53
|
+
Raises:
|
|
54
|
+
ConfigurationError: If required arguments are missing
|
|
55
|
+
"""
|
|
56
|
+
args = parse_args()
|
|
57
|
+
|
|
58
|
+
# Get API key from environment
|
|
59
|
+
try:
|
|
60
|
+
api_key = get_env_var("MARKER_PDF_KEY")
|
|
61
|
+
except Exception as e:
|
|
62
|
+
raise ConfigurationError(f"API key not found: {e}. Set the MARKER_PDF_KEY environment variable.")
|
|
63
|
+
|
|
64
|
+
# If --no-chunk is specified, override chunk size to effectively disable chunking
|
|
65
|
+
chunk_size = 1_000_000 if args.no_chunk else args.chunk_size
|
|
66
|
+
|
|
67
|
+
config = Config(
|
|
68
|
+
api_key=api_key,
|
|
69
|
+
input_path=args.input,
|
|
70
|
+
output_dir=Path(args.output_dir),
|
|
71
|
+
cache_dir=Path(args.cache_dir),
|
|
72
|
+
output_format="json" if args.json else "markdown",
|
|
73
|
+
langs=args.langs,
|
|
74
|
+
use_llm=args.llm or args.max,
|
|
75
|
+
strip_existing_ocr=args.strip or args.max,
|
|
76
|
+
disable_image_extraction=args.noimg,
|
|
77
|
+
force_ocr=args.force or args.max,
|
|
78
|
+
paginate=args.pages,
|
|
79
|
+
chunk_size=chunk_size,
|
|
80
|
+
max_pages=args.max_pages
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
# Validate the configuration
|
|
84
|
+
config.validate()
|
|
85
|
+
|
|
86
|
+
return config
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Optional, List
|
|
4
|
+
import logging
|
|
5
|
+
import os
|
|
6
|
+
|
|
7
|
+
from docs_to_md.utils.exceptions import ConfigurationError
|
|
8
|
+
from docs_to_md.utils.file_utils import ensure_directory, get_env_var
|
|
9
|
+
|
|
10
|
+
logger = logging.getLogger(__name__)
|
|
11
|
+
|
|
12
|
+
@dataclass
|
|
13
|
+
class Config:
|
|
14
|
+
"""Global configuration for marker PDF conversion."""
|
|
15
|
+
# API settings
|
|
16
|
+
api_key: str
|
|
17
|
+
|
|
18
|
+
# Input/output settings
|
|
19
|
+
input_path: str
|
|
20
|
+
output_dir: Path = Path("converted")
|
|
21
|
+
cache_dir: Path = Path(".docs_to_md_cache")
|
|
22
|
+
tmp_dir: Path = Path("chunks")
|
|
23
|
+
|
|
24
|
+
# Conversion settings
|
|
25
|
+
output_format: str = "markdown"
|
|
26
|
+
langs: str = "English"
|
|
27
|
+
chunk_size: int = 25
|
|
28
|
+
|
|
29
|
+
# Feature flags
|
|
30
|
+
use_llm: bool = False
|
|
31
|
+
strip_existing_ocr: bool = False
|
|
32
|
+
disable_image_extraction: bool = False
|
|
33
|
+
force_ocr: bool = False
|
|
34
|
+
paginate: bool = False
|
|
35
|
+
max_pages: Optional[int] = None
|
|
36
|
+
|
|
37
|
+
@classmethod
|
|
38
|
+
def from_env(cls, api_key_var: str = "MARKER_PDF_KEY") -> "Config":
|
|
39
|
+
"""
|
|
40
|
+
Create a configuration from environment variables.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
api_key_var: Environment variable name for API key
|
|
44
|
+
|
|
45
|
+
Returns:
|
|
46
|
+
Config instance with settings from environment
|
|
47
|
+
|
|
48
|
+
Raises:
|
|
49
|
+
ConfigurationError: If required environment variables are missing
|
|
50
|
+
"""
|
|
51
|
+
try:
|
|
52
|
+
api_key = get_env_var(api_key_var)
|
|
53
|
+
|
|
54
|
+
# This will raise an error if MARKER_INPUT_PATH is not set
|
|
55
|
+
input_path = get_env_var("MARKER_INPUT_PATH", required=False)
|
|
56
|
+
|
|
57
|
+
if not input_path:
|
|
58
|
+
raise ConfigurationError("No input path specified. Use MARKER_INPUT_PATH environment variable.")
|
|
59
|
+
|
|
60
|
+
return cls(
|
|
61
|
+
api_key=api_key,
|
|
62
|
+
input_path=input_path,
|
|
63
|
+
output_dir=Path(get_env_var("MARKER_OUTPUT_DIR", required=False) or "converted"),
|
|
64
|
+
cache_dir=Path(get_env_var("DOCS_TO_MD_CACHE_DIR", required=False) or ".docs_to_md_cache"),
|
|
65
|
+
output_format=get_env_var("MARKER_OUTPUT_FORMAT", required=False) or "markdown",
|
|
66
|
+
langs=get_env_var("MARKER_LANGS", required=False) or "English",
|
|
67
|
+
use_llm=get_env_var("MARKER_USE_LLM", required=False) == "1",
|
|
68
|
+
chunk_size=int(get_env_var("MARKER_CHUNK_SIZE", required=False) or "25")
|
|
69
|
+
)
|
|
70
|
+
except ValueError as e:
|
|
71
|
+
raise ConfigurationError(f"Invalid environment configuration: {e}")
|
|
72
|
+
|
|
73
|
+
def ensure_directories(self) -> None:
|
|
74
|
+
"""Ensure all required directories exist."""
|
|
75
|
+
ensure_directory(self.output_dir)
|
|
76
|
+
ensure_directory(self.cache_dir)
|
|
77
|
+
ensure_directory(self.tmp_dir)
|
|
78
|
+
|
|
79
|
+
def validate(self) -> None:
|
|
80
|
+
"""
|
|
81
|
+
Validate configuration values.
|
|
82
|
+
|
|
83
|
+
Raises:
|
|
84
|
+
ConfigurationError: If configuration is invalid
|
|
85
|
+
"""
|
|
86
|
+
if not self.api_key:
|
|
87
|
+
raise ConfigurationError("API key is required")
|
|
88
|
+
|
|
89
|
+
if not self.input_path:
|
|
90
|
+
raise ConfigurationError("Input path is required")
|
|
91
|
+
|
|
92
|
+
if self.chunk_size < 1:
|
|
93
|
+
raise ConfigurationError("Chunk size must be at least 1")
|
|
94
|
+
|
|
95
|
+
if self.max_pages is not None and self.max_pages < 1:
|
|
96
|
+
raise ConfigurationError("Max pages must be at least 1")
|
|
97
|
+
|
|
98
|
+
if self.output_format not in ["markdown", "json", "html", "txt"]:
|
|
99
|
+
raise ConfigurationError(f"Unsupported output format: {self.output_format}")
|
|
100
|
+
|
|
101
|
+
# Additional validations can be added here
|
|
102
|
+
|
|
103
|
+
if not self.output_dir.exists():
|
|
104
|
+
try:
|
|
105
|
+
ensure_directory(self.output_dir)
|
|
106
|
+
logger.info(f"Created output directory: {self.output_dir}")
|
|
107
|
+
except Exception as e:
|
|
108
|
+
raise ConfigurationError(f"Failed to create output directory: {e}")
|
|
File without changes
|