pdf-to-markdown-cli 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
docs_to_md/__init__.py ADDED
File without changes
docs_to_md/__main__.py ADDED
@@ -0,0 +1,5 @@
1
+ import sys
2
+ from .main import main
3
+
4
+ if __name__ == "__main__":
5
+ sys.exit(main())
File without changes
@@ -0,0 +1,218 @@
1
+ import json
2
+ import logging
3
+ from pathlib import Path
4
+ from typing import Dict, Optional, Any
5
+
6
+ import backoff
7
+ import filetype
8
+ import requests
9
+ from ratelimit import limits, sleep_and_retry
10
+
11
+ from docs_to_md.api.models import MarkerStatus, StatusEnum, SubmitResponse, SUPPORTED_MIME_TYPES
12
+ from docs_to_md.utils.exceptions import APIError
13
+ from docs_to_md.utils.file_utils import FileIO
14
+
15
+ # Constants
16
+ MAX_REQUESTS_PER_MINUTE = 150
17
+ REQUEST_TIMEOUT = 30 # seconds
18
+ MAX_RETRIES = 3
19
+
20
+ logger = logging.getLogger(__name__)
21
+
22
+
23
+ class MarkerClient:
24
+ """Client for interacting with the Marker API."""
25
+
26
+ BASE_URL = "https://www.datalab.to/api/v1/marker"
27
+
28
+ def __init__(self, api_key: str):
29
+ """
30
+ Initialize the API client.
31
+
32
+ Args:
33
+ api_key: API key for authentication
34
+
35
+ Raises:
36
+ APIError: If API key is not provided or invalid
37
+ """
38
+ # Check for empty API key
39
+ if not api_key:
40
+ raise APIError("API key is required")
41
+
42
+ # Basic validation - API keys should typically be alphanumeric
43
+ # and have a reasonable length. The exact format depends on Marker's specs.
44
+ api_key = api_key.strip() # Remove accidental whitespace
45
+ if len(api_key) < 8: # Assuming a minimum sensible length
46
+ raise APIError("API key appears to be too short")
47
+
48
+ self.headers = {"X-Api-Key": api_key}
49
+
50
+ @sleep_and_retry
51
+ @limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
52
+ @backoff.on_exception(
53
+ backoff.expo,
54
+ (requests.exceptions.RequestException, json.JSONDecodeError),
55
+ max_tries=MAX_RETRIES
56
+ )
57
+ def submit_file(
58
+ self,
59
+ file_path: Path,
60
+ output_format: str = "markdown",
61
+ langs: str = "English",
62
+ use_llm: bool = False,
63
+ strip_existing_ocr: bool = False,
64
+ disable_image_extraction: bool = False,
65
+ force_ocr: bool = False,
66
+ paginate: bool = False,
67
+ max_pages: Optional[int] = None,
68
+ ) -> Optional[str]:
69
+ """
70
+ Submit a file for conversion.
71
+
72
+ Args:
73
+ file_path: Path to the file to submit
74
+ output_format: Desired output format (markdown, json, html)
75
+ langs: Comma-separated OCR languages
76
+ use_llm: Whether to use LLM for enhanced processing
77
+ strip_existing_ocr: Whether to redo OCR processing
78
+ disable_image_extraction: Whether to disable image extraction
79
+ force_ocr: Whether to force OCR on all pages
80
+ paginate: Whether to add page delimiters
81
+ max_pages: Maximum pages to process
82
+
83
+ Returns:
84
+ Request ID if successful, None otherwise
85
+
86
+ Raises:
87
+ APIError: If file is invalid or API request fails
88
+ """
89
+ try:
90
+ # Validate file exists
91
+ if not file_path.exists():
92
+ raise APIError(f"File not found: {file_path}")
93
+
94
+ # Read file and check type
95
+ file_data = FileIO.read_file(file_path)
96
+ kind = filetype.guess(file_data)
97
+ if not kind or kind.mime not in SUPPORTED_MIME_TYPES:
98
+ raise APIError(f"Unsupported file type: {kind.mime if kind else 'unknown'}")
99
+
100
+ # Build form data
101
+ files = {
102
+ 'file': (file_path.name, file_data, kind.mime),
103
+ 'langs': (None, langs),
104
+ 'force_ocr': (None, force_ocr),
105
+ 'paginate': (None, paginate),
106
+ 'strip_existing_ocr': (None, strip_existing_ocr),
107
+ 'disable_image_extraction': (None, disable_image_extraction),
108
+ 'use_llm': (None, use_llm),
109
+ 'output_format': (None, output_format),
110
+ }
111
+ # Add max_pages only if it's provided
112
+ if max_pages is not None:
113
+ files['max_pages'] = (None, max_pages)
114
+
115
+ # Send request
116
+ response = requests.post(
117
+ self.BASE_URL,
118
+ files=files,
119
+ headers=self.headers,
120
+ timeout=REQUEST_TIMEOUT
121
+ )
122
+ response.raise_for_status()
123
+
124
+ # Parse response
125
+ data = response.json()
126
+ submit_response = SubmitResponse.model_validate(data)
127
+
128
+ if not submit_response.success:
129
+ logger.error(f"API request failed: {submit_response.error or 'Unknown error'}")
130
+ return None
131
+
132
+ logger.info(f"Successfully submitted file. Request ID: {submit_response.request_id}")
133
+ return submit_response.request_id
134
+
135
+ except Exception as e:
136
+ logger.error(f"Error submitting file {file_path}: {e}")
137
+ return None
138
+
139
+ @sleep_and_retry
140
+ @limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
141
+ @backoff.on_exception(
142
+ backoff.expo,
143
+ (requests.exceptions.RequestException, json.JSONDecodeError),
144
+ max_tries=MAX_RETRIES
145
+ )
146
+ def check_status(self, request_id: str) -> Optional[MarkerStatus]:
147
+ """
148
+ Check the status of a conversion request.
149
+
150
+ Args:
151
+ request_id: Request ID to check
152
+
153
+ Returns:
154
+ MarkerStatus object with current status, or None if request fails
155
+ """
156
+ if not request_id:
157
+ logger.error("Cannot check status: empty request_id provided")
158
+ return None
159
+
160
+ try:
161
+ # Make API request with timeout
162
+ response = requests.get(
163
+ f"{self.BASE_URL}/{request_id}",
164
+ headers=self.headers,
165
+ timeout=REQUEST_TIMEOUT
166
+ )
167
+
168
+ # Handle non-200 responses properly
169
+ if response.status_code != 200:
170
+ logger.error(f"API returned status code {response.status_code} for request {request_id}")
171
+ if response.status_code == 404:
172
+ return MarkerStatus(status=StatusEnum.PROCESSING, error="Request not found")
173
+ elif response.status_code == 401:
174
+ return MarkerStatus(status=StatusEnum.FAILED, error="Authentication failed")
175
+ elif response.status_code == 429:
176
+ return MarkerStatus(status=StatusEnum.PROCESSING, error="Rate limit exceeded")
177
+ return None
178
+
179
+ # Parse response JSON
180
+ try:
181
+ data = response.json()
182
+ except json.JSONDecodeError as e:
183
+ logger.error(f"Invalid JSON response for request {request_id}: {e}")
184
+ return None
185
+
186
+ # Handle empty response
187
+ if not data:
188
+ logger.error(f"Empty response for request {request_id}")
189
+ return None
190
+
191
+ # Validate and create status object
192
+ try:
193
+ status = MarkerStatus.model_validate(data)
194
+ return status
195
+ except Exception as e:
196
+ logger.error(f"Failed to parse status response for {request_id}: {e}")
197
+ return None
198
+
199
+ except requests.exceptions.Timeout:
200
+ logger.error(f"Timeout checking status for {request_id}")
201
+ return None
202
+ except requests.exceptions.ConnectionError:
203
+ logger.error(f"Connection error checking status for {request_id}")
204
+ return None
205
+ except requests.exceptions.RequestException as e:
206
+ logger.error(f"Request error checking status for {request_id}: {e}")
207
+ return None
208
+ except Exception as e:
209
+ logger.error(f"Unexpected error checking status for {request_id}: {e}")
210
+ return None
211
+
212
+ def __enter__(self):
213
+ """Support for context manager."""
214
+ return self
215
+
216
+ def __exit__(self, exc_type, exc_val, exc_tb):
217
+ """Clean up resources if needed."""
218
+ pass # No cleanup needed for API client
@@ -0,0 +1,66 @@
1
+ from enum import Enum
2
+ from typing import Any, Dict, Optional, Set
3
+ from dataclasses import dataclass
4
+
5
+ from pydantic import BaseModel
6
+
7
+
8
+ class StatusEnum(str, Enum):
9
+ """Enum for API status values."""
10
+ COMPLETE = "complete"
11
+ PROCESSING = "processing"
12
+ FAILED = "failed"
13
+
14
+
15
+ class MarkerStatus(BaseModel):
16
+ """Model for API status response."""
17
+ status: StatusEnum # Indicates the status of the request (`complete`, or `processing`).
18
+ output_format: Optional[str] = None # The requested output format, `json`, `html`, or `markdown`.
19
+ success: Optional[bool] = None # Indicates if the request completed successfully. `True` or `False`.
20
+ error: Optional[str] = None # If there was an error, this contains the error message.
21
+ markdown: Optional[str] = None # The output from the file if `output_format` is `markdown`.
22
+ json_data: Optional[Dict[str, Any]] = None # The output from the file if `output_format` is `json`.
23
+ images: Optional[Dict[str, str]] = None # Dictionary of image filenames (keys) and base64 encoded images (values).
24
+ meta: Optional[Dict[str, Any]] = None # Metadata about the markdown conversion.
25
+ page_count: Optional[int] = None # Number of pages that were converted.
26
+
27
+
28
+ class SubmitResponse(BaseModel):
29
+ """Model for API submit response."""
30
+ success: bool
31
+ error: Optional[str] = None
32
+ request_id: str
33
+ request_check_url: Optional[str] = None
34
+
35
+
36
+ @dataclass
37
+ class ApiParams:
38
+ """Parameters for the Marker API submit call."""
39
+ output_format: str = "markdown"
40
+ langs: str = "English"
41
+ use_llm: bool = False
42
+ strip_existing_ocr: bool = False
43
+ disable_image_extraction: bool = False
44
+ force_ocr: bool = False
45
+ paginate: bool = False
46
+ max_pages: Optional[int] = None
47
+
48
+
49
+ # Supported mime types according to API docs
50
+ SUPPORTED_MIME_TYPES: Set[str] = {
51
+ # PDF
52
+ 'application/pdf',
53
+ # Word documents
54
+ 'application/msword',
55
+ 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
56
+ # Powerpoint
57
+ 'application/vnd.ms-powerpoint',
58
+ 'application/vnd.openxmlformats-officedocument.presentationml.presentation',
59
+ # Images
60
+ 'image/png',
61
+ 'image/jpeg',
62
+ 'image/webp',
63
+ 'image/gif',
64
+ 'image/tiff',
65
+ 'image/jpg'
66
+ }
File without changes
@@ -0,0 +1,86 @@
1
+ import argparse
2
+ from pathlib import Path
3
+
4
+ from docs_to_md.config.settings import Config
5
+ from docs_to_md.utils.exceptions import ConfigurationError
6
+ from docs_to_md.utils.file_utils import get_env_var
7
+
8
+
9
+ def parse_args() -> argparse.Namespace:
10
+ """
11
+ Parse command line arguments.
12
+
13
+ Returns:
14
+ Parsed arguments namespace
15
+ """
16
+ parser = argparse.ArgumentParser(
17
+ description="Process PDF files using Marker API.",
18
+ formatter_class=argparse.ArgumentDefaultsHelpFormatter
19
+ )
20
+
21
+ # Required arguments
22
+ parser.add_argument("input", help="Input file or directory path")
23
+
24
+ # Output format
25
+ parser.add_argument("--json", action="store_true", help="Output in JSON format")
26
+
27
+ # OCR settings
28
+ parser.add_argument("--langs", default="English", help="Comma-separated OCR languages")
29
+ parser.add_argument("--llm", action="store_true", help="Use LLM for enhanced processing")
30
+ parser.add_argument("--strip", action="store_true", help="Redo OCR processing")
31
+ parser.add_argument("--noimg", action="store_true", help="Disable image extraction")
32
+ parser.add_argument("--force", action="store_true", help="Force OCR on all pages")
33
+ parser.add_argument("--pages", action="store_true", help="Add page delimiters")
34
+ parser.add_argument("--max-pages", type=int, help="Maximum number of pages to process from the start of the file")
35
+
36
+ # Advanced settings
37
+ parser.add_argument("--max", action="store_true", help="Enable all OCR enhancements (LLM, strip OCR, force OCR)")
38
+ parser.add_argument("--no-chunk", action="store_true", help="Disable PDF chunking (sets chunk size to 1 million)")
39
+ parser.add_argument("-cs", "--chunk-size", type=int, help="Set PDF chunk size in pages", default=25)
40
+ parser.add_argument("--output-dir", help="Output directory", default="converted")
41
+ parser.add_argument("--cache-dir", help="Cache directory", default=".docs_to_md_cache")
42
+
43
+ return parser.parse_args()
44
+
45
+
46
+ def create_config_from_args() -> Config:
47
+ """
48
+ Create configuration from command line arguments.
49
+
50
+ Returns:
51
+ Config object with settings from command line
52
+
53
+ Raises:
54
+ ConfigurationError: If required arguments are missing
55
+ """
56
+ args = parse_args()
57
+
58
+ # Get API key from environment
59
+ try:
60
+ api_key = get_env_var("MARKER_PDF_KEY")
61
+ except Exception as e:
62
+ raise ConfigurationError(f"API key not found: {e}. Set the MARKER_PDF_KEY environment variable.")
63
+
64
+ # If --no-chunk is specified, override chunk size to effectively disable chunking
65
+ chunk_size = 1_000_000 if args.no_chunk else args.chunk_size
66
+
67
+ config = Config(
68
+ api_key=api_key,
69
+ input_path=args.input,
70
+ output_dir=Path(args.output_dir),
71
+ cache_dir=Path(args.cache_dir),
72
+ output_format="json" if args.json else "markdown",
73
+ langs=args.langs,
74
+ use_llm=args.llm or args.max,
75
+ strip_existing_ocr=args.strip or args.max,
76
+ disable_image_extraction=args.noimg,
77
+ force_ocr=args.force or args.max,
78
+ paginate=args.pages,
79
+ chunk_size=chunk_size,
80
+ max_pages=args.max_pages
81
+ )
82
+
83
+ # Validate the configuration
84
+ config.validate()
85
+
86
+ return config
@@ -0,0 +1,108 @@
1
+ from dataclasses import dataclass
2
+ from pathlib import Path
3
+ from typing import Optional, List
4
+ import logging
5
+ import os
6
+
7
+ from docs_to_md.utils.exceptions import ConfigurationError
8
+ from docs_to_md.utils.file_utils import ensure_directory, get_env_var
9
+
10
+ logger = logging.getLogger(__name__)
11
+
12
+ @dataclass
13
+ class Config:
14
+ """Global configuration for marker PDF conversion."""
15
+ # API settings
16
+ api_key: str
17
+
18
+ # Input/output settings
19
+ input_path: str
20
+ output_dir: Path = Path("converted")
21
+ cache_dir: Path = Path(".docs_to_md_cache")
22
+ tmp_dir: Path = Path("chunks")
23
+
24
+ # Conversion settings
25
+ output_format: str = "markdown"
26
+ langs: str = "English"
27
+ chunk_size: int = 25
28
+
29
+ # Feature flags
30
+ use_llm: bool = False
31
+ strip_existing_ocr: bool = False
32
+ disable_image_extraction: bool = False
33
+ force_ocr: bool = False
34
+ paginate: bool = False
35
+ max_pages: Optional[int] = None
36
+
37
+ @classmethod
38
+ def from_env(cls, api_key_var: str = "MARKER_PDF_KEY") -> "Config":
39
+ """
40
+ Create a configuration from environment variables.
41
+
42
+ Args:
43
+ api_key_var: Environment variable name for API key
44
+
45
+ Returns:
46
+ Config instance with settings from environment
47
+
48
+ Raises:
49
+ ConfigurationError: If required environment variables are missing
50
+ """
51
+ try:
52
+ api_key = get_env_var(api_key_var)
53
+
54
+ # This will raise an error if MARKER_INPUT_PATH is not set
55
+ input_path = get_env_var("MARKER_INPUT_PATH", required=False)
56
+
57
+ if not input_path:
58
+ raise ConfigurationError("No input path specified. Use MARKER_INPUT_PATH environment variable.")
59
+
60
+ return cls(
61
+ api_key=api_key,
62
+ input_path=input_path,
63
+ output_dir=Path(get_env_var("MARKER_OUTPUT_DIR", required=False) or "converted"),
64
+ cache_dir=Path(get_env_var("DOCS_TO_MD_CACHE_DIR", required=False) or ".docs_to_md_cache"),
65
+ output_format=get_env_var("MARKER_OUTPUT_FORMAT", required=False) or "markdown",
66
+ langs=get_env_var("MARKER_LANGS", required=False) or "English",
67
+ use_llm=get_env_var("MARKER_USE_LLM", required=False) == "1",
68
+ chunk_size=int(get_env_var("MARKER_CHUNK_SIZE", required=False) or "25")
69
+ )
70
+ except ValueError as e:
71
+ raise ConfigurationError(f"Invalid environment configuration: {e}")
72
+
73
+ def ensure_directories(self) -> None:
74
+ """Ensure all required directories exist."""
75
+ ensure_directory(self.output_dir)
76
+ ensure_directory(self.cache_dir)
77
+ ensure_directory(self.tmp_dir)
78
+
79
+ def validate(self) -> None:
80
+ """
81
+ Validate configuration values.
82
+
83
+ Raises:
84
+ ConfigurationError: If configuration is invalid
85
+ """
86
+ if not self.api_key:
87
+ raise ConfigurationError("API key is required")
88
+
89
+ if not self.input_path:
90
+ raise ConfigurationError("Input path is required")
91
+
92
+ if self.chunk_size < 1:
93
+ raise ConfigurationError("Chunk size must be at least 1")
94
+
95
+ if self.max_pages is not None and self.max_pages < 1:
96
+ raise ConfigurationError("Max pages must be at least 1")
97
+
98
+ if self.output_format not in ["markdown", "json", "html", "txt"]:
99
+ raise ConfigurationError(f"Unsupported output format: {self.output_format}")
100
+
101
+ # Additional validations can be added here
102
+
103
+ if not self.output_dir.exists():
104
+ try:
105
+ ensure_directory(self.output_dir)
106
+ logger.info(f"Created output directory: {self.output_dir}")
107
+ except Exception as e:
108
+ raise ConfigurationError(f"Failed to create output directory: {e}")
File without changes