pdf-to-markdown-cli 0.2.1__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0}/MANIFEST.in +2 -2
  2. {pdf_to_markdown_cli-0.2.1/pdf_to_markdown_cli.egg-info → pdf_to_markdown_cli-0.4.0}/PKG-INFO +40 -16
  3. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0}/README.md +17 -5
  4. pdf_to_markdown_cli-0.4.0/pyproject.toml +51 -0
  5. pdf_to_markdown_cli-0.4.0/src/docs_to_md/api/client.py +183 -0
  6. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/api/models.py +29 -1
  7. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/config/cli.py +32 -31
  8. pdf_to_markdown_cli-0.4.0/src/docs_to_md/config/settings.py +54 -0
  9. pdf_to_markdown_cli-0.4.0/src/docs_to_md/core/paths.py +80 -0
  10. pdf_to_markdown_cli-0.4.0/src/docs_to_md/core/processor.py +344 -0
  11. pdf_to_markdown_cli-0.4.0/src/docs_to_md/core/result_handler.py +601 -0
  12. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/main.py +20 -11
  13. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/storage/cache.py +2 -2
  14. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/storage/models.py +5 -2
  15. pdf_to_markdown_cli-0.4.0/src/docs_to_md/utils/file_utils.py +219 -0
  16. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/utils/logging.py +51 -39
  17. pdf_to_markdown_cli-0.2.1/docs_to_md/pdf/splitter.py → pdf_to_markdown_cli-0.4.0/src/docs_to_md/utils/pdf_splitter.py +3 -35
  18. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src/pdf_to_markdown_cli.egg-info}/PKG-INFO +40 -16
  19. pdf_to_markdown_cli-0.4.0/src/pdf_to_markdown_cli.egg-info/SOURCES.txt +31 -0
  20. pdf_to_markdown_cli-0.2.1/docs_to_md/api/client.py +0 -218
  21. pdf_to_markdown_cli-0.2.1/docs_to_md/config/settings.py +0 -108
  22. pdf_to_markdown_cli-0.2.1/docs_to_md/core/processor.py +0 -229
  23. pdf_to_markdown_cli-0.2.1/docs_to_md/core/result_handler.py +0 -496
  24. pdf_to_markdown_cli-0.2.1/docs_to_md/utils/__init__.py +0 -0
  25. pdf_to_markdown_cli-0.2.1/docs_to_md/utils/file_utils.py +0 -189
  26. pdf_to_markdown_cli-0.2.1/pdf_to_markdown_cli.egg-info/SOURCES.txt +0 -31
  27. pdf_to_markdown_cli-0.2.1/setup.py +0 -67
  28. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0}/LICENSE +0 -0
  29. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0}/setup.cfg +0 -0
  30. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/__init__.py +0 -0
  31. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/__main__.py +0 -0
  32. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/api/__init__.py +0 -0
  33. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/config/__init__.py +0 -0
  34. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/core/__init__.py +0 -0
  35. {pdf_to_markdown_cli-0.2.1/docs_to_md/pdf → pdf_to_markdown_cli-0.4.0/src/docs_to_md/storage}/__init__.py +0 -0
  36. {pdf_to_markdown_cli-0.2.1/docs_to_md/storage → pdf_to_markdown_cli-0.4.0/src/docs_to_md/utils}/__init__.py +0 -0
  37. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/utils/exceptions.py +0 -0
  38. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/pdf_to_markdown_cli.egg-info/dependency_links.txt +0 -0
  39. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/pdf_to_markdown_cli.egg-info/entry_points.txt +0 -0
  40. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/pdf_to_markdown_cli.egg-info/requires.txt +0 -0
  41. {pdf_to_markdown_cli-0.2.1 → pdf_to_markdown_cli-0.4.0/src}/pdf_to_markdown_cli.egg-info/top_level.txt +0 -0
@@ -2,8 +2,8 @@
2
2
  include README.md
3
3
  include LICENSE
4
4
 
5
- # Include the package source directory
6
- recursive-include docs_to_md *
5
+ # Include the package source directory (now within src/)
6
+ recursive-include src/docs_to_md *
7
7
 
8
8
  # Exclude development/cache/temporary files and directories
9
9
  global-exclude *.py[cod]
@@ -1,9 +1,31 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pdf-to-markdown-cli
3
- Version: 0.2.1
3
+ Version: 0.4.0
4
4
  Summary: CLI tool to convert PDF files (and other documents) to markdown using the Marker API.
5
- Home-page: https://github.com/SokolskyNikita/pdf-to-markdown-cli
6
5
  Author: Nikita Sokolsky
6
+ License: MIT License
7
+
8
+ Copyright (c) 2025 SokolskyNikita
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Project-URL: Homepage, https://github.com/SokolskyNikita/pdf-to-markdown-cli
7
29
  Project-URL: Bug Tracker, https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues
8
30
  Keywords: pdf,markdown,converter,cli,document,marker,md
9
31
  Classifier: Development Status :: 4 - Beta
@@ -28,17 +50,7 @@ Requires-Dist: pydantic>=2.0
28
50
  Requires-Dist: ratelimit>=2.0
29
51
  Requires-Dist: requests>=2.0
30
52
  Requires-Dist: tqdm>=4.0
31
- Dynamic: author
32
- Dynamic: classifier
33
- Dynamic: description
34
- Dynamic: description-content-type
35
- Dynamic: home-page
36
- Dynamic: keywords
37
53
  Dynamic: license-file
38
- Dynamic: project-url
39
- Dynamic: requires-dist
40
- Dynamic: requires-python
41
- Dynamic: summary
42
54
 
43
55
  # PDF to Markdown CLI (via the Datalab Marker API)
44
56
 
@@ -108,11 +120,19 @@ pdf-to-md /path/to/file.pdf --max
108
120
  - `--force`: Force OCR on all pages
109
121
  - `--pages`: Add page delimiters
110
122
  - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
111
- - `--max-pages`: Maximum number of pages to process from the start of the file
123
+ - `-mp`, `--max-pages`: Maximum number of pages to process from the start of the file
112
124
  - `--no-chunk`: Disable PDF chunking
113
- - `--chunk-size`: Set PDF chunk size in pages (default: 25)
114
- - `--output-dir`: Output directory (default: "converted")
115
- - `--cache-dir`: Cache directory (default: ".marker_cache")
125
+ - `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
126
+ - `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
127
+ - `-v`, `--verbose`: Enable verbose (DEBUG level) logging
128
+
129
+ ### Output Structure
130
+
131
+ By default, output files are saved in the same directory as the input file with the format `[input-filename].[format]`. For example, converting `/data/report.pdf` to markdown will result in `/data/report.md`.
132
+
133
+ If an output file with the same name already exists, the new file will be automatically renamed using a numeric suffix (e.g., `[input-filename]_1.[format]`, `[input-filename]_2.[format]`, etc.) to avoid overwriting.
134
+
135
+ If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
116
136
 
117
137
  ### API
118
138
 
@@ -176,3 +196,7 @@ For regular use after installation, use the `pdf-to-md` command.
176
196
  ## License
177
197
 
178
198
  This project is licensed under the MIT License - see the LICENSE file for details.
199
+
200
+ ## Future Considerations
201
+
202
+ While this project currently uses a modern structure (`pyproject.toml`, `src` layout), future development might involve migrating to a standardized project template, such as [simonw/python-lib](https://github.com/simonw/python-lib), to further align with community best practices and potentially simplify workflows like automated PyPI publishing via Trusted Publishers.
@@ -66,11 +66,19 @@ pdf-to-md /path/to/file.pdf --max
66
66
  - `--force`: Force OCR on all pages
67
67
  - `--pages`: Add page delimiters
68
68
  - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
69
- - `--max-pages`: Maximum number of pages to process from the start of the file
69
+ - `-mp`, `--max-pages`: Maximum number of pages to process from the start of the file
70
70
  - `--no-chunk`: Disable PDF chunking
71
- - `--chunk-size`: Set PDF chunk size in pages (default: 25)
72
- - `--output-dir`: Output directory (default: "converted")
73
- - `--cache-dir`: Cache directory (default: ".marker_cache")
71
+ - `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
72
+ - `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
73
+ - `-v`, `--verbose`: Enable verbose (DEBUG level) logging
74
+
75
+ ### Output Structure
76
+
77
+ By default, output files are saved in the same directory as the input file with the format `[input-filename].[format]`. For example, converting `/data/report.pdf` to markdown will result in `/data/report.md`.
78
+
79
+ If an output file with the same name already exists, the new file will be automatically renamed using a numeric suffix (e.g., `[input-filename]_1.[format]`, `[input-filename]_2.[format]`, etc.) to avoid overwriting.
80
+
81
+ If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
74
82
 
75
83
  ### API
76
84
 
@@ -133,4 +141,8 @@ For regular use after installation, use the `pdf-to-md` command.
133
141
 
134
142
  ## License
135
143
 
136
- This project is licensed under the MIT License - see the LICENSE file for details.
144
+ This project is licensed under the MIT License - see the LICENSE file for details.
145
+
146
+ ## Future Considerations
147
+
148
+ While this project currently uses a modern structure (`pyproject.toml`, `src` layout), future development might involve migrating to a standardized project template, such as [simonw/python-lib](https://github.com/simonw/python-lib), to further align with community best practices and potentially simplify workflows like automated PyPI publishing via Trusted Publishers.
@@ -0,0 +1,51 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "pdf-to-markdown-cli"
7
+ version = "0.4.0"
8
+ description = "CLI tool to convert PDF files (and other documents) to markdown using the Marker API."
9
+ readme = "README.md"
10
+ authors = [
11
+ { name = "Nikita Sokolsky" },
12
+ ]
13
+ license = { file = "LICENSE" }
14
+ requires-python = ">=3.10"
15
+ classifiers = [
16
+ "Development Status :: 4 - Beta",
17
+ "Intended Audience :: Developers",
18
+ "Intended Audience :: End Users/Desktop",
19
+ "License :: OSI Approved :: MIT License",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.10",
22
+ "Programming Language :: Python :: 3.11",
23
+ "Programming Language :: Python :: 3.12",
24
+ "Operating System :: OS Independent",
25
+ "Topic :: Text Processing",
26
+ "Topic :: Utilities",
27
+ ]
28
+ keywords = ["pdf", "markdown", "converter", "cli", "document", "marker", "md"]
29
+ dependencies = [
30
+ "backoff>=2.0",
31
+ "diskcache>=5.0",
32
+ "filetype>=1.0",
33
+ "pikepdf>=8.0",
34
+ "pydantic>=2.0",
35
+ "ratelimit>=2.0",
36
+ "requests>=2.0",
37
+ "tqdm>=4.0",
38
+ ]
39
+
40
+ # Define project URLs
41
+ [project.urls]
42
+ "Homepage" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli"
43
+ "Bug Tracker" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues"
44
+
45
+ # Define console script entry point
46
+ [project.scripts]
47
+ pdf-to-md = "docs_to_md.main:main"
48
+
49
+ # Just in case: Configure setuptools to find packages in src/
50
+ [tool.setuptools.packages.find]
51
+ where = ["src"]
@@ -0,0 +1,183 @@
1
+ import json
2
+ import logging
3
+ from pathlib import Path
4
+ from typing import Optional
5
+
6
+ import backoff
7
+ import filetype
8
+ import requests
9
+ from ratelimit import limits, sleep_and_retry
10
+
11
+ from docs_to_md.api.models import (
12
+ MarkerStatus,
13
+ StatusEnum,
14
+ SubmitResponse,
15
+ SUPPORTED_MIME_TYPES,
16
+ )
17
+ from docs_to_md.utils.exceptions import APIError
18
+ from docs_to_md.utils.file_utils import FileIO
19
+
20
+ # Client-side constants
21
+ MAX_REQUESTS_PER_MINUTE = 150
22
+ REQUEST_TIMEOUT_SECONDS = 30
23
+ MAX_RETRIES = 3
24
+
25
+ logger = logging.getLogger(__name__)
26
+
27
+
28
+ class MarkerClient:
29
+ BASE_MARKER_API_ENDPOINT = "https://www.datalab.to/api/v1/marker"
30
+
31
+ # See datalab_marker_api_docs.md#authentication for API key details
32
+ def __init__(self, api_key: str):
33
+ if not api_key or not api_key.strip():
34
+ raise APIError("API key is required")
35
+
36
+ self.headers = {"X-Api-Key": api_key.strip()}
37
+
38
+ @sleep_and_retry
39
+ @limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
40
+ @backoff.on_exception(
41
+ backoff.expo,
42
+ (requests.exceptions.RequestException, json.JSONDecodeError),
43
+ max_tries=MAX_RETRIES,
44
+ )
45
+ def submit_file(
46
+ self,
47
+ file_path: Path,
48
+ output_format: str = "markdown",
49
+ langs: str = "English",
50
+ use_llm: bool = False,
51
+ strip_existing_ocr: bool = False,
52
+ disable_image_extraction: bool = False,
53
+ force_ocr: bool = False,
54
+ paginate: bool = False,
55
+ max_pages: Optional[int] = None,
56
+ ) -> Optional[str]:
57
+ """Submit a file for conversion via the Marker API.
58
+ See datalab_marker_api_docs.md#marker for API parameter details.
59
+ """
60
+ try:
61
+ if not file_path.exists():
62
+ raise APIError(f"File not found: {file_path}")
63
+
64
+ file_data = FileIO.read_file(file_path)
65
+ kind = filetype.guess(file_data)
66
+
67
+ # Supported types listed in datalab_marker_api_docs.md#supported-file-types
68
+ if not kind or kind.mime not in SUPPORTED_MIME_TYPES:
69
+ raise APIError(
70
+ f"Unsupported file type: {kind.mime if kind else 'unknown'}"
71
+ )
72
+
73
+ form_data = {
74
+ "file": (file_path.name, file_data, kind.mime),
75
+ "langs": (None, langs),
76
+ "force_ocr": (None, force_ocr),
77
+ "paginate": (None, paginate),
78
+ "strip_existing_ocr": (None, strip_existing_ocr),
79
+ "disable_image_extraction": (None, disable_image_extraction),
80
+ "use_llm": (None, use_llm),
81
+ "output_format": (None, output_format),
82
+ "max_pages": (None, max_pages),
83
+ }
84
+
85
+ response = requests.post(
86
+ self.BASE_MARKER_API_ENDPOINT,
87
+ files=form_data,
88
+ headers=self.headers,
89
+ timeout=REQUEST_TIMEOUT_SECONDS,
90
+ )
91
+ response.raise_for_status() # Default handling for HTTP errors,
92
+ submit_response = SubmitResponse.model_validate(response.json())
93
+
94
+ if not submit_response.success:
95
+ logger.error(
96
+ f"API request failed: {submit_response.error or 'Unknown error'}"
97
+ )
98
+ return None
99
+
100
+ logger.info(
101
+ f"Successfully submitted file {file_path.name}. Request ID: {submit_response.request_id}"
102
+ )
103
+ return submit_response.request_id
104
+
105
+ except Exception as e:
106
+ logger.error(f"Error submitting file {file_path}: {e}")
107
+ return None
108
+
109
+ def _handle_status_error(
110
+ self, status_code: int, request_id: str
111
+ ) -> Optional[MarkerStatus]:
112
+ """Handle non-200 status codes from the check_status endpoint."""
113
+ logger.error(f"API returned status code {status_code} for request {request_id}")
114
+ # Specific handling for non-fatal polling errors
115
+ if status_code == 404:
116
+ # Treat not found as still processing, might appear later
117
+ return MarkerStatus(status=StatusEnum.PROCESSING, error="Request not found")
118
+ elif status_code == 401:
119
+ return MarkerStatus(status=StatusEnum.FAILED, error="Authentication failed")
120
+ elif status_code == 429:
121
+ # Treat rate limit as still processing, should retry later
122
+ return MarkerStatus(
123
+ status=StatusEnum.PROCESSING, error="Rate limit exceeded"
124
+ )
125
+ # For other non-200 errors, return None to indicate failure to get status
126
+ return None
127
+
128
+ @sleep_and_retry
129
+ @limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
130
+ @backoff.on_exception(
131
+ backoff.expo,
132
+ (requests.exceptions.RequestException, json.JSONDecodeError),
133
+ max_tries=MAX_RETRIES,
134
+ )
135
+ def check_status(self, request_id: str) -> Optional[MarkerStatus]:
136
+ """
137
+ Check the status of a conversion request.
138
+ See datalab_marker_api_docs.md#marker for polling details.
139
+
140
+ Returns:
141
+ MarkerStatus object with current status, or None if the check fails.
142
+ """
143
+ if not request_id:
144
+ logger.error("Empty Marker request ID provided, skipping status check")
145
+ return None
146
+
147
+ try:
148
+ response = requests.get(
149
+ f"{self.BASE_MARKER_API_ENDPOINT}/{request_id}",
150
+ headers=self.headers,
151
+ timeout=REQUEST_TIMEOUT_SECONDS,
152
+ )
153
+
154
+ if response.status_code != 200:
155
+ return self._handle_status_error(response.status_code, request_id)
156
+
157
+ data = response.json()
158
+ # Handle potential empty response from API
159
+ if not data:
160
+ logger.error(f"Empty response for request {request_id}")
161
+ return None
162
+
163
+ status = MarkerStatus.model_validate(data)
164
+ return status
165
+
166
+ except json.JSONDecodeError as e:
167
+ logger.error(f"Invalid JSON response for request {request_id}: {e}")
168
+ return None
169
+ except (
170
+ requests.exceptions.RequestException
171
+ ) as e: # Consolidated network/request errors
172
+ logger.error(f"Request error checking status for {request_id}: {e}")
173
+ return None
174
+ except Exception as e: # Catch-all for validation or other unexpected errors
175
+ logger.error(f"Unexpected error checking status for {request_id}: {e}")
176
+ return None
177
+
178
+ def __enter__(self):
179
+ return self
180
+
181
+ def __exit__(self, exc_type, exc_val, exc_tb):
182
+ # No specific cleanup needed for this client
183
+ pass
@@ -44,9 +44,37 @@ class ApiParams:
44
44
  force_ocr: bool = False
45
45
  paginate: bool = False
46
46
  max_pages: Optional[int] = None
47
+
48
+ # Map of supported output formats to their extensions
49
+ SUPPORTED_FORMAT_EXTENSIONS = {
50
+ "markdown": ".md",
51
+ "json": ".json",
52
+ "html": ".html",
53
+ "txt": ".txt"
54
+ }
47
55
 
56
+ SUPPORTED_IMAGE_EXTENSIONS: Set[str] = {
57
+ "jpg",
58
+ "jpeg",
59
+ "png",
60
+ "gif",
61
+ "tiff"
62
+ }
48
63
 
49
- # Supported mime types according to API docs
64
+ SUPPORTED_INPUT_EXTENSIONS: Set[str] = {
65
+ "pdf",
66
+ "docx",
67
+ "doc",
68
+ "pptx",
69
+ "ppt",
70
+ "jpg",
71
+ "jpeg",
72
+ "png",
73
+ "gif",
74
+ "tiff"
75
+ }
76
+
77
+ # Supported mime types according to datalab_marker_api_docs.md#supported-file-types
50
78
  SUPPORTED_MIME_TYPES: Set[str] = {
51
79
  # PDF
52
80
  'application/pdf',
@@ -1,61 +1,57 @@
1
1
  import argparse
2
- from pathlib import Path
2
+ from pathlib import Path
3
+ import os
4
+ from typing import Optional
5
+ import importlib.metadata
3
6
 
4
7
  from docs_to_md.config.settings import Config
5
- from docs_to_md.utils.exceptions import ConfigurationError
6
- from docs_to_md.utils.file_utils import get_env_var
8
+ from docs_to_md.utils.exceptions import ConfigurationError, FileError
7
9
 
8
10
 
9
11
  def parse_args() -> argparse.Namespace:
10
- """
11
- Parse command line arguments.
12
-
13
- Returns:
14
- Parsed arguments namespace
15
- """
12
+ # Get package version dynamically
13
+ try:
14
+ __version__ = importlib.metadata.version('pdf-to-markdown-cli')
15
+ except importlib.metadata.PackageNotFoundError:
16
+ __version__ = 'unknown' # Fallback if package not installed
17
+
16
18
  parser = argparse.ArgumentParser(
17
19
  description="Process PDF files using Marker API.",
18
20
  formatter_class=argparse.ArgumentDefaultsHelpFormatter
19
21
  )
20
22
 
21
- # Required arguments
23
+ # Add version argument
24
+ parser.add_argument(
25
+ '--version',
26
+ action='version',
27
+ version=f'pdf-to-markdown-cli version: {__version__}'
28
+ )
29
+
22
30
  parser.add_argument("input", help="Input file or directory path")
23
31
 
24
- # Output format
25
32
  parser.add_argument("--json", action="store_true", help="Output in JSON format")
26
33
 
27
- # OCR settings
28
- parser.add_argument("--langs", default="English", help="Comma-separated OCR languages")
34
+ parser.add_argument("-l", "--langs", default="English", help="Comma-separated OCR languages")
29
35
  parser.add_argument("--llm", action="store_true", help="Use LLM for enhanced processing")
30
36
  parser.add_argument("--strip", action="store_true", help="Redo OCR processing")
31
37
  parser.add_argument("--noimg", action="store_true", help="Disable image extraction")
32
38
  parser.add_argument("--force", action="store_true", help="Force OCR on all pages")
33
39
  parser.add_argument("--pages", action="store_true", help="Add page delimiters")
34
- parser.add_argument("--max-pages", type=int, help="Maximum number of pages to process from the start of the file")
40
+ parser.add_argument("-mp", "--max-pages", type=int, help="Maximum number of pages to process from the start of the file")
35
41
 
36
- # Advanced settings
37
42
  parser.add_argument("--max", action="store_true", help="Enable all OCR enhancements (LLM, strip OCR, force OCR)")
38
43
  parser.add_argument("--no-chunk", action="store_true", help="Disable PDF chunking (sets chunk size to 1 million)")
39
44
  parser.add_argument("-cs", "--chunk-size", type=int, help="Set PDF chunk size in pages", default=25)
40
- parser.add_argument("--output-dir", help="Output directory", default="converted")
41
- parser.add_argument("--cache-dir", help="Cache directory", default=".docs_to_md_cache")
45
+ parser.add_argument("-o", "--output-dir", help="Absolute path to the output directory (default: same directory as input file)", default=None)
46
+
47
+ parser.add_argument("-v", "--verbose", action="store_true", help="Enable verbose (DEBUG level) logging")
42
48
 
43
49
  return parser.parse_args()
44
50
 
45
51
 
46
52
  def create_config_from_args() -> Config:
47
- """
48
- Create configuration from command line arguments.
49
-
50
- Returns:
51
- Config object with settings from command line
52
-
53
- Raises:
54
- ConfigurationError: If required arguments are missing
55
- """
56
53
  args = parse_args()
57
54
 
58
- # Get API key from environment
59
55
  try:
60
56
  api_key = get_env_var("MARKER_PDF_KEY")
61
57
  except Exception as e:
@@ -67,8 +63,7 @@ def create_config_from_args() -> Config:
67
63
  config = Config(
68
64
  api_key=api_key,
69
65
  input_path=args.input,
70
- output_dir=Path(args.output_dir),
71
- cache_dir=Path(args.cache_dir),
66
+ output_dir=Path(args.output_dir) if args.output_dir else None,
72
67
  output_format="json" if args.json else "markdown",
73
68
  langs=args.langs,
74
69
  use_llm=args.llm or args.max,
@@ -80,7 +75,13 @@ def create_config_from_args() -> Config:
80
75
  max_pages=args.max_pages
81
76
  )
82
77
 
83
- # Validate the configuration
84
78
  config.validate()
85
79
 
86
- return config
80
+ return config
81
+
82
+ def get_env_var(name: str, required: bool = True) -> Optional[str]:
83
+ """Get environment variable with optional requirement."""
84
+ value = os.getenv(name)
85
+ if required and not value:
86
+ raise FileError(f"Required environment variable {name} is not set")
87
+ return value
@@ -0,0 +1,54 @@
1
+ from dataclasses import dataclass
2
+ from pathlib import Path
3
+ from typing import Optional
4
+ import logging
5
+
6
+ from docs_to_md.api.models import SUPPORTED_FORMAT_EXTENSIONS
7
+ from docs_to_md.utils.exceptions import ConfigurationError
8
+
9
+ logger = logging.getLogger(__name__)
10
+ SETTINGS_DIR_NAME = ".docs_to_md"
11
+
12
+
13
+ @dataclass
14
+ class Config:
15
+ """Global configuration for marker PDF conversion."""
16
+ api_key: str
17
+
18
+ input_path: str
19
+ output_dir: Optional[Path] = None
20
+ cache_dir: Path = Path.home() / SETTINGS_DIR_NAME / "cache" # Root directory for cache files
21
+ root_tmp_dir: Path = Path.home() / SETTINGS_DIR_NAME / "tmp" # Root directory for temporary files
22
+
23
+ output_format: str = "markdown"
24
+ langs: str = "English"
25
+ chunk_size: int = 25
26
+
27
+ use_llm: bool = False
28
+ strip_existing_ocr: bool = False
29
+ disable_image_extraction: bool = False
30
+ force_ocr: bool = False
31
+ paginate: bool = False
32
+ max_pages: Optional[int] = None
33
+
34
+ def validate(self) -> None:
35
+ if not self.api_key:
36
+ raise ConfigurationError("API key is required")
37
+
38
+ if not self.input_path:
39
+ raise ConfigurationError("Input path is required")
40
+
41
+ if not Path(self.input_path).exists():
42
+ raise ConfigurationError(f"Input path does not exist: {self.input_path}")
43
+
44
+ if self.chunk_size < 1:
45
+ raise ConfigurationError("Chunk size must be at least 1")
46
+
47
+ if self.max_pages is not None and self.max_pages < 1:
48
+ raise ConfigurationError("Max pages must be at least 1")
49
+
50
+ if not self.output_format or self.output_format not in SUPPORTED_FORMAT_EXTENSIONS:
51
+ raise ConfigurationError(f"Unsupported output format: {self.output_format}")
52
+
53
+ if self.output_dir is not None and not self.output_dir.is_absolute():
54
+ raise ConfigurationError(f"Output directory must be an absolute path: {self.output_dir}")
@@ -0,0 +1,80 @@
1
+ import logging
2
+ import random
3
+ import string
4
+ from dataclasses import dataclass
5
+ from pathlib import Path
6
+ from typing import Optional
7
+
8
+ # import os # Not used
9
+
10
+ from docs_to_md.utils.exceptions import FileError
11
+ from docs_to_md.api.models import SUPPORTED_FORMAT_EXTENSIONS
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+
16
+ def generate_unique_key(length: int = 8) -> str:
17
+ """Generates a random alphanumeric key of the specified length."""
18
+ characters = string.ascii_letters + string.digits
19
+ return "".join(random.choices(characters, k=length))
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class OutputPaths:
24
+ """Holds the determined output paths for a conversion."""
25
+
26
+ markdown_path: Path
27
+ images_dir: Path
28
+ unique_key: str # Add the key for potential later use
29
+
30
+
31
+ def determine_output_paths(
32
+ input_file: Path, output_dir_config: Optional[Path], output_format: str
33
+ ) -> OutputPaths:
34
+ """
35
+ Determines the final output markdown path and images directory path
36
+ using a unique key for each run.
37
+ The image directory path is determined but not created here.
38
+ """
39
+ if not input_file.is_file():
40
+ raise ValueError(f"Input path must be a file: {input_file}")
41
+
42
+ unique_key = generate_unique_key()
43
+ logger.debug(f"Generated unique key for run: {unique_key}")
44
+
45
+ base_output_dir = output_dir_config if output_dir_config else input_file.parent
46
+ try:
47
+ # Still ensure the base output directory exists
48
+ base_output_dir.mkdir(parents=True, exist_ok=True)
49
+ except OSError as e:
50
+ raise FileError(
51
+ f"Could not create or access base output directory {base_output_dir}: {e}"
52
+ ) from e
53
+
54
+ # Use the imported mapping to get the desired file extension (remove the leading dot)
55
+ file_extension = SUPPORTED_FORMAT_EXTENSIONS.get(output_format, output_format)
56
+ if file_extension.startswith("."):
57
+ file_extension = file_extension[1:] # Remove leading dot if present
58
+
59
+ markdown_filename_base = input_file.stem
60
+ final_markdown_filename = f"{markdown_filename_base}_{unique_key}.{file_extension}"
61
+ final_markdown_path = base_output_dir / final_markdown_filename
62
+
63
+ logger.debug(f"Determined final markdown path: {final_markdown_path}")
64
+
65
+ # Determine image directory path (placed in the same dir as the markdown file)
66
+ image_dir_name = f"images_{unique_key}"
67
+ final_images_dir = base_output_dir / image_dir_name
68
+
69
+ # Note: The image directory is NOT created here.
70
+ # Creation should happen later, only if images are actually extracted.
71
+
72
+ logger.info(
73
+ f"Determined final paths: Markdown='{final_markdown_path}', Images='{final_images_dir}'"
74
+ )
75
+
76
+ return OutputPaths(
77
+ markdown_path=final_markdown_path,
78
+ images_dir=final_images_dir,
79
+ unique_key=unique_key
80
+ )