pdf-to-markdown-cli 0.2.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0}/MANIFEST.in +2 -2
  2. {pdf_to_markdown_cli-0.2.0/pdf_to_markdown_cli.egg-info → pdf_to_markdown_cli-0.4.0}/PKG-INFO +61 -38
  3. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0}/README.md +38 -27
  4. pdf_to_markdown_cli-0.4.0/pyproject.toml +51 -0
  5. pdf_to_markdown_cli-0.4.0/src/docs_to_md/api/client.py +183 -0
  6. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/api/models.py +29 -1
  7. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/config/cli.py +32 -31
  8. pdf_to_markdown_cli-0.4.0/src/docs_to_md/config/settings.py +54 -0
  9. pdf_to_markdown_cli-0.4.0/src/docs_to_md/core/paths.py +80 -0
  10. pdf_to_markdown_cli-0.4.0/src/docs_to_md/core/processor.py +344 -0
  11. pdf_to_markdown_cli-0.4.0/src/docs_to_md/core/result_handler.py +601 -0
  12. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/main.py +20 -11
  13. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/storage/cache.py +2 -2
  14. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/storage/models.py +5 -2
  15. pdf_to_markdown_cli-0.4.0/src/docs_to_md/utils/file_utils.py +219 -0
  16. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/utils/logging.py +51 -39
  17. pdf_to_markdown_cli-0.2.0/docs_to_md/pdf/splitter.py → pdf_to_markdown_cli-0.4.0/src/docs_to_md/utils/pdf_splitter.py +3 -35
  18. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src/pdf_to_markdown_cli.egg-info}/PKG-INFO +61 -38
  19. pdf_to_markdown_cli-0.4.0/src/pdf_to_markdown_cli.egg-info/SOURCES.txt +31 -0
  20. pdf_to_markdown_cli-0.2.0/docs_to_md/api/client.py +0 -218
  21. pdf_to_markdown_cli-0.2.0/docs_to_md/config/settings.py +0 -108
  22. pdf_to_markdown_cli-0.2.0/docs_to_md/core/processor.py +0 -229
  23. pdf_to_markdown_cli-0.2.0/docs_to_md/core/result_handler.py +0 -496
  24. pdf_to_markdown_cli-0.2.0/docs_to_md/utils/__init__.py +0 -0
  25. pdf_to_markdown_cli-0.2.0/docs_to_md/utils/file_utils.py +0 -189
  26. pdf_to_markdown_cli-0.2.0/pdf_to_markdown_cli.egg-info/SOURCES.txt +0 -31
  27. pdf_to_markdown_cli-0.2.0/setup.py +0 -67
  28. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0}/LICENSE +0 -0
  29. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0}/setup.cfg +0 -0
  30. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/__init__.py +0 -0
  31. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/__main__.py +0 -0
  32. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/api/__init__.py +0 -0
  33. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/config/__init__.py +0 -0
  34. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/core/__init__.py +0 -0
  35. {pdf_to_markdown_cli-0.2.0/docs_to_md/pdf → pdf_to_markdown_cli-0.4.0/src/docs_to_md/storage}/__init__.py +0 -0
  36. {pdf_to_markdown_cli-0.2.0/docs_to_md/storage → pdf_to_markdown_cli-0.4.0/src/docs_to_md/utils}/__init__.py +0 -0
  37. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/docs_to_md/utils/exceptions.py +0 -0
  38. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/pdf_to_markdown_cli.egg-info/dependency_links.txt +0 -0
  39. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/pdf_to_markdown_cli.egg-info/entry_points.txt +0 -0
  40. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/pdf_to_markdown_cli.egg-info/requires.txt +0 -0
  41. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.4.0/src}/pdf_to_markdown_cli.egg-info/top_level.txt +0 -0
@@ -2,8 +2,8 @@
2
2
  include README.md
3
3
  include LICENSE
4
4
 
5
- # Include the package source directory
6
- recursive-include docs_to_md *
5
+ # Include the package source directory (now within src/)
6
+ recursive-include src/docs_to_md *
7
7
 
8
8
  # Exclude development/cache/temporary files and directories
9
9
  global-exclude *.py[cod]
@@ -1,9 +1,31 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pdf-to-markdown-cli
3
- Version: 0.2.0
3
+ Version: 0.4.0
4
4
  Summary: CLI tool to convert PDF files (and other documents) to markdown using the Marker API.
5
- Home-page: https://github.com/SokolskyNikita/pdf-to-markdown-cli
6
5
  Author: Nikita Sokolsky
6
+ License: MIT License
7
+
8
+ Copyright (c) 2025 SokolskyNikita
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Project-URL: Homepage, https://github.com/SokolskyNikita/pdf-to-markdown-cli
7
29
  Project-URL: Bug Tracker, https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues
8
30
  Keywords: pdf,markdown,converter,cli,document,marker,md
9
31
  Classifier: Development Status :: 4 - Beta
@@ -28,21 +50,11 @@ Requires-Dist: pydantic>=2.0
28
50
  Requires-Dist: ratelimit>=2.0
29
51
  Requires-Dist: requests>=2.0
30
52
  Requires-Dist: tqdm>=4.0
31
- Dynamic: author
32
- Dynamic: classifier
33
- Dynamic: description
34
- Dynamic: description-content-type
35
- Dynamic: home-page
36
- Dynamic: keywords
37
53
  Dynamic: license-file
38
- Dynamic: project-url
39
- Dynamic: requires-dist
40
- Dynamic: requires-python
41
- Dynamic: summary
42
54
 
43
- # PDF to Markdown CLI (using Marker API)
55
+ # PDF to Markdown CLI (via the Datalab Marker API)
44
56
 
45
- Convert PDF files (and other documents) to markdown using the [Marker API](https://www.marker.io/) via a command-line tool.
57
+ Convert PDF files (and other documents) to Markdown using the [Marker API](https://www.datalab.to/marker) via a convenient CLI tool.
46
58
 
47
59
  ## Overview
48
60
 
@@ -50,13 +62,12 @@ This package provides a convenient command-line interface (`pdf-to-md`) for conv
50
62
 
51
63
  ## Features
52
64
 
53
- - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to markdown
54
- - Optionally use the Marker API for enhanced PDF conversion
55
- - Handle large documents by splitting them into chunks
65
+ - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to Markdown using the best-in-class Marker API by Datalab
66
+ - Handle large documents by splitting them into chunks, which massively speeds up output speed
56
67
  - Progress tracking for long-running operations
57
- - Customizable OCR options
58
- - Local caching of results
59
- - Output in markdown, JSON, or HTML format
68
+ - Customizable OCR options, fully reflecting Marker's API as of April 2025
69
+ - Local caching of in-progress conversions, allowing for idempotence
70
+ - Output in Markdown, JSON, or HTML format
60
71
 
61
72
  ## Installation
62
73
 
@@ -79,7 +90,7 @@ pip install -e .
79
90
  ### Command-line interface
80
91
 
81
92
  ```bash
82
- # Set your Marker API key
93
+ # Obtain an API key by signing up on https://www.datalab.to/marker
83
94
  export MARKER_PDF_KEY=your_api_key_here
84
95
 
85
96
  # Basic usage
@@ -98,6 +109,31 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
98
109
  pdf-to-md /path/to/file.pdf --max
99
110
  ```
100
111
 
112
+ ### Full list of CLI options
113
+
114
+ - `input`: Input file or directory path
115
+ - `--json`: Output in JSON format (default is markdown)
116
+ - `--langs`: Comma-separated OCR languages (default: "English")
117
+ - `--llm`: Use LLM for enhanced processing
118
+ - `--strip`: Redo OCR processing
119
+ - `--noimg`: Disable image extraction
120
+ - `--force`: Force OCR on all pages
121
+ - `--pages`: Add page delimiters
122
+ - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
123
+ - `-mp`, `--max-pages`: Maximum number of pages to process from the start of the file
124
+ - `--no-chunk`: Disable PDF chunking
125
+ - `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
126
+ - `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
127
+ - `-v`, `--verbose`: Enable verbose (DEBUG level) logging
128
+
129
+ ### Output Structure
130
+
131
+ By default, output files are saved in the same directory as the input file with the format `[input-filename].[format]`. For example, converting `/data/report.pdf` to markdown will result in `/data/report.md`.
132
+
133
+ If an output file with the same name already exists, the new file will be automatically renamed using a numeric suffix (e.g., `[input-filename]_1.[format]`, `[input-filename]_2.[format]`, etc.) to avoid overwriting.
134
+
135
+ If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
136
+
101
137
  ### API
102
138
 
103
139
  ```python
@@ -118,23 +154,6 @@ processor = MarkerProcessor(config)
118
154
  processor.process()
119
155
  ```
120
156
 
121
- ## Command-line Options
122
-
123
- - `input`: Input file or directory path
124
- - `--json`: Output in JSON format (default is markdown)
125
- - `--langs`: Comma-separated OCR languages (default: "English")
126
- - `--llm`: Use LLM for enhanced processing
127
- - `--strip`: Redo OCR processing
128
- - `--noimg`: Disable image extraction
129
- - `--force`: Force OCR on all pages
130
- - `--pages`: Add page delimiters
131
- - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
132
- - `--max-pages`: Maximum number of pages to process from the start of the file
133
- - `--no-chunk`: Disable PDF chunking
134
- - `--chunk-size`: Set PDF chunk size in pages (default: 25)
135
- - `--output-dir`: Output directory (default: "converted")
136
- - `--cache-dir`: Cache directory (default: ".marker_cache")
137
-
138
157
  ## Project Structure
139
158
 
140
159
  The package is organized as follows:
@@ -177,3 +196,7 @@ For regular use after installation, use the `pdf-to-md` command.
177
196
  ## License
178
197
 
179
198
  This project is licensed under the MIT License - see the LICENSE file for details.
199
+
200
+ ## Future Considerations
201
+
202
+ While this project currently uses a modern structure (`pyproject.toml`, `src` layout), future development might involve migrating to a standardized project template, such as [simonw/python-lib](https://github.com/simonw/python-lib), to further align with community best practices and potentially simplify workflows like automated PyPI publishing via Trusted Publishers.
@@ -1,6 +1,6 @@
1
- # PDF to Markdown CLI (using Marker API)
1
+ # PDF to Markdown CLI (via the Datalab Marker API)
2
2
 
3
- Convert PDF files (and other documents) to markdown using the [Marker API](https://www.marker.io/) via a command-line tool.
3
+ Convert PDF files (and other documents) to Markdown using the [Marker API](https://www.datalab.to/marker) via a convenient CLI tool.
4
4
 
5
5
  ## Overview
6
6
 
@@ -8,13 +8,12 @@ This package provides a convenient command-line interface (`pdf-to-md`) for conv
8
8
 
9
9
  ## Features
10
10
 
11
- - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to markdown
12
- - Optionally use the Marker API for enhanced PDF conversion
13
- - Handle large documents by splitting them into chunks
11
+ - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to Markdown using the best-in-class Marker API by Datalab
12
+ - Handle large documents by splitting them into chunks, which massively speeds up output speed
14
13
  - Progress tracking for long-running operations
15
- - Customizable OCR options
16
- - Local caching of results
17
- - Output in markdown, JSON, or HTML format
14
+ - Customizable OCR options, fully reflecting Marker's API as of April 2025
15
+ - Local caching of in-progress conversions, allowing for idempotence
16
+ - Output in Markdown, JSON, or HTML format
18
17
 
19
18
  ## Installation
20
19
 
@@ -37,7 +36,7 @@ pip install -e .
37
36
  ### Command-line interface
38
37
 
39
38
  ```bash
40
- # Set your Marker API key
39
+ # Obtain an API key by signing up on https://www.datalab.to/marker
41
40
  export MARKER_PDF_KEY=your_api_key_here
42
41
 
43
42
  # Basic usage
@@ -56,6 +55,31 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
56
55
  pdf-to-md /path/to/file.pdf --max
57
56
  ```
58
57
 
58
+ ### Full list of CLI options
59
+
60
+ - `input`: Input file or directory path
61
+ - `--json`: Output in JSON format (default is markdown)
62
+ - `--langs`: Comma-separated OCR languages (default: "English")
63
+ - `--llm`: Use LLM for enhanced processing
64
+ - `--strip`: Redo OCR processing
65
+ - `--noimg`: Disable image extraction
66
+ - `--force`: Force OCR on all pages
67
+ - `--pages`: Add page delimiters
68
+ - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
69
+ - `-mp`, `--max-pages`: Maximum number of pages to process from the start of the file
70
+ - `--no-chunk`: Disable PDF chunking
71
+ - `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
72
+ - `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
73
+ - `-v`, `--verbose`: Enable verbose (DEBUG level) logging
74
+
75
+ ### Output Structure
76
+
77
+ By default, output files are saved in the same directory as the input file with the format `[input-filename].[format]`. For example, converting `/data/report.pdf` to markdown will result in `/data/report.md`.
78
+
79
+ If an output file with the same name already exists, the new file will be automatically renamed using a numeric suffix (e.g., `[input-filename]_1.[format]`, `[input-filename]_2.[format]`, etc.) to avoid overwriting.
80
+
81
+ If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
82
+
59
83
  ### API
60
84
 
61
85
  ```python
@@ -76,23 +100,6 @@ processor = MarkerProcessor(config)
76
100
  processor.process()
77
101
  ```
78
102
 
79
- ## Command-line Options
80
-
81
- - `input`: Input file or directory path
82
- - `--json`: Output in JSON format (default is markdown)
83
- - `--langs`: Comma-separated OCR languages (default: "English")
84
- - `--llm`: Use LLM for enhanced processing
85
- - `--strip`: Redo OCR processing
86
- - `--noimg`: Disable image extraction
87
- - `--force`: Force OCR on all pages
88
- - `--pages`: Add page delimiters
89
- - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
90
- - `--max-pages`: Maximum number of pages to process from the start of the file
91
- - `--no-chunk`: Disable PDF chunking
92
- - `--chunk-size`: Set PDF chunk size in pages (default: 25)
93
- - `--output-dir`: Output directory (default: "converted")
94
- - `--cache-dir`: Cache directory (default: ".marker_cache")
95
-
96
103
  ## Project Structure
97
104
 
98
105
  The package is organized as follows:
@@ -134,4 +141,8 @@ For regular use after installation, use the `pdf-to-md` command.
134
141
 
135
142
  ## License
136
143
 
137
- This project is licensed under the MIT License - see the LICENSE file for details.
144
+ This project is licensed under the MIT License - see the LICENSE file for details.
145
+
146
+ ## Future Considerations
147
+
148
+ While this project currently uses a modern structure (`pyproject.toml`, `src` layout), future development might involve migrating to a standardized project template, such as [simonw/python-lib](https://github.com/simonw/python-lib), to further align with community best practices and potentially simplify workflows like automated PyPI publishing via Trusted Publishers.
@@ -0,0 +1,51 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "pdf-to-markdown-cli"
7
+ version = "0.4.0"
8
+ description = "CLI tool to convert PDF files (and other documents) to markdown using the Marker API."
9
+ readme = "README.md"
10
+ authors = [
11
+ { name = "Nikita Sokolsky" },
12
+ ]
13
+ license = { file = "LICENSE" }
14
+ requires-python = ">=3.10"
15
+ classifiers = [
16
+ "Development Status :: 4 - Beta",
17
+ "Intended Audience :: Developers",
18
+ "Intended Audience :: End Users/Desktop",
19
+ "License :: OSI Approved :: MIT License",
20
+ "Programming Language :: Python :: 3",
21
+ "Programming Language :: Python :: 3.10",
22
+ "Programming Language :: Python :: 3.11",
23
+ "Programming Language :: Python :: 3.12",
24
+ "Operating System :: OS Independent",
25
+ "Topic :: Text Processing",
26
+ "Topic :: Utilities",
27
+ ]
28
+ keywords = ["pdf", "markdown", "converter", "cli", "document", "marker", "md"]
29
+ dependencies = [
30
+ "backoff>=2.0",
31
+ "diskcache>=5.0",
32
+ "filetype>=1.0",
33
+ "pikepdf>=8.0",
34
+ "pydantic>=2.0",
35
+ "ratelimit>=2.0",
36
+ "requests>=2.0",
37
+ "tqdm>=4.0",
38
+ ]
39
+
40
+ # Define project URLs
41
+ [project.urls]
42
+ "Homepage" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli"
43
+ "Bug Tracker" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues"
44
+
45
+ # Define console script entry point
46
+ [project.scripts]
47
+ pdf-to-md = "docs_to_md.main:main"
48
+
49
+ # Just in case: Configure setuptools to find packages in src/
50
+ [tool.setuptools.packages.find]
51
+ where = ["src"]
@@ -0,0 +1,183 @@
1
+ import json
2
+ import logging
3
+ from pathlib import Path
4
+ from typing import Optional
5
+
6
+ import backoff
7
+ import filetype
8
+ import requests
9
+ from ratelimit import limits, sleep_and_retry
10
+
11
+ from docs_to_md.api.models import (
12
+ MarkerStatus,
13
+ StatusEnum,
14
+ SubmitResponse,
15
+ SUPPORTED_MIME_TYPES,
16
+ )
17
+ from docs_to_md.utils.exceptions import APIError
18
+ from docs_to_md.utils.file_utils import FileIO
19
+
20
+ # Client-side constants
21
+ MAX_REQUESTS_PER_MINUTE = 150
22
+ REQUEST_TIMEOUT_SECONDS = 30
23
+ MAX_RETRIES = 3
24
+
25
+ logger = logging.getLogger(__name__)
26
+
27
+
28
+ class MarkerClient:
29
+ BASE_MARKER_API_ENDPOINT = "https://www.datalab.to/api/v1/marker"
30
+
31
+ # See datalab_marker_api_docs.md#authentication for API key details
32
+ def __init__(self, api_key: str):
33
+ if not api_key or not api_key.strip():
34
+ raise APIError("API key is required")
35
+
36
+ self.headers = {"X-Api-Key": api_key.strip()}
37
+
38
+ @sleep_and_retry
39
+ @limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
40
+ @backoff.on_exception(
41
+ backoff.expo,
42
+ (requests.exceptions.RequestException, json.JSONDecodeError),
43
+ max_tries=MAX_RETRIES,
44
+ )
45
+ def submit_file(
46
+ self,
47
+ file_path: Path,
48
+ output_format: str = "markdown",
49
+ langs: str = "English",
50
+ use_llm: bool = False,
51
+ strip_existing_ocr: bool = False,
52
+ disable_image_extraction: bool = False,
53
+ force_ocr: bool = False,
54
+ paginate: bool = False,
55
+ max_pages: Optional[int] = None,
56
+ ) -> Optional[str]:
57
+ """Submit a file for conversion via the Marker API.
58
+ See datalab_marker_api_docs.md#marker for API parameter details.
59
+ """
60
+ try:
61
+ if not file_path.exists():
62
+ raise APIError(f"File not found: {file_path}")
63
+
64
+ file_data = FileIO.read_file(file_path)
65
+ kind = filetype.guess(file_data)
66
+
67
+ # Supported types listed in datalab_marker_api_docs.md#supported-file-types
68
+ if not kind or kind.mime not in SUPPORTED_MIME_TYPES:
69
+ raise APIError(
70
+ f"Unsupported file type: {kind.mime if kind else 'unknown'}"
71
+ )
72
+
73
+ form_data = {
74
+ "file": (file_path.name, file_data, kind.mime),
75
+ "langs": (None, langs),
76
+ "force_ocr": (None, force_ocr),
77
+ "paginate": (None, paginate),
78
+ "strip_existing_ocr": (None, strip_existing_ocr),
79
+ "disable_image_extraction": (None, disable_image_extraction),
80
+ "use_llm": (None, use_llm),
81
+ "output_format": (None, output_format),
82
+ "max_pages": (None, max_pages),
83
+ }
84
+
85
+ response = requests.post(
86
+ self.BASE_MARKER_API_ENDPOINT,
87
+ files=form_data,
88
+ headers=self.headers,
89
+ timeout=REQUEST_TIMEOUT_SECONDS,
90
+ )
91
+ response.raise_for_status() # Default handling for HTTP errors,
92
+ submit_response = SubmitResponse.model_validate(response.json())
93
+
94
+ if not submit_response.success:
95
+ logger.error(
96
+ f"API request failed: {submit_response.error or 'Unknown error'}"
97
+ )
98
+ return None
99
+
100
+ logger.info(
101
+ f"Successfully submitted file {file_path.name}. Request ID: {submit_response.request_id}"
102
+ )
103
+ return submit_response.request_id
104
+
105
+ except Exception as e:
106
+ logger.error(f"Error submitting file {file_path}: {e}")
107
+ return None
108
+
109
+ def _handle_status_error(
110
+ self, status_code: int, request_id: str
111
+ ) -> Optional[MarkerStatus]:
112
+ """Handle non-200 status codes from the check_status endpoint."""
113
+ logger.error(f"API returned status code {status_code} for request {request_id}")
114
+ # Specific handling for non-fatal polling errors
115
+ if status_code == 404:
116
+ # Treat not found as still processing, might appear later
117
+ return MarkerStatus(status=StatusEnum.PROCESSING, error="Request not found")
118
+ elif status_code == 401:
119
+ return MarkerStatus(status=StatusEnum.FAILED, error="Authentication failed")
120
+ elif status_code == 429:
121
+ # Treat rate limit as still processing, should retry later
122
+ return MarkerStatus(
123
+ status=StatusEnum.PROCESSING, error="Rate limit exceeded"
124
+ )
125
+ # For other non-200 errors, return None to indicate failure to get status
126
+ return None
127
+
128
+ @sleep_and_retry
129
+ @limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
130
+ @backoff.on_exception(
131
+ backoff.expo,
132
+ (requests.exceptions.RequestException, json.JSONDecodeError),
133
+ max_tries=MAX_RETRIES,
134
+ )
135
+ def check_status(self, request_id: str) -> Optional[MarkerStatus]:
136
+ """
137
+ Check the status of a conversion request.
138
+ See datalab_marker_api_docs.md#marker for polling details.
139
+
140
+ Returns:
141
+ MarkerStatus object with current status, or None if the check fails.
142
+ """
143
+ if not request_id:
144
+ logger.error("Empty Marker request ID provided, skipping status check")
145
+ return None
146
+
147
+ try:
148
+ response = requests.get(
149
+ f"{self.BASE_MARKER_API_ENDPOINT}/{request_id}",
150
+ headers=self.headers,
151
+ timeout=REQUEST_TIMEOUT_SECONDS,
152
+ )
153
+
154
+ if response.status_code != 200:
155
+ return self._handle_status_error(response.status_code, request_id)
156
+
157
+ data = response.json()
158
+ # Handle potential empty response from API
159
+ if not data:
160
+ logger.error(f"Empty response for request {request_id}")
161
+ return None
162
+
163
+ status = MarkerStatus.model_validate(data)
164
+ return status
165
+
166
+ except json.JSONDecodeError as e:
167
+ logger.error(f"Invalid JSON response for request {request_id}: {e}")
168
+ return None
169
+ except (
170
+ requests.exceptions.RequestException
171
+ ) as e: # Consolidated network/request errors
172
+ logger.error(f"Request error checking status for {request_id}: {e}")
173
+ return None
174
+ except Exception as e: # Catch-all for validation or other unexpected errors
175
+ logger.error(f"Unexpected error checking status for {request_id}: {e}")
176
+ return None
177
+
178
+ def __enter__(self):
179
+ return self
180
+
181
+ def __exit__(self, exc_type, exc_val, exc_tb):
182
+ # No specific cleanup needed for this client
183
+ pass
@@ -44,9 +44,37 @@ class ApiParams:
44
44
  force_ocr: bool = False
45
45
  paginate: bool = False
46
46
  max_pages: Optional[int] = None
47
+
48
+ # Map of supported output formats to their extensions
49
+ SUPPORTED_FORMAT_EXTENSIONS = {
50
+ "markdown": ".md",
51
+ "json": ".json",
52
+ "html": ".html",
53
+ "txt": ".txt"
54
+ }
47
55
 
56
+ SUPPORTED_IMAGE_EXTENSIONS: Set[str] = {
57
+ "jpg",
58
+ "jpeg",
59
+ "png",
60
+ "gif",
61
+ "tiff"
62
+ }
48
63
 
49
- # Supported mime types according to API docs
64
+ SUPPORTED_INPUT_EXTENSIONS: Set[str] = {
65
+ "pdf",
66
+ "docx",
67
+ "doc",
68
+ "pptx",
69
+ "ppt",
70
+ "jpg",
71
+ "jpeg",
72
+ "png",
73
+ "gif",
74
+ "tiff"
75
+ }
76
+
77
+ # Supported mime types according to datalab_marker_api_docs.md#supported-file-types
50
78
  SUPPORTED_MIME_TYPES: Set[str] = {
51
79
  # PDF
52
80
  'application/pdf',
@@ -1,61 +1,57 @@
1
1
  import argparse
2
- from pathlib import Path
2
+ from pathlib import Path
3
+ import os
4
+ from typing import Optional
5
+ import importlib.metadata
3
6
 
4
7
  from docs_to_md.config.settings import Config
5
- from docs_to_md.utils.exceptions import ConfigurationError
6
- from docs_to_md.utils.file_utils import get_env_var
8
+ from docs_to_md.utils.exceptions import ConfigurationError, FileError
7
9
 
8
10
 
9
11
  def parse_args() -> argparse.Namespace:
10
- """
11
- Parse command line arguments.
12
-
13
- Returns:
14
- Parsed arguments namespace
15
- """
12
+ # Get package version dynamically
13
+ try:
14
+ __version__ = importlib.metadata.version('pdf-to-markdown-cli')
15
+ except importlib.metadata.PackageNotFoundError:
16
+ __version__ = 'unknown' # Fallback if package not installed
17
+
16
18
  parser = argparse.ArgumentParser(
17
19
  description="Process PDF files using Marker API.",
18
20
  formatter_class=argparse.ArgumentDefaultsHelpFormatter
19
21
  )
20
22
 
21
- # Required arguments
23
+ # Add version argument
24
+ parser.add_argument(
25
+ '--version',
26
+ action='version',
27
+ version=f'pdf-to-markdown-cli version: {__version__}'
28
+ )
29
+
22
30
  parser.add_argument("input", help="Input file or directory path")
23
31
 
24
- # Output format
25
32
  parser.add_argument("--json", action="store_true", help="Output in JSON format")
26
33
 
27
- # OCR settings
28
- parser.add_argument("--langs", default="English", help="Comma-separated OCR languages")
34
+ parser.add_argument("-l", "--langs", default="English", help="Comma-separated OCR languages")
29
35
  parser.add_argument("--llm", action="store_true", help="Use LLM for enhanced processing")
30
36
  parser.add_argument("--strip", action="store_true", help="Redo OCR processing")
31
37
  parser.add_argument("--noimg", action="store_true", help="Disable image extraction")
32
38
  parser.add_argument("--force", action="store_true", help="Force OCR on all pages")
33
39
  parser.add_argument("--pages", action="store_true", help="Add page delimiters")
34
- parser.add_argument("--max-pages", type=int, help="Maximum number of pages to process from the start of the file")
40
+ parser.add_argument("-mp", "--max-pages", type=int, help="Maximum number of pages to process from the start of the file")
35
41
 
36
- # Advanced settings
37
42
  parser.add_argument("--max", action="store_true", help="Enable all OCR enhancements (LLM, strip OCR, force OCR)")
38
43
  parser.add_argument("--no-chunk", action="store_true", help="Disable PDF chunking (sets chunk size to 1 million)")
39
44
  parser.add_argument("-cs", "--chunk-size", type=int, help="Set PDF chunk size in pages", default=25)
40
- parser.add_argument("--output-dir", help="Output directory", default="converted")
41
- parser.add_argument("--cache-dir", help="Cache directory", default=".docs_to_md_cache")
45
+ parser.add_argument("-o", "--output-dir", help="Absolute path to the output directory (default: same directory as input file)", default=None)
46
+
47
+ parser.add_argument("-v", "--verbose", action="store_true", help="Enable verbose (DEBUG level) logging")
42
48
 
43
49
  return parser.parse_args()
44
50
 
45
51
 
46
52
  def create_config_from_args() -> Config:
47
- """
48
- Create configuration from command line arguments.
49
-
50
- Returns:
51
- Config object with settings from command line
52
-
53
- Raises:
54
- ConfigurationError: If required arguments are missing
55
- """
56
53
  args = parse_args()
57
54
 
58
- # Get API key from environment
59
55
  try:
60
56
  api_key = get_env_var("MARKER_PDF_KEY")
61
57
  except Exception as e:
@@ -67,8 +63,7 @@ def create_config_from_args() -> Config:
67
63
  config = Config(
68
64
  api_key=api_key,
69
65
  input_path=args.input,
70
- output_dir=Path(args.output_dir),
71
- cache_dir=Path(args.cache_dir),
66
+ output_dir=Path(args.output_dir) if args.output_dir else None,
72
67
  output_format="json" if args.json else "markdown",
73
68
  langs=args.langs,
74
69
  use_llm=args.llm or args.max,
@@ -80,7 +75,13 @@ def create_config_from_args() -> Config:
80
75
  max_pages=args.max_pages
81
76
  )
82
77
 
83
- # Validate the configuration
84
78
  config.validate()
85
79
 
86
- return config
80
+ return config
81
+
82
+ def get_env_var(name: str, required: bool = True) -> Optional[str]:
83
+ """Get environment variable with optional requirement."""
84
+ value = os.getenv(name)
85
+ if required and not value:
86
+ raise FileError(f"Required environment variable {name} is not set")
87
+ return value