pdf-to-markdown-cli 0.5.1__tar.gz → 0.5.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pdf_to_markdown_cli-0.5.2/CHANGELOG.md +25 -0
- pdf_to_markdown_cli-0.5.2/CONTRIBUTING.md +38 -0
- pdf_to_markdown_cli-0.5.2/MANIFEST.in +13 -0
- pdf_to_markdown_cli-0.5.2/PKG-INFO +143 -0
- pdf_to_markdown_cli-0.5.2/README.md +95 -0
- pdf_to_markdown_cli-0.5.2/pyproject.toml +74 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/api/client.py +32 -4
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/api/models.py +32 -14
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/config/cli.py +13 -5
- pdf_to_markdown_cli-0.5.2/src/docs_to_md/config/settings.py +99 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/core/paths.py +17 -21
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/core/processor.py +2 -3
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/core/result_handler.py +188 -159
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/storage/cache.py +1 -3
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/storage/models.py +36 -21
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/logging.py +8 -3
- pdf_to_markdown_cli-0.5.2/src/pdf_to_markdown_cli.egg-info/PKG-INFO +143 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/SOURCES.txt +2 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/requires.txt +8 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_cli.py +17 -2
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_cli_equations.py +33 -19
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_settings.py +26 -0
- pdf_to_markdown_cli-0.5.1/MANIFEST.in +0 -24
- pdf_to_markdown_cli-0.5.1/PKG-INFO +0 -119
- pdf_to_markdown_cli-0.5.1/README.md +0 -64
- pdf_to_markdown_cli-0.5.1/pyproject.toml +0 -54
- pdf_to_markdown_cli-0.5.1/src/docs_to_md/config/settings.py +0 -54
- pdf_to_markdown_cli-0.5.1/src/pdf_to_markdown_cli.egg-info/PKG-INFO +0 -119
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/LICENSE +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/setup.cfg +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/__init__.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/__main__.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/api/__init__.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/config/__init__.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/core/__init__.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/main.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/storage/__init__.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/__init__.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/exceptions.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/file_utils.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/pdf_splitter.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/dependency_links.txt +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/entry_points.txt +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/top_level.txt +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_paths.py +0 -0
- {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_utils.py +0 -0
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project are documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project follows [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
|
|
8
|
+
## [Unreleased]
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
|
|
12
|
+
- Added fallback handling for non-writable cache/tmp directories.
|
|
13
|
+
- Added support tests for HTML output CLI configuration and both example PDFs.
|
|
14
|
+
|
|
15
|
+
### Changed
|
|
16
|
+
|
|
17
|
+
- Expanded supported input MIME/extension mappings for document and image formats.
|
|
18
|
+
- Improved result polling with per-chunk exponential backoff and cleaner logging.
|
|
19
|
+
- Modernized packaging and project metadata in `pyproject.toml`.
|
|
20
|
+
|
|
21
|
+
### Fixed
|
|
22
|
+
|
|
23
|
+
- Fixed false "unsupported file type" failures by adding MIME detection fallbacks.
|
|
24
|
+
- Fixed cache retrieval edge case where empty payloads were treated as missing.
|
|
25
|
+
- Fixed request state tracking semantics for completion/failure handling.
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Thanks for contributing to `pdf-to-markdown-cli`.
|
|
4
|
+
|
|
5
|
+
## Development setup
|
|
6
|
+
|
|
7
|
+
1. Fork and clone the repository.
|
|
8
|
+
2. Create a virtual environment.
|
|
9
|
+
3. Install in editable mode:
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install -e .
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
4. Run tests before submitting changes:
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
python -m unittest discover -s tests -v
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
## Pull request checklist
|
|
22
|
+
|
|
23
|
+
- Keep changes focused and explain the motivation.
|
|
24
|
+
- Add or update tests when behavior changes.
|
|
25
|
+
- Keep CLI behavior backward-compatible unless the change is intentional.
|
|
26
|
+
- Update docs (`README.md`, examples, or this guide) when user-facing behavior changes.
|
|
27
|
+
- Ensure all tests pass locally.
|
|
28
|
+
|
|
29
|
+
## Coding guidelines
|
|
30
|
+
|
|
31
|
+
- Target Python 3.10+.
|
|
32
|
+
- Prefer clear, small functions with explicit error handling.
|
|
33
|
+
- Avoid silent failures; log contextual errors where useful.
|
|
34
|
+
- Do not commit API keys, credentials, or private documents.
|
|
35
|
+
|
|
36
|
+
## Reporting issues
|
|
37
|
+
|
|
38
|
+
Use GitHub issues for bug reports and feature requests.
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Include the README and LICENSE
|
|
2
|
+
include README.md
|
|
3
|
+
include LICENSE
|
|
4
|
+
include CHANGELOG.md
|
|
5
|
+
include CONTRIBUTING.md
|
|
6
|
+
|
|
7
|
+
# Include package source
|
|
8
|
+
recursive-include src/docs_to_md *.py
|
|
9
|
+
|
|
10
|
+
# Exclude generated files
|
|
11
|
+
global-exclude *.py[cod]
|
|
12
|
+
global-exclude __pycache__
|
|
13
|
+
global-exclude .DS_Store
|
|
@@ -0,0 +1,143 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pdf-to-markdown-cli
|
|
3
|
+
Version: 0.5.2
|
|
4
|
+
Summary: CLI utility to convert PDFs and supported document formats to Markdown/JSON/HTML with Marker API.
|
|
5
|
+
Author-email: Nikita Sokolsky <sokolx@gmail.com>
|
|
6
|
+
Maintainer-email: Nikita Sokolsky <sokolx@gmail.com>
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
Project-URL: Homepage, https://github.com/SokolskyNikita/pdf-to-markdown-cli
|
|
9
|
+
Project-URL: Repository, https://github.com/SokolskyNikita/pdf-to-markdown-cli
|
|
10
|
+
Project-URL: Documentation, https://github.com/SokolskyNikita/pdf-to-markdown-cli#readme
|
|
11
|
+
Project-URL: Issues, https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues
|
|
12
|
+
Project-URL: Changelog, https://github.com/SokolskyNikita/pdf-to-markdown-cli/releases
|
|
13
|
+
Keywords: pdf,markdown,converter,cli,document,marker,md
|
|
14
|
+
Classifier: Development Status :: 4 - Beta
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Operating System :: OS Independent
|
|
23
|
+
Classifier: Environment :: Console
|
|
24
|
+
Classifier: Topic :: Office/Business
|
|
25
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
26
|
+
Classifier: Topic :: Text Processing
|
|
27
|
+
Classifier: Topic :: Utilities
|
|
28
|
+
Requires-Python: >=3.10
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
License-File: LICENSE
|
|
31
|
+
Requires-Dist: backoff>=2.0
|
|
32
|
+
Requires-Dist: diskcache>=5.0
|
|
33
|
+
Requires-Dist: filetype>=1.0
|
|
34
|
+
Requires-Dist: pikepdf>=8.0
|
|
35
|
+
Requires-Dist: pydantic>=2.0
|
|
36
|
+
Requires-Dist: ratelimit>=2.0
|
|
37
|
+
Requires-Dist: requests>=2.0
|
|
38
|
+
Requires-Dist: tqdm>=4.0
|
|
39
|
+
Provides-Extra: test
|
|
40
|
+
Requires-Dist: pytest>=8.3.0; extra == "test"
|
|
41
|
+
Requires-Dist: pytest-cov>=5.0.0; extra == "test"
|
|
42
|
+
Provides-Extra: dev
|
|
43
|
+
Requires-Dist: build>=1.2.0; extra == "dev"
|
|
44
|
+
Requires-Dist: twine>=5.1.0; extra == "dev"
|
|
45
|
+
Requires-Dist: ruff>=0.11.0; extra == "dev"
|
|
46
|
+
Requires-Dist: types-requests>=2.32.0; extra == "dev"
|
|
47
|
+
Dynamic: license-file
|
|
48
|
+
|
|
49
|
+
# PDF to markdown CLI
|
|
50
|
+
|
|
51
|
+
[](https://pypi.org/project/pdf-to-markdown-cli/)
|
|
52
|
+
[](https://pypi.org/project/pdf-to-markdown-cli/)
|
|
53
|
+
[](LICENSE)
|
|
54
|
+
|
|
55
|
+
Command-line utility for converting PDFs and other supported documents into Markdown, JSON, or HTML using the [Marker API](https://www.datalab.to/marker).
|
|
56
|
+
|
|
57
|
+
## Why use this tool
|
|
58
|
+
|
|
59
|
+
- Converts single files or entire directories
|
|
60
|
+
- Automatically splits large PDFs into chunks and merges results
|
|
61
|
+
- Persists request state locally so interrupted runs can recover
|
|
62
|
+
- Rewrites and copies extracted images into deterministic output folders
|
|
63
|
+
- Supports OCR/LLM tuning flags from the Marker API
|
|
64
|
+
|
|
65
|
+
## Supported formats
|
|
66
|
+
|
|
67
|
+
### Input
|
|
68
|
+
|
|
69
|
+
- PDF (`.pdf`)
|
|
70
|
+
- Word (`.doc`, `.docx`, `.odt`)
|
|
71
|
+
- PowerPoint (`.ppt`, `.pptx`, `.odp`)
|
|
72
|
+
- Spreadsheets (`.xls`, `.xlsx`, `.ods`)
|
|
73
|
+
- EPUB/HTML (`.epub`, `.html`)
|
|
74
|
+
- Images (`.png`, `.jpg`, `.jpeg`, `.webp`, `.gif`, `.tiff`)
|
|
75
|
+
|
|
76
|
+
### Output
|
|
77
|
+
|
|
78
|
+
- Markdown (`.md`, default)
|
|
79
|
+
- JSON (`.json`)
|
|
80
|
+
- HTML (`.html`)
|
|
81
|
+
|
|
82
|
+
## Installation
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
pip install pdf-to-markdown-cli
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
From source:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
git clone https://github.com/SokolskyNikita/pdf-to-markdown-cli.git
|
|
92
|
+
cd pdf-to-markdown-cli
|
|
93
|
+
pip install -e .
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## Quick start
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
export MARKER_PDF_KEY="your_api_key"
|
|
100
|
+
pdf-to-md ./examples/equations.pdf
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
Process a directory:
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
pdf-to-md ./docs
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Use JSON or HTML output:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
pdf-to-md ./examples/equations.pdf --json
|
|
113
|
+
pdf-to-md ./examples/equations.pdf --html
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
## CLI options
|
|
117
|
+
|
|
118
|
+
- `input`: input file or directory path
|
|
119
|
+
- `--json`: output JSON instead of Markdown
|
|
120
|
+
- `--html`: output HTML instead of Markdown
|
|
121
|
+
- `--langs`: comma-separated OCR languages (default: `English`)
|
|
122
|
+
- `--llm`: enable LLM-enhanced processing
|
|
123
|
+
- `--strip`: redo OCR
|
|
124
|
+
- `--noimg`: disable image extraction
|
|
125
|
+
- `--force`: force OCR on all pages
|
|
126
|
+
- `--pages`: include page delimiters
|
|
127
|
+
- `--max`: enable all OCR enhancement flags (`--llm --strip --force`)
|
|
128
|
+
- `-mp`, `--max-pages`: process only the first N pages
|
|
129
|
+
- `--no-chunk`: disable PDF chunking
|
|
130
|
+
- `-cs`, `--chunk-size`: PDF pages per chunk (default: `25`)
|
|
131
|
+
- `-o`, `--output-dir`: absolute output directory path
|
|
132
|
+
- `-v`, `--verbose`: debug logging
|
|
133
|
+
- `--version`: show installed package version
|
|
134
|
+
|
|
135
|
+
## Development
|
|
136
|
+
|
|
137
|
+
Run tests:
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
python -m unittest discover -s tests -v
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
For contributions or questions, open a GitHub issue.
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
# PDF to markdown CLI
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/pdf-to-markdown-cli/)
|
|
4
|
+
[](https://pypi.org/project/pdf-to-markdown-cli/)
|
|
5
|
+
[](LICENSE)
|
|
6
|
+
|
|
7
|
+
Command-line utility for converting PDFs and other supported documents into Markdown, JSON, or HTML using the [Marker API](https://www.datalab.to/marker).
|
|
8
|
+
|
|
9
|
+
## Why use this tool
|
|
10
|
+
|
|
11
|
+
- Converts single files or entire directories
|
|
12
|
+
- Automatically splits large PDFs into chunks and merges results
|
|
13
|
+
- Persists request state locally so interrupted runs can recover
|
|
14
|
+
- Rewrites and copies extracted images into deterministic output folders
|
|
15
|
+
- Supports OCR/LLM tuning flags from the Marker API
|
|
16
|
+
|
|
17
|
+
## Supported formats
|
|
18
|
+
|
|
19
|
+
### Input
|
|
20
|
+
|
|
21
|
+
- PDF (`.pdf`)
|
|
22
|
+
- Word (`.doc`, `.docx`, `.odt`)
|
|
23
|
+
- PowerPoint (`.ppt`, `.pptx`, `.odp`)
|
|
24
|
+
- Spreadsheets (`.xls`, `.xlsx`, `.ods`)
|
|
25
|
+
- EPUB/HTML (`.epub`, `.html`)
|
|
26
|
+
- Images (`.png`, `.jpg`, `.jpeg`, `.webp`, `.gif`, `.tiff`)
|
|
27
|
+
|
|
28
|
+
### Output
|
|
29
|
+
|
|
30
|
+
- Markdown (`.md`, default)
|
|
31
|
+
- JSON (`.json`)
|
|
32
|
+
- HTML (`.html`)
|
|
33
|
+
|
|
34
|
+
## Installation
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
pip install pdf-to-markdown-cli
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
From source:
|
|
41
|
+
|
|
42
|
+
```bash
|
|
43
|
+
git clone https://github.com/SokolskyNikita/pdf-to-markdown-cli.git
|
|
44
|
+
cd pdf-to-markdown-cli
|
|
45
|
+
pip install -e .
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
## Quick start
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
export MARKER_PDF_KEY="your_api_key"
|
|
52
|
+
pdf-to-md ./examples/equations.pdf
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Process a directory:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
pdf-to-md ./docs
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Use JSON or HTML output:
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
pdf-to-md ./examples/equations.pdf --json
|
|
65
|
+
pdf-to-md ./examples/equations.pdf --html
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
## CLI options
|
|
69
|
+
|
|
70
|
+
- `input`: input file or directory path
|
|
71
|
+
- `--json`: output JSON instead of Markdown
|
|
72
|
+
- `--html`: output HTML instead of Markdown
|
|
73
|
+
- `--langs`: comma-separated OCR languages (default: `English`)
|
|
74
|
+
- `--llm`: enable LLM-enhanced processing
|
|
75
|
+
- `--strip`: redo OCR
|
|
76
|
+
- `--noimg`: disable image extraction
|
|
77
|
+
- `--force`: force OCR on all pages
|
|
78
|
+
- `--pages`: include page delimiters
|
|
79
|
+
- `--max`: enable all OCR enhancement flags (`--llm --strip --force`)
|
|
80
|
+
- `-mp`, `--max-pages`: process only the first N pages
|
|
81
|
+
- `--no-chunk`: disable PDF chunking
|
|
82
|
+
- `-cs`, `--chunk-size`: PDF pages per chunk (default: `25`)
|
|
83
|
+
- `-o`, `--output-dir`: absolute output directory path
|
|
84
|
+
- `-v`, `--verbose`: debug logging
|
|
85
|
+
- `--version`: show installed package version
|
|
86
|
+
|
|
87
|
+
## Development
|
|
88
|
+
|
|
89
|
+
Run tests:
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
python -m unittest discover -s tests -v
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
For contributions or questions, open a GitHub issue.
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=69.0", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "pdf-to-markdown-cli"
|
|
7
|
+
version = "0.5.2"
|
|
8
|
+
description = "CLI utility to convert PDFs and supported document formats to Markdown/JSON/HTML with Marker API."
|
|
9
|
+
readme = { file = "README.md", content-type = "text/markdown" }
|
|
10
|
+
authors = [
|
|
11
|
+
{ name = "Nikita Sokolsky", email = "sokolx@gmail.com" },
|
|
12
|
+
]
|
|
13
|
+
maintainers = [
|
|
14
|
+
{ name = "Nikita Sokolsky", email = "sokolx@gmail.com" },
|
|
15
|
+
]
|
|
16
|
+
license = "MIT"
|
|
17
|
+
license-files = ["LICENSE"]
|
|
18
|
+
requires-python = ">=3.10"
|
|
19
|
+
classifiers = [
|
|
20
|
+
"Development Status :: 4 - Beta",
|
|
21
|
+
"Intended Audience :: Developers",
|
|
22
|
+
"Intended Audience :: End Users/Desktop",
|
|
23
|
+
"Programming Language :: Python :: 3",
|
|
24
|
+
"Programming Language :: Python :: 3.10",
|
|
25
|
+
"Programming Language :: Python :: 3.11",
|
|
26
|
+
"Programming Language :: Python :: 3.12",
|
|
27
|
+
"Programming Language :: Python :: 3.13",
|
|
28
|
+
"Operating System :: OS Independent",
|
|
29
|
+
"Environment :: Console",
|
|
30
|
+
"Topic :: Office/Business",
|
|
31
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
32
|
+
"Topic :: Text Processing",
|
|
33
|
+
"Topic :: Utilities",
|
|
34
|
+
]
|
|
35
|
+
keywords = ["pdf", "markdown", "converter", "cli", "document", "marker", "md"]
|
|
36
|
+
dependencies = [
|
|
37
|
+
"backoff>=2.0",
|
|
38
|
+
"diskcache>=5.0",
|
|
39
|
+
"filetype>=1.0",
|
|
40
|
+
"pikepdf>=8.0",
|
|
41
|
+
"pydantic>=2.0",
|
|
42
|
+
"ratelimit>=2.0",
|
|
43
|
+
"requests>=2.0",
|
|
44
|
+
"tqdm>=4.0",
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
[project.urls]
|
|
48
|
+
"Homepage" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli"
|
|
49
|
+
"Repository" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli"
|
|
50
|
+
"Documentation" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli#readme"
|
|
51
|
+
"Issues" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues"
|
|
52
|
+
"Changelog" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli/releases"
|
|
53
|
+
|
|
54
|
+
[project.scripts]
|
|
55
|
+
pdf-to-md = "docs_to_md.main:main"
|
|
56
|
+
|
|
57
|
+
[project.optional-dependencies]
|
|
58
|
+
test = [
|
|
59
|
+
"pytest>=8.3.0",
|
|
60
|
+
"pytest-cov>=5.0.0",
|
|
61
|
+
]
|
|
62
|
+
dev = [
|
|
63
|
+
"build>=1.2.0",
|
|
64
|
+
"twine>=5.1.0",
|
|
65
|
+
"ruff>=0.11.0",
|
|
66
|
+
"types-requests>=2.32.0",
|
|
67
|
+
]
|
|
68
|
+
|
|
69
|
+
[tool.setuptools]
|
|
70
|
+
package-dir = { "" = "src" }
|
|
71
|
+
include-package-data = true
|
|
72
|
+
|
|
73
|
+
[tool.setuptools.packages.find]
|
|
74
|
+
where = ["src"]
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import json
|
|
2
2
|
import logging
|
|
3
|
+
import mimetypes
|
|
3
4
|
from pathlib import Path
|
|
4
5
|
from typing import Optional
|
|
5
6
|
|
|
@@ -27,6 +28,19 @@ logger = logging.getLogger(__name__)
|
|
|
27
28
|
|
|
28
29
|
class MarkerClient:
|
|
29
30
|
BASE_MARKER_API_ENDPOINT = "https://www.datalab.to/api/v1/marker"
|
|
31
|
+
_EXTENSION_TO_MIME = {
|
|
32
|
+
".doc": "application/msword",
|
|
33
|
+
".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
34
|
+
".odt": "application/vnd.oasis.opendocument.text",
|
|
35
|
+
".ppt": "application/vnd.ms-powerpoint",
|
|
36
|
+
".pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
|
37
|
+
".odp": "application/vnd.oasis.opendocument.presentation",
|
|
38
|
+
".xls": "application/vnd.ms-excel",
|
|
39
|
+
".xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
|
40
|
+
".ods": "application/vnd.oasis.opendocument.spreadsheet",
|
|
41
|
+
".epub": "application/epub+zip",
|
|
42
|
+
".html": "text/html",
|
|
43
|
+
}
|
|
30
44
|
|
|
31
45
|
# See datalab_marker_api_docs.md#authentication for API key details
|
|
32
46
|
def __init__(self, api_key: str):
|
|
@@ -35,6 +49,19 @@ class MarkerClient:
|
|
|
35
49
|
|
|
36
50
|
self.headers = {"X-Api-Key": api_key.strip()}
|
|
37
51
|
|
|
52
|
+
def _detect_mime_type(self, file_path: Path, file_data: bytes) -> Optional[str]:
|
|
53
|
+
"""Infer MIME type from file header first, then file extension."""
|
|
54
|
+
detected = filetype.guess(file_data)
|
|
55
|
+
if detected and detected.mime:
|
|
56
|
+
return detected.mime
|
|
57
|
+
|
|
58
|
+
suffix = file_path.suffix.lower()
|
|
59
|
+
if suffix in self._EXTENSION_TO_MIME:
|
|
60
|
+
return self._EXTENSION_TO_MIME[suffix]
|
|
61
|
+
|
|
62
|
+
guessed, _ = mimetypes.guess_type(file_path.name)
|
|
63
|
+
return guessed
|
|
64
|
+
|
|
38
65
|
@sleep_and_retry
|
|
39
66
|
@limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
|
|
40
67
|
@backoff.on_exception(
|
|
@@ -62,16 +89,17 @@ class MarkerClient:
|
|
|
62
89
|
raise APIError(f"File not found: {file_path}")
|
|
63
90
|
|
|
64
91
|
file_data = FileIO.read_file(file_path)
|
|
65
|
-
|
|
92
|
+
mime_type = self._detect_mime_type(file_path, file_data)
|
|
66
93
|
|
|
67
94
|
# Supported types listed in datalab_marker_api_docs.md#supported-file-types
|
|
68
|
-
if not
|
|
95
|
+
if not mime_type or mime_type not in SUPPORTED_MIME_TYPES:
|
|
69
96
|
raise APIError(
|
|
70
|
-
f"Unsupported file type
|
|
97
|
+
f"Unsupported file type for '{file_path.name}': "
|
|
98
|
+
f"{mime_type or 'unknown'}"
|
|
71
99
|
)
|
|
72
100
|
|
|
73
101
|
form_data = {
|
|
74
|
-
"file": (file_path.name, file_data,
|
|
102
|
+
"file": (file_path.name, file_data, mime_type),
|
|
75
103
|
"langs": (None, langs),
|
|
76
104
|
"output_format": (None, output_format),
|
|
77
105
|
}
|
|
@@ -50,45 +50,63 @@ SUPPORTED_FORMAT_EXTENSIONS = {
|
|
|
50
50
|
"markdown": ".md",
|
|
51
51
|
"json": ".json",
|
|
52
52
|
"html": ".html",
|
|
53
|
-
"txt": ".txt"
|
|
54
53
|
}
|
|
55
54
|
|
|
56
55
|
SUPPORTED_IMAGE_EXTENSIONS: Set[str] = {
|
|
57
56
|
"jpg",
|
|
58
57
|
"jpeg",
|
|
59
58
|
"png",
|
|
59
|
+
"webp",
|
|
60
60
|
"gif",
|
|
61
|
-
"tiff"
|
|
61
|
+
"tiff",
|
|
62
62
|
}
|
|
63
63
|
|
|
64
64
|
SUPPORTED_INPUT_EXTENSIONS: Set[str] = {
|
|
65
65
|
"pdf",
|
|
66
66
|
"docx",
|
|
67
67
|
"doc",
|
|
68
|
+
"odt",
|
|
68
69
|
"pptx",
|
|
69
70
|
"ppt",
|
|
71
|
+
"odp",
|
|
72
|
+
"xlsx",
|
|
73
|
+
"xls",
|
|
74
|
+
"ods",
|
|
75
|
+
"epub",
|
|
76
|
+
"html",
|
|
70
77
|
"jpg",
|
|
71
78
|
"jpeg",
|
|
72
79
|
"png",
|
|
80
|
+
"webp",
|
|
73
81
|
"gif",
|
|
74
|
-
"tiff"
|
|
82
|
+
"tiff",
|
|
75
83
|
}
|
|
76
84
|
|
|
77
85
|
# Supported mime types according to datalab_marker_api_docs.md#supported-file-types
|
|
78
86
|
SUPPORTED_MIME_TYPES: Set[str] = {
|
|
79
87
|
# PDF
|
|
80
|
-
|
|
88
|
+
"application/pdf",
|
|
81
89
|
# Word documents
|
|
82
|
-
|
|
83
|
-
|
|
90
|
+
"application/msword",
|
|
91
|
+
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
92
|
+
"application/vnd.oasis.opendocument.text",
|
|
84
93
|
# Powerpoint
|
|
85
|
-
|
|
86
|
-
|
|
94
|
+
"application/vnd.ms-powerpoint",
|
|
95
|
+
"application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
|
96
|
+
"application/vnd.oasis.opendocument.presentation",
|
|
97
|
+
# Spreadsheets
|
|
98
|
+
"application/vnd.ms-excel",
|
|
99
|
+
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
|
100
|
+
"application/vnd.oasis.opendocument.spreadsheet",
|
|
101
|
+
# HTML/EPUB
|
|
102
|
+
"text/html",
|
|
103
|
+
"application/xhtml+xml",
|
|
104
|
+
"application/epub+zip",
|
|
87
105
|
# Images
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
106
|
+
"image/png",
|
|
107
|
+
"image/jpeg",
|
|
108
|
+
"image/webp",
|
|
109
|
+
"image/gif",
|
|
110
|
+
"image/tiff",
|
|
111
|
+
"image/jpg",
|
|
94
112
|
}
|
|
@@ -16,7 +16,7 @@ def parse_args() -> argparse.Namespace:
|
|
|
16
16
|
__version__ = 'unknown' # Fallback if package not installed
|
|
17
17
|
|
|
18
18
|
parser = argparse.ArgumentParser(
|
|
19
|
-
description="Process
|
|
19
|
+
description="Process documents with the Marker API.",
|
|
20
20
|
formatter_class=argparse.ArgumentDefaultsHelpFormatter
|
|
21
21
|
)
|
|
22
22
|
|
|
@@ -28,8 +28,10 @@ def parse_args() -> argparse.Namespace:
|
|
|
28
28
|
)
|
|
29
29
|
|
|
30
30
|
parser.add_argument("input", help="Input file or directory path")
|
|
31
|
-
|
|
32
|
-
parser.
|
|
31
|
+
|
|
32
|
+
output_group = parser.add_mutually_exclusive_group()
|
|
33
|
+
output_group.add_argument("--json", action="store_true", help="Output in JSON format")
|
|
34
|
+
output_group.add_argument("--html", action="store_true", help="Output in HTML format")
|
|
33
35
|
|
|
34
36
|
parser.add_argument("-l", "--langs", default="English", help="Comma-separated OCR languages")
|
|
35
37
|
parser.add_argument("--llm", action="store_true", help="Use LLM for enhanced processing")
|
|
@@ -54,17 +56,23 @@ def create_config_from_args() -> Config:
|
|
|
54
56
|
|
|
55
57
|
try:
|
|
56
58
|
api_key = get_env_var("MARKER_PDF_KEY")
|
|
57
|
-
except
|
|
59
|
+
except FileError as e:
|
|
58
60
|
raise ConfigurationError(f"API key not found: {e}. Set the MARKER_PDF_KEY environment variable.")
|
|
59
61
|
|
|
60
62
|
# If --no-chunk is specified, override chunk size to effectively disable chunking
|
|
61
63
|
chunk_size = 1_000_000 if args.no_chunk else args.chunk_size
|
|
62
64
|
|
|
65
|
+
output_format = "markdown"
|
|
66
|
+
if args.json:
|
|
67
|
+
output_format = "json"
|
|
68
|
+
elif args.html:
|
|
69
|
+
output_format = "html"
|
|
70
|
+
|
|
63
71
|
config = Config(
|
|
64
72
|
api_key=api_key,
|
|
65
73
|
input_path=args.input,
|
|
66
74
|
output_dir=Path(args.output_dir) if args.output_dir else None,
|
|
67
|
-
output_format=
|
|
75
|
+
output_format=output_format,
|
|
68
76
|
langs=args.langs,
|
|
69
77
|
use_llm=args.llm or args.max,
|
|
70
78
|
strip_existing_ocr=args.strip or args.max,
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
from dataclasses import dataclass
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
from typing import Optional
|
|
4
|
+
import logging
|
|
5
|
+
import tempfile
|
|
6
|
+
|
|
7
|
+
from docs_to_md.api.models import SUPPORTED_FORMAT_EXTENSIONS
|
|
8
|
+
from docs_to_md.utils.exceptions import ConfigurationError
|
|
9
|
+
|
|
10
|
+
logger = logging.getLogger(__name__)
|
|
11
|
+
SETTINGS_DIR_NAME = ".docs_to_md"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class Config:
|
|
16
|
+
"""Global configuration for marker PDF conversion."""
|
|
17
|
+
api_key: str
|
|
18
|
+
|
|
19
|
+
input_path: str
|
|
20
|
+
output_dir: Optional[Path] = None
|
|
21
|
+
cache_dir: Path = Path.home() / SETTINGS_DIR_NAME / "cache" # Root directory for cache files
|
|
22
|
+
root_tmp_dir: Path = Path.home() / SETTINGS_DIR_NAME / "tmp" # Root directory for temporary files
|
|
23
|
+
|
|
24
|
+
output_format: str = "markdown"
|
|
25
|
+
langs: str = "English"
|
|
26
|
+
chunk_size: int = 25
|
|
27
|
+
|
|
28
|
+
use_llm: bool = False
|
|
29
|
+
strip_existing_ocr: bool = False
|
|
30
|
+
disable_image_extraction: bool = False
|
|
31
|
+
force_ocr: bool = False
|
|
32
|
+
paginate: bool = False
|
|
33
|
+
max_pages: Optional[int] = None
|
|
34
|
+
|
|
35
|
+
@staticmethod
|
|
36
|
+
def _resolve_writable_directory(path: Path, label: str) -> Path:
|
|
37
|
+
resolved = Path(path).expanduser().resolve(strict=False)
|
|
38
|
+
|
|
39
|
+
try:
|
|
40
|
+
resolved.mkdir(parents=True, exist_ok=True)
|
|
41
|
+
probe_path = resolved / ".write_probe"
|
|
42
|
+
probe_path.write_text("")
|
|
43
|
+
probe_path.unlink(missing_ok=True)
|
|
44
|
+
return resolved
|
|
45
|
+
except OSError as error:
|
|
46
|
+
fallback = (
|
|
47
|
+
Path(tempfile.gettempdir()) / SETTINGS_DIR_NAME / label
|
|
48
|
+
).resolve(strict=False)
|
|
49
|
+
try:
|
|
50
|
+
fallback.mkdir(parents=True, exist_ok=True)
|
|
51
|
+
probe_path = fallback / ".write_probe"
|
|
52
|
+
probe_path.write_text("")
|
|
53
|
+
probe_path.unlink(missing_ok=True)
|
|
54
|
+
except OSError as fallback_error:
|
|
55
|
+
raise ConfigurationError(
|
|
56
|
+
f"{label} directory is not writable: '{resolved}' "
|
|
57
|
+
f"(fallback '{fallback}' also failed: {fallback_error})"
|
|
58
|
+
) from fallback_error
|
|
59
|
+
|
|
60
|
+
logger.warning(
|
|
61
|
+
"Directory '%s' is not writable (%s); falling back to '%s'.",
|
|
62
|
+
resolved,
|
|
63
|
+
error,
|
|
64
|
+
fallback,
|
|
65
|
+
)
|
|
66
|
+
return fallback
|
|
67
|
+
|
|
68
|
+
def validate(self) -> None:
|
|
69
|
+
self.output_format = (self.output_format or "").strip().lower()
|
|
70
|
+
if not self.api_key:
|
|
71
|
+
raise ConfigurationError("API key is required")
|
|
72
|
+
|
|
73
|
+
if not self.input_path:
|
|
74
|
+
raise ConfigurationError("Input path is required")
|
|
75
|
+
|
|
76
|
+
normalized_input = Path(self.input_path).expanduser()
|
|
77
|
+
if not normalized_input.exists():
|
|
78
|
+
raise ConfigurationError(f"Input path does not exist: {self.input_path}")
|
|
79
|
+
self.input_path = str(normalized_input.resolve())
|
|
80
|
+
|
|
81
|
+
if self.chunk_size < 1:
|
|
82
|
+
raise ConfigurationError("Chunk size must be at least 1")
|
|
83
|
+
|
|
84
|
+
if self.max_pages is not None and self.max_pages < 1:
|
|
85
|
+
raise ConfigurationError("Max pages must be at least 1")
|
|
86
|
+
|
|
87
|
+
if not self.output_format or self.output_format not in SUPPORTED_FORMAT_EXTENSIONS:
|
|
88
|
+
raise ConfigurationError(f"Unsupported output format: {self.output_format}")
|
|
89
|
+
|
|
90
|
+
if self.output_dir is not None:
|
|
91
|
+
output_dir = Path(self.output_dir).expanduser()
|
|
92
|
+
if not output_dir.is_absolute():
|
|
93
|
+
raise ConfigurationError(
|
|
94
|
+
f"Output directory must be an absolute path: {self.output_dir}"
|
|
95
|
+
)
|
|
96
|
+
self.output_dir = output_dir.resolve(strict=False)
|
|
97
|
+
|
|
98
|
+
self.cache_dir = self._resolve_writable_directory(self.cache_dir, "cache")
|
|
99
|
+
self.root_tmp_dir = self._resolve_writable_directory(self.root_tmp_dir, "tmp")
|