pdf-to-markdown-cli 0.5.1__tar.gz → 0.5.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. pdf_to_markdown_cli-0.5.2/CHANGELOG.md +25 -0
  2. pdf_to_markdown_cli-0.5.2/CONTRIBUTING.md +38 -0
  3. pdf_to_markdown_cli-0.5.2/MANIFEST.in +13 -0
  4. pdf_to_markdown_cli-0.5.2/PKG-INFO +143 -0
  5. pdf_to_markdown_cli-0.5.2/README.md +95 -0
  6. pdf_to_markdown_cli-0.5.2/pyproject.toml +74 -0
  7. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/api/client.py +32 -4
  8. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/api/models.py +32 -14
  9. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/config/cli.py +13 -5
  10. pdf_to_markdown_cli-0.5.2/src/docs_to_md/config/settings.py +99 -0
  11. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/core/paths.py +17 -21
  12. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/core/processor.py +2 -3
  13. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/core/result_handler.py +188 -159
  14. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/storage/cache.py +1 -3
  15. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/storage/models.py +36 -21
  16. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/logging.py +8 -3
  17. pdf_to_markdown_cli-0.5.2/src/pdf_to_markdown_cli.egg-info/PKG-INFO +143 -0
  18. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/SOURCES.txt +2 -0
  19. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/requires.txt +8 -0
  20. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_cli.py +17 -2
  21. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_cli_equations.py +33 -19
  22. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_settings.py +26 -0
  23. pdf_to_markdown_cli-0.5.1/MANIFEST.in +0 -24
  24. pdf_to_markdown_cli-0.5.1/PKG-INFO +0 -119
  25. pdf_to_markdown_cli-0.5.1/README.md +0 -64
  26. pdf_to_markdown_cli-0.5.1/pyproject.toml +0 -54
  27. pdf_to_markdown_cli-0.5.1/src/docs_to_md/config/settings.py +0 -54
  28. pdf_to_markdown_cli-0.5.1/src/pdf_to_markdown_cli.egg-info/PKG-INFO +0 -119
  29. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/LICENSE +0 -0
  30. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/setup.cfg +0 -0
  31. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/__init__.py +0 -0
  32. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/__main__.py +0 -0
  33. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/api/__init__.py +0 -0
  34. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/config/__init__.py +0 -0
  35. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/core/__init__.py +0 -0
  36. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/main.py +0 -0
  37. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/storage/__init__.py +0 -0
  38. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/__init__.py +0 -0
  39. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/exceptions.py +0 -0
  40. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/file_utils.py +0 -0
  41. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/docs_to_md/utils/pdf_splitter.py +0 -0
  42. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/dependency_links.txt +0 -0
  43. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/entry_points.txt +0 -0
  44. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/src/pdf_to_markdown_cli.egg-info/top_level.txt +0 -0
  45. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_paths.py +0 -0
  46. {pdf_to_markdown_cli-0.5.1 → pdf_to_markdown_cli-0.5.2}/tests/test_utils.py +0 -0
@@ -0,0 +1,25 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented in this file.
4
+
5
+ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
6
+ and this project follows [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
+
8
+ ## [Unreleased]
9
+
10
+ ### Added
11
+
12
+ - Added fallback handling for non-writable cache/tmp directories.
13
+ - Added support tests for HTML output CLI configuration and both example PDFs.
14
+
15
+ ### Changed
16
+
17
+ - Expanded supported input MIME/extension mappings for document and image formats.
18
+ - Improved result polling with per-chunk exponential backoff and cleaner logging.
19
+ - Modernized packaging and project metadata in `pyproject.toml`.
20
+
21
+ ### Fixed
22
+
23
+ - Fixed false "unsupported file type" failures by adding MIME detection fallbacks.
24
+ - Fixed cache retrieval edge case where empty payloads were treated as missing.
25
+ - Fixed request state tracking semantics for completion/failure handling.
@@ -0,0 +1,38 @@
1
+ # Contributing
2
+
3
+ Thanks for contributing to `pdf-to-markdown-cli`.
4
+
5
+ ## Development setup
6
+
7
+ 1. Fork and clone the repository.
8
+ 2. Create a virtual environment.
9
+ 3. Install in editable mode:
10
+
11
+ ```bash
12
+ pip install -e .
13
+ ```
14
+
15
+ 4. Run tests before submitting changes:
16
+
17
+ ```bash
18
+ python -m unittest discover -s tests -v
19
+ ```
20
+
21
+ ## Pull request checklist
22
+
23
+ - Keep changes focused and explain the motivation.
24
+ - Add or update tests when behavior changes.
25
+ - Keep CLI behavior backward-compatible unless the change is intentional.
26
+ - Update docs (`README.md`, examples, or this guide) when user-facing behavior changes.
27
+ - Ensure all tests pass locally.
28
+
29
+ ## Coding guidelines
30
+
31
+ - Target Python 3.10+.
32
+ - Prefer clear, small functions with explicit error handling.
33
+ - Avoid silent failures; log contextual errors where useful.
34
+ - Do not commit API keys, credentials, or private documents.
35
+
36
+ ## Reporting issues
37
+
38
+ Use GitHub issues for bug reports and feature requests.
@@ -0,0 +1,13 @@
1
+ # Include the README and LICENSE
2
+ include README.md
3
+ include LICENSE
4
+ include CHANGELOG.md
5
+ include CONTRIBUTING.md
6
+
7
+ # Include package source
8
+ recursive-include src/docs_to_md *.py
9
+
10
+ # Exclude generated files
11
+ global-exclude *.py[cod]
12
+ global-exclude __pycache__
13
+ global-exclude .DS_Store
@@ -0,0 +1,143 @@
1
+ Metadata-Version: 2.4
2
+ Name: pdf-to-markdown-cli
3
+ Version: 0.5.2
4
+ Summary: CLI utility to convert PDFs and supported document formats to Markdown/JSON/HTML with Marker API.
5
+ Author-email: Nikita Sokolsky <sokolx@gmail.com>
6
+ Maintainer-email: Nikita Sokolsky <sokolx@gmail.com>
7
+ License-Expression: MIT
8
+ Project-URL: Homepage, https://github.com/SokolskyNikita/pdf-to-markdown-cli
9
+ Project-URL: Repository, https://github.com/SokolskyNikita/pdf-to-markdown-cli
10
+ Project-URL: Documentation, https://github.com/SokolskyNikita/pdf-to-markdown-cli#readme
11
+ Project-URL: Issues, https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues
12
+ Project-URL: Changelog, https://github.com/SokolskyNikita/pdf-to-markdown-cli/releases
13
+ Keywords: pdf,markdown,converter,cli,document,marker,md
14
+ Classifier: Development Status :: 4 - Beta
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Intended Audience :: End Users/Desktop
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Operating System :: OS Independent
23
+ Classifier: Environment :: Console
24
+ Classifier: Topic :: Office/Business
25
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
26
+ Classifier: Topic :: Text Processing
27
+ Classifier: Topic :: Utilities
28
+ Requires-Python: >=3.10
29
+ Description-Content-Type: text/markdown
30
+ License-File: LICENSE
31
+ Requires-Dist: backoff>=2.0
32
+ Requires-Dist: diskcache>=5.0
33
+ Requires-Dist: filetype>=1.0
34
+ Requires-Dist: pikepdf>=8.0
35
+ Requires-Dist: pydantic>=2.0
36
+ Requires-Dist: ratelimit>=2.0
37
+ Requires-Dist: requests>=2.0
38
+ Requires-Dist: tqdm>=4.0
39
+ Provides-Extra: test
40
+ Requires-Dist: pytest>=8.3.0; extra == "test"
41
+ Requires-Dist: pytest-cov>=5.0.0; extra == "test"
42
+ Provides-Extra: dev
43
+ Requires-Dist: build>=1.2.0; extra == "dev"
44
+ Requires-Dist: twine>=5.1.0; extra == "dev"
45
+ Requires-Dist: ruff>=0.11.0; extra == "dev"
46
+ Requires-Dist: types-requests>=2.32.0; extra == "dev"
47
+ Dynamic: license-file
48
+
49
+ # PDF to markdown CLI
50
+
51
+ [![PyPI](https://img.shields.io/pypi/v/pdf-to-markdown-cli.svg)](https://pypi.org/project/pdf-to-markdown-cli/)
52
+ [![Python versions](https://img.shields.io/pypi/pyversions/pdf-to-markdown-cli.svg)](https://pypi.org/project/pdf-to-markdown-cli/)
53
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](LICENSE)
54
+
55
+ Command-line utility for converting PDFs and other supported documents into Markdown, JSON, or HTML using the [Marker API](https://www.datalab.to/marker).
56
+
57
+ ## Why use this tool
58
+
59
+ - Converts single files or entire directories
60
+ - Automatically splits large PDFs into chunks and merges results
61
+ - Persists request state locally so interrupted runs can recover
62
+ - Rewrites and copies extracted images into deterministic output folders
63
+ - Supports OCR/LLM tuning flags from the Marker API
64
+
65
+ ## Supported formats
66
+
67
+ ### Input
68
+
69
+ - PDF (`.pdf`)
70
+ - Word (`.doc`, `.docx`, `.odt`)
71
+ - PowerPoint (`.ppt`, `.pptx`, `.odp`)
72
+ - Spreadsheets (`.xls`, `.xlsx`, `.ods`)
73
+ - EPUB/HTML (`.epub`, `.html`)
74
+ - Images (`.png`, `.jpg`, `.jpeg`, `.webp`, `.gif`, `.tiff`)
75
+
76
+ ### Output
77
+
78
+ - Markdown (`.md`, default)
79
+ - JSON (`.json`)
80
+ - HTML (`.html`)
81
+
82
+ ## Installation
83
+
84
+ ```bash
85
+ pip install pdf-to-markdown-cli
86
+ ```
87
+
88
+ From source:
89
+
90
+ ```bash
91
+ git clone https://github.com/SokolskyNikita/pdf-to-markdown-cli.git
92
+ cd pdf-to-markdown-cli
93
+ pip install -e .
94
+ ```
95
+
96
+ ## Quick start
97
+
98
+ ```bash
99
+ export MARKER_PDF_KEY="your_api_key"
100
+ pdf-to-md ./examples/equations.pdf
101
+ ```
102
+
103
+ Process a directory:
104
+
105
+ ```bash
106
+ pdf-to-md ./docs
107
+ ```
108
+
109
+ Use JSON or HTML output:
110
+
111
+ ```bash
112
+ pdf-to-md ./examples/equations.pdf --json
113
+ pdf-to-md ./examples/equations.pdf --html
114
+ ```
115
+
116
+ ## CLI options
117
+
118
+ - `input`: input file or directory path
119
+ - `--json`: output JSON instead of Markdown
120
+ - `--html`: output HTML instead of Markdown
121
+ - `--langs`: comma-separated OCR languages (default: `English`)
122
+ - `--llm`: enable LLM-enhanced processing
123
+ - `--strip`: redo OCR
124
+ - `--noimg`: disable image extraction
125
+ - `--force`: force OCR on all pages
126
+ - `--pages`: include page delimiters
127
+ - `--max`: enable all OCR enhancement flags (`--llm --strip --force`)
128
+ - `-mp`, `--max-pages`: process only the first N pages
129
+ - `--no-chunk`: disable PDF chunking
130
+ - `-cs`, `--chunk-size`: PDF pages per chunk (default: `25`)
131
+ - `-o`, `--output-dir`: absolute output directory path
132
+ - `-v`, `--verbose`: debug logging
133
+ - `--version`: show installed package version
134
+
135
+ ## Development
136
+
137
+ Run tests:
138
+
139
+ ```bash
140
+ python -m unittest discover -s tests -v
141
+ ```
142
+
143
+ For contributions or questions, open a GitHub issue.
@@ -0,0 +1,95 @@
1
+ # PDF to markdown CLI
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/pdf-to-markdown-cli.svg)](https://pypi.org/project/pdf-to-markdown-cli/)
4
+ [![Python versions](https://img.shields.io/pypi/pyversions/pdf-to-markdown-cli.svg)](https://pypi.org/project/pdf-to-markdown-cli/)
5
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green.svg)](LICENSE)
6
+
7
+ Command-line utility for converting PDFs and other supported documents into Markdown, JSON, or HTML using the [Marker API](https://www.datalab.to/marker).
8
+
9
+ ## Why use this tool
10
+
11
+ - Converts single files or entire directories
12
+ - Automatically splits large PDFs into chunks and merges results
13
+ - Persists request state locally so interrupted runs can recover
14
+ - Rewrites and copies extracted images into deterministic output folders
15
+ - Supports OCR/LLM tuning flags from the Marker API
16
+
17
+ ## Supported formats
18
+
19
+ ### Input
20
+
21
+ - PDF (`.pdf`)
22
+ - Word (`.doc`, `.docx`, `.odt`)
23
+ - PowerPoint (`.ppt`, `.pptx`, `.odp`)
24
+ - Spreadsheets (`.xls`, `.xlsx`, `.ods`)
25
+ - EPUB/HTML (`.epub`, `.html`)
26
+ - Images (`.png`, `.jpg`, `.jpeg`, `.webp`, `.gif`, `.tiff`)
27
+
28
+ ### Output
29
+
30
+ - Markdown (`.md`, default)
31
+ - JSON (`.json`)
32
+ - HTML (`.html`)
33
+
34
+ ## Installation
35
+
36
+ ```bash
37
+ pip install pdf-to-markdown-cli
38
+ ```
39
+
40
+ From source:
41
+
42
+ ```bash
43
+ git clone https://github.com/SokolskyNikita/pdf-to-markdown-cli.git
44
+ cd pdf-to-markdown-cli
45
+ pip install -e .
46
+ ```
47
+
48
+ ## Quick start
49
+
50
+ ```bash
51
+ export MARKER_PDF_KEY="your_api_key"
52
+ pdf-to-md ./examples/equations.pdf
53
+ ```
54
+
55
+ Process a directory:
56
+
57
+ ```bash
58
+ pdf-to-md ./docs
59
+ ```
60
+
61
+ Use JSON or HTML output:
62
+
63
+ ```bash
64
+ pdf-to-md ./examples/equations.pdf --json
65
+ pdf-to-md ./examples/equations.pdf --html
66
+ ```
67
+
68
+ ## CLI options
69
+
70
+ - `input`: input file or directory path
71
+ - `--json`: output JSON instead of Markdown
72
+ - `--html`: output HTML instead of Markdown
73
+ - `--langs`: comma-separated OCR languages (default: `English`)
74
+ - `--llm`: enable LLM-enhanced processing
75
+ - `--strip`: redo OCR
76
+ - `--noimg`: disable image extraction
77
+ - `--force`: force OCR on all pages
78
+ - `--pages`: include page delimiters
79
+ - `--max`: enable all OCR enhancement flags (`--llm --strip --force`)
80
+ - `-mp`, `--max-pages`: process only the first N pages
81
+ - `--no-chunk`: disable PDF chunking
82
+ - `-cs`, `--chunk-size`: PDF pages per chunk (default: `25`)
83
+ - `-o`, `--output-dir`: absolute output directory path
84
+ - `-v`, `--verbose`: debug logging
85
+ - `--version`: show installed package version
86
+
87
+ ## Development
88
+
89
+ Run tests:
90
+
91
+ ```bash
92
+ python -m unittest discover -s tests -v
93
+ ```
94
+
95
+ For contributions or questions, open a GitHub issue.
@@ -0,0 +1,74 @@
1
+ [build-system]
2
+ requires = ["setuptools>=69.0", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "pdf-to-markdown-cli"
7
+ version = "0.5.2"
8
+ description = "CLI utility to convert PDFs and supported document formats to Markdown/JSON/HTML with Marker API."
9
+ readme = { file = "README.md", content-type = "text/markdown" }
10
+ authors = [
11
+ { name = "Nikita Sokolsky", email = "sokolx@gmail.com" },
12
+ ]
13
+ maintainers = [
14
+ { name = "Nikita Sokolsky", email = "sokolx@gmail.com" },
15
+ ]
16
+ license = "MIT"
17
+ license-files = ["LICENSE"]
18
+ requires-python = ">=3.10"
19
+ classifiers = [
20
+ "Development Status :: 4 - Beta",
21
+ "Intended Audience :: Developers",
22
+ "Intended Audience :: End Users/Desktop",
23
+ "Programming Language :: Python :: 3",
24
+ "Programming Language :: Python :: 3.10",
25
+ "Programming Language :: Python :: 3.11",
26
+ "Programming Language :: Python :: 3.12",
27
+ "Programming Language :: Python :: 3.13",
28
+ "Operating System :: OS Independent",
29
+ "Environment :: Console",
30
+ "Topic :: Office/Business",
31
+ "Topic :: Software Development :: Libraries :: Python Modules",
32
+ "Topic :: Text Processing",
33
+ "Topic :: Utilities",
34
+ ]
35
+ keywords = ["pdf", "markdown", "converter", "cli", "document", "marker", "md"]
36
+ dependencies = [
37
+ "backoff>=2.0",
38
+ "diskcache>=5.0",
39
+ "filetype>=1.0",
40
+ "pikepdf>=8.0",
41
+ "pydantic>=2.0",
42
+ "ratelimit>=2.0",
43
+ "requests>=2.0",
44
+ "tqdm>=4.0",
45
+ ]
46
+
47
+ [project.urls]
48
+ "Homepage" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli"
49
+ "Repository" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli"
50
+ "Documentation" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli#readme"
51
+ "Issues" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues"
52
+ "Changelog" = "https://github.com/SokolskyNikita/pdf-to-markdown-cli/releases"
53
+
54
+ [project.scripts]
55
+ pdf-to-md = "docs_to_md.main:main"
56
+
57
+ [project.optional-dependencies]
58
+ test = [
59
+ "pytest>=8.3.0",
60
+ "pytest-cov>=5.0.0",
61
+ ]
62
+ dev = [
63
+ "build>=1.2.0",
64
+ "twine>=5.1.0",
65
+ "ruff>=0.11.0",
66
+ "types-requests>=2.32.0",
67
+ ]
68
+
69
+ [tool.setuptools]
70
+ package-dir = { "" = "src" }
71
+ include-package-data = true
72
+
73
+ [tool.setuptools.packages.find]
74
+ where = ["src"]
@@ -1,5 +1,6 @@
1
1
  import json
2
2
  import logging
3
+ import mimetypes
3
4
  from pathlib import Path
4
5
  from typing import Optional
5
6
 
@@ -27,6 +28,19 @@ logger = logging.getLogger(__name__)
27
28
 
28
29
  class MarkerClient:
29
30
  BASE_MARKER_API_ENDPOINT = "https://www.datalab.to/api/v1/marker"
31
+ _EXTENSION_TO_MIME = {
32
+ ".doc": "application/msword",
33
+ ".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
34
+ ".odt": "application/vnd.oasis.opendocument.text",
35
+ ".ppt": "application/vnd.ms-powerpoint",
36
+ ".pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
37
+ ".odp": "application/vnd.oasis.opendocument.presentation",
38
+ ".xls": "application/vnd.ms-excel",
39
+ ".xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
40
+ ".ods": "application/vnd.oasis.opendocument.spreadsheet",
41
+ ".epub": "application/epub+zip",
42
+ ".html": "text/html",
43
+ }
30
44
 
31
45
  # See datalab_marker_api_docs.md#authentication for API key details
32
46
  def __init__(self, api_key: str):
@@ -35,6 +49,19 @@ class MarkerClient:
35
49
 
36
50
  self.headers = {"X-Api-Key": api_key.strip()}
37
51
 
52
+ def _detect_mime_type(self, file_path: Path, file_data: bytes) -> Optional[str]:
53
+ """Infer MIME type from file header first, then file extension."""
54
+ detected = filetype.guess(file_data)
55
+ if detected and detected.mime:
56
+ return detected.mime
57
+
58
+ suffix = file_path.suffix.lower()
59
+ if suffix in self._EXTENSION_TO_MIME:
60
+ return self._EXTENSION_TO_MIME[suffix]
61
+
62
+ guessed, _ = mimetypes.guess_type(file_path.name)
63
+ return guessed
64
+
38
65
  @sleep_and_retry
39
66
  @limits(calls=MAX_REQUESTS_PER_MINUTE, period=60)
40
67
  @backoff.on_exception(
@@ -62,16 +89,17 @@ class MarkerClient:
62
89
  raise APIError(f"File not found: {file_path}")
63
90
 
64
91
  file_data = FileIO.read_file(file_path)
65
- kind = filetype.guess(file_data)
92
+ mime_type = self._detect_mime_type(file_path, file_data)
66
93
 
67
94
  # Supported types listed in datalab_marker_api_docs.md#supported-file-types
68
- if not kind or kind.mime not in SUPPORTED_MIME_TYPES:
95
+ if not mime_type or mime_type not in SUPPORTED_MIME_TYPES:
69
96
  raise APIError(
70
- f"Unsupported file type: {kind.mime if kind else 'unknown'}"
97
+ f"Unsupported file type for '{file_path.name}': "
98
+ f"{mime_type or 'unknown'}"
71
99
  )
72
100
 
73
101
  form_data = {
74
- "file": (file_path.name, file_data, kind.mime),
102
+ "file": (file_path.name, file_data, mime_type),
75
103
  "langs": (None, langs),
76
104
  "output_format": (None, output_format),
77
105
  }
@@ -50,45 +50,63 @@ SUPPORTED_FORMAT_EXTENSIONS = {
50
50
  "markdown": ".md",
51
51
  "json": ".json",
52
52
  "html": ".html",
53
- "txt": ".txt"
54
53
  }
55
54
 
56
55
  SUPPORTED_IMAGE_EXTENSIONS: Set[str] = {
57
56
  "jpg",
58
57
  "jpeg",
59
58
  "png",
59
+ "webp",
60
60
  "gif",
61
- "tiff"
61
+ "tiff",
62
62
  }
63
63
 
64
64
  SUPPORTED_INPUT_EXTENSIONS: Set[str] = {
65
65
  "pdf",
66
66
  "docx",
67
67
  "doc",
68
+ "odt",
68
69
  "pptx",
69
70
  "ppt",
71
+ "odp",
72
+ "xlsx",
73
+ "xls",
74
+ "ods",
75
+ "epub",
76
+ "html",
70
77
  "jpg",
71
78
  "jpeg",
72
79
  "png",
80
+ "webp",
73
81
  "gif",
74
- "tiff"
82
+ "tiff",
75
83
  }
76
84
 
77
85
  # Supported mime types according to datalab_marker_api_docs.md#supported-file-types
78
86
  SUPPORTED_MIME_TYPES: Set[str] = {
79
87
  # PDF
80
- 'application/pdf',
88
+ "application/pdf",
81
89
  # Word documents
82
- 'application/msword',
83
- 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
90
+ "application/msword",
91
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
92
+ "application/vnd.oasis.opendocument.text",
84
93
  # Powerpoint
85
- 'application/vnd.ms-powerpoint',
86
- 'application/vnd.openxmlformats-officedocument.presentationml.presentation',
94
+ "application/vnd.ms-powerpoint",
95
+ "application/vnd.openxmlformats-officedocument.presentationml.presentation",
96
+ "application/vnd.oasis.opendocument.presentation",
97
+ # Spreadsheets
98
+ "application/vnd.ms-excel",
99
+ "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
100
+ "application/vnd.oasis.opendocument.spreadsheet",
101
+ # HTML/EPUB
102
+ "text/html",
103
+ "application/xhtml+xml",
104
+ "application/epub+zip",
87
105
  # Images
88
- 'image/png',
89
- 'image/jpeg',
90
- 'image/webp',
91
- 'image/gif',
92
- 'image/tiff',
93
- 'image/jpg'
106
+ "image/png",
107
+ "image/jpeg",
108
+ "image/webp",
109
+ "image/gif",
110
+ "image/tiff",
111
+ "image/jpg",
94
112
  }
@@ -16,7 +16,7 @@ def parse_args() -> argparse.Namespace:
16
16
  __version__ = 'unknown' # Fallback if package not installed
17
17
 
18
18
  parser = argparse.ArgumentParser(
19
- description="Process PDF files using Marker API.",
19
+ description="Process documents with the Marker API.",
20
20
  formatter_class=argparse.ArgumentDefaultsHelpFormatter
21
21
  )
22
22
 
@@ -28,8 +28,10 @@ def parse_args() -> argparse.Namespace:
28
28
  )
29
29
 
30
30
  parser.add_argument("input", help="Input file or directory path")
31
-
32
- parser.add_argument("--json", action="store_true", help="Output in JSON format")
31
+
32
+ output_group = parser.add_mutually_exclusive_group()
33
+ output_group.add_argument("--json", action="store_true", help="Output in JSON format")
34
+ output_group.add_argument("--html", action="store_true", help="Output in HTML format")
33
35
 
34
36
  parser.add_argument("-l", "--langs", default="English", help="Comma-separated OCR languages")
35
37
  parser.add_argument("--llm", action="store_true", help="Use LLM for enhanced processing")
@@ -54,17 +56,23 @@ def create_config_from_args() -> Config:
54
56
 
55
57
  try:
56
58
  api_key = get_env_var("MARKER_PDF_KEY")
57
- except Exception as e:
59
+ except FileError as e:
58
60
  raise ConfigurationError(f"API key not found: {e}. Set the MARKER_PDF_KEY environment variable.")
59
61
 
60
62
  # If --no-chunk is specified, override chunk size to effectively disable chunking
61
63
  chunk_size = 1_000_000 if args.no_chunk else args.chunk_size
62
64
 
65
+ output_format = "markdown"
66
+ if args.json:
67
+ output_format = "json"
68
+ elif args.html:
69
+ output_format = "html"
70
+
63
71
  config = Config(
64
72
  api_key=api_key,
65
73
  input_path=args.input,
66
74
  output_dir=Path(args.output_dir) if args.output_dir else None,
67
- output_format="json" if args.json else "markdown",
75
+ output_format=output_format,
68
76
  langs=args.langs,
69
77
  use_llm=args.llm or args.max,
70
78
  strip_existing_ocr=args.strip or args.max,
@@ -0,0 +1,99 @@
1
+ from dataclasses import dataclass
2
+ from pathlib import Path
3
+ from typing import Optional
4
+ import logging
5
+ import tempfile
6
+
7
+ from docs_to_md.api.models import SUPPORTED_FORMAT_EXTENSIONS
8
+ from docs_to_md.utils.exceptions import ConfigurationError
9
+
10
+ logger = logging.getLogger(__name__)
11
+ SETTINGS_DIR_NAME = ".docs_to_md"
12
+
13
+
14
+ @dataclass
15
+ class Config:
16
+ """Global configuration for marker PDF conversion."""
17
+ api_key: str
18
+
19
+ input_path: str
20
+ output_dir: Optional[Path] = None
21
+ cache_dir: Path = Path.home() / SETTINGS_DIR_NAME / "cache" # Root directory for cache files
22
+ root_tmp_dir: Path = Path.home() / SETTINGS_DIR_NAME / "tmp" # Root directory for temporary files
23
+
24
+ output_format: str = "markdown"
25
+ langs: str = "English"
26
+ chunk_size: int = 25
27
+
28
+ use_llm: bool = False
29
+ strip_existing_ocr: bool = False
30
+ disable_image_extraction: bool = False
31
+ force_ocr: bool = False
32
+ paginate: bool = False
33
+ max_pages: Optional[int] = None
34
+
35
+ @staticmethod
36
+ def _resolve_writable_directory(path: Path, label: str) -> Path:
37
+ resolved = Path(path).expanduser().resolve(strict=False)
38
+
39
+ try:
40
+ resolved.mkdir(parents=True, exist_ok=True)
41
+ probe_path = resolved / ".write_probe"
42
+ probe_path.write_text("")
43
+ probe_path.unlink(missing_ok=True)
44
+ return resolved
45
+ except OSError as error:
46
+ fallback = (
47
+ Path(tempfile.gettempdir()) / SETTINGS_DIR_NAME / label
48
+ ).resolve(strict=False)
49
+ try:
50
+ fallback.mkdir(parents=True, exist_ok=True)
51
+ probe_path = fallback / ".write_probe"
52
+ probe_path.write_text("")
53
+ probe_path.unlink(missing_ok=True)
54
+ except OSError as fallback_error:
55
+ raise ConfigurationError(
56
+ f"{label} directory is not writable: '{resolved}' "
57
+ f"(fallback '{fallback}' also failed: {fallback_error})"
58
+ ) from fallback_error
59
+
60
+ logger.warning(
61
+ "Directory '%s' is not writable (%s); falling back to '%s'.",
62
+ resolved,
63
+ error,
64
+ fallback,
65
+ )
66
+ return fallback
67
+
68
+ def validate(self) -> None:
69
+ self.output_format = (self.output_format or "").strip().lower()
70
+ if not self.api_key:
71
+ raise ConfigurationError("API key is required")
72
+
73
+ if not self.input_path:
74
+ raise ConfigurationError("Input path is required")
75
+
76
+ normalized_input = Path(self.input_path).expanduser()
77
+ if not normalized_input.exists():
78
+ raise ConfigurationError(f"Input path does not exist: {self.input_path}")
79
+ self.input_path = str(normalized_input.resolve())
80
+
81
+ if self.chunk_size < 1:
82
+ raise ConfigurationError("Chunk size must be at least 1")
83
+
84
+ if self.max_pages is not None and self.max_pages < 1:
85
+ raise ConfigurationError("Max pages must be at least 1")
86
+
87
+ if not self.output_format or self.output_format not in SUPPORTED_FORMAT_EXTENSIONS:
88
+ raise ConfigurationError(f"Unsupported output format: {self.output_format}")
89
+
90
+ if self.output_dir is not None:
91
+ output_dir = Path(self.output_dir).expanduser()
92
+ if not output_dir.is_absolute():
93
+ raise ConfigurationError(
94
+ f"Output directory must be an absolute path: {self.output_dir}"
95
+ )
96
+ self.output_dir = output_dir.resolve(strict=False)
97
+
98
+ self.cache_dir = self._resolve_writable_directory(self.cache_dir, "cache")
99
+ self.root_tmp_dir = self._resolve_writable_directory(self.root_tmp_dir, "tmp")