pdf-to-markdown-cli 0.5.2__tar.gz → 1.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. pdf_to_markdown_cli-1.0.1/CHANGELOG.md +146 -0
  2. pdf_to_markdown_cli-1.0.1/CONTRIBUTING.md +83 -0
  3. {pdf_to_markdown_cli-0.5.2 → pdf_to_markdown_cli-1.0.1}/LICENSE +0 -0
  4. pdf_to_markdown_cli-1.0.1/MANIFEST.in +7 -0
  5. pdf_to_markdown_cli-1.0.1/PKG-INFO +305 -0
  6. pdf_to_markdown_cli-1.0.1/README.md +261 -0
  7. pdf_to_markdown_cli-1.0.1/examples/README.md +28 -0
  8. pdf_to_markdown_cli-1.0.1/examples/alice_in_wonderland_sample.md +47 -0
  9. pdf_to_markdown_cli-1.0.1/examples/alice_in_wonderland_sample.pdf +0 -0
  10. pdf_to_markdown_cli-1.0.1/examples/alice_in_wonderland_sample_images/61972112770d9c8fb00883057954f885_img.jpg +0 -0
  11. pdf_to_markdown_cli-1.0.1/examples/darwin_origin_of_species_en.md +13 -0
  12. pdf_to_markdown_cli-1.0.1/examples/darwin_origin_of_species_en.pdf +0 -0
  13. pdf_to_markdown_cli-1.0.1/examples/darwin_origin_of_species_en_images/68ac34ff111db52afaa786afcb8346c3_img.jpg +0 -0
  14. pdf_to_markdown_cli-1.0.1/examples/equations.md +29 -0
  15. pdf_to_markdown_cli-1.0.1/examples/equations.pdf +0 -0
  16. pdf_to_markdown_cli-1.0.1/examples/equations_images/30a26f2d17ca95672702bf50fb4f0242_img.jpg +0 -0
  17. pdf_to_markdown_cli-1.0.1/examples/grimm_fairy_tales_de.md +7 -0
  18. pdf_to_markdown_cli-1.0.1/examples/grimm_fairy_tales_de.pdf +0 -0
  19. pdf_to_markdown_cli-1.0.1/examples/shijing_gupu_zh.md +48 -0
  20. pdf_to_markdown_cli-1.0.1/examples/shijing_gupu_zh.pdf +0 -0
  21. pdf_to_markdown_cli-1.0.1/examples/shijing_gupu_zh_images/13f8cc7f38c868748825f3a80b201b57_img.jpg +0 -0
  22. pdf_to_markdown_cli-1.0.1/examples/shijing_gupu_zh_images/8c378a184b5ae4d1605cb74d7b7a7e3f_img.jpg +0 -0
  23. pdf_to_markdown_cli-1.0.1/examples/shijing_gupu_zh_images/d5fc881e4328d6a2e76c9576408ced49_img.jpg +0 -0
  24. pdf_to_markdown_cli-1.0.1/examples/shijing_gupu_zh_images/eb70ac3e793aea24b49d2cd9f7b0b269_img.jpg +0 -0
  25. pdf_to_markdown_cli-1.0.1/examples/tolstoy_war_and_peace_ru.md +45 -0
  26. pdf_to_markdown_cli-1.0.1/examples/tolstoy_war_and_peace_ru.pdf +0 -0
  27. {pdf_to_markdown_cli-0.5.2 → pdf_to_markdown_cli-1.0.1}/pyproject.toml +33 -14
  28. pdf_to_markdown_cli-1.0.1/src/docs_to_md/__init__.py +8 -0
  29. pdf_to_markdown_cli-1.0.1/src/docs_to_md/__main__.py +3 -0
  30. pdf_to_markdown_cli-1.0.1/src/docs_to_md/assemble.py +212 -0
  31. pdf_to_markdown_cli-1.0.1/src/docs_to_md/cli.py +297 -0
  32. pdf_to_markdown_cli-1.0.1/src/docs_to_md/client.py +198 -0
  33. pdf_to_markdown_cli-1.0.1/src/docs_to_md/config.py +81 -0
  34. pdf_to_markdown_cli-1.0.1/src/docs_to_md/console.py +107 -0
  35. pdf_to_markdown_cli-1.0.1/src/docs_to_md/discovery.py +173 -0
  36. pdf_to_markdown_cli-1.0.1/src/docs_to_md/errors.py +50 -0
  37. pdf_to_markdown_cli-1.0.1/src/docs_to_md/markdown.py +90 -0
  38. pdf_to_markdown_cli-1.0.1/src/docs_to_md/models.py +124 -0
  39. pdf_to_markdown_cli-1.0.1/src/docs_to_md/pdf.py +108 -0
  40. pdf_to_markdown_cli-1.0.1/src/docs_to_md/pipeline.py +295 -0
  41. pdf_to_markdown_cli-1.0.1/src/pdf_to_markdown_cli.egg-info/PKG-INFO +305 -0
  42. pdf_to_markdown_cli-1.0.1/src/pdf_to_markdown_cli.egg-info/SOURCES.txt +59 -0
  43. {pdf_to_markdown_cli-0.5.2 → pdf_to_markdown_cli-1.0.1}/src/pdf_to_markdown_cli.egg-info/dependency_links.txt +0 -0
  44. pdf_to_markdown_cli-1.0.1/src/pdf_to_markdown_cli.egg-info/entry_points.txt +2 -0
  45. pdf_to_markdown_cli-1.0.1/src/pdf_to_markdown_cli.egg-info/requires.txt +13 -0
  46. {pdf_to_markdown_cli-0.5.2 → pdf_to_markdown_cli-1.0.1}/src/pdf_to_markdown_cli.egg-info/top_level.txt +0 -0
  47. {pdf_to_markdown_cli-0.5.2/src/docs_to_md → pdf_to_markdown_cli-1.0.1/tests}/__init__.py +0 -0
  48. pdf_to_markdown_cli-1.0.1/tests/conftest.py +93 -0
  49. pdf_to_markdown_cli-1.0.1/tests/samples.py +72 -0
  50. pdf_to_markdown_cli-1.0.1/tests/test_assemble.py +194 -0
  51. pdf_to_markdown_cli-1.0.1/tests/test_cli.py +261 -0
  52. pdf_to_markdown_cli-1.0.1/tests/test_client.py +214 -0
  53. pdf_to_markdown_cli-1.0.1/tests/test_console.py +85 -0
  54. pdf_to_markdown_cli-1.0.1/tests/test_discovery.py +159 -0
  55. pdf_to_markdown_cli-1.0.1/tests/test_live.py +82 -0
  56. pdf_to_markdown_cli-1.0.1/tests/test_markdown.py +53 -0
  57. pdf_to_markdown_cli-1.0.1/tests/test_models.py +66 -0
  58. pdf_to_markdown_cli-1.0.1/tests/test_pdf.py +76 -0
  59. pdf_to_markdown_cli-1.0.1/tests/test_pipeline.py +297 -0
  60. pdf_to_markdown_cli-1.0.1/tests/test_samples.py +64 -0
  61. pdf_to_markdown_cli-0.5.2/CHANGELOG.md +0 -25
  62. pdf_to_markdown_cli-0.5.2/CONTRIBUTING.md +0 -38
  63. pdf_to_markdown_cli-0.5.2/MANIFEST.in +0 -13
  64. pdf_to_markdown_cli-0.5.2/PKG-INFO +0 -143
  65. pdf_to_markdown_cli-0.5.2/README.md +0 -95
  66. pdf_to_markdown_cli-0.5.2/src/docs_to_md/__main__.py +0 -5
  67. pdf_to_markdown_cli-0.5.2/src/docs_to_md/api/__init__.py +0 -0
  68. pdf_to_markdown_cli-0.5.2/src/docs_to_md/api/client.py +0 -221
  69. pdf_to_markdown_cli-0.5.2/src/docs_to_md/api/models.py +0 -112
  70. pdf_to_markdown_cli-0.5.2/src/docs_to_md/config/__init__.py +0 -0
  71. pdf_to_markdown_cli-0.5.2/src/docs_to_md/config/cli.py +0 -95
  72. pdf_to_markdown_cli-0.5.2/src/docs_to_md/config/settings.py +0 -99
  73. pdf_to_markdown_cli-0.5.2/src/docs_to_md/core/__init__.py +0 -0
  74. pdf_to_markdown_cli-0.5.2/src/docs_to_md/core/paths.py +0 -76
  75. pdf_to_markdown_cli-0.5.2/src/docs_to_md/core/processor.py +0 -346
  76. pdf_to_markdown_cli-0.5.2/src/docs_to_md/core/result_handler.py +0 -635
  77. pdf_to_markdown_cli-0.5.2/src/docs_to_md/main.py +0 -64
  78. pdf_to_markdown_cli-0.5.2/src/docs_to_md/storage/__init__.py +0 -0
  79. pdf_to_markdown_cli-0.5.2/src/docs_to_md/storage/cache.py +0 -123
  80. pdf_to_markdown_cli-0.5.2/src/docs_to_md/storage/models.py +0 -104
  81. pdf_to_markdown_cli-0.5.2/src/docs_to_md/utils/__init__.py +0 -0
  82. pdf_to_markdown_cli-0.5.2/src/docs_to_md/utils/exceptions.py +0 -33
  83. pdf_to_markdown_cli-0.5.2/src/docs_to_md/utils/file_utils.py +0 -220
  84. pdf_to_markdown_cli-0.5.2/src/docs_to_md/utils/logging.py +0 -151
  85. pdf_to_markdown_cli-0.5.2/src/docs_to_md/utils/pdf_splitter.py +0 -167
  86. pdf_to_markdown_cli-0.5.2/src/pdf_to_markdown_cli.egg-info/PKG-INFO +0 -143
  87. pdf_to_markdown_cli-0.5.2/src/pdf_to_markdown_cli.egg-info/SOURCES.txt +0 -38
  88. pdf_to_markdown_cli-0.5.2/src/pdf_to_markdown_cli.egg-info/entry_points.txt +0 -2
  89. pdf_to_markdown_cli-0.5.2/src/pdf_to_markdown_cli.egg-info/requires.txt +0 -18
  90. pdf_to_markdown_cli-0.5.2/tests/test_cli.py +0 -52
  91. pdf_to_markdown_cli-0.5.2/tests/test_cli_equations.py +0 -127
  92. pdf_to_markdown_cli-0.5.2/tests/test_paths.py +0 -25
  93. pdf_to_markdown_cli-0.5.2/tests/test_settings.py +0 -91
  94. pdf_to_markdown_cli-0.5.2/tests/test_utils.py +0 -118
  95. {pdf_to_markdown_cli-0.5.2 → pdf_to_markdown_cli-1.0.1}/setup.cfg +0 -0
@@ -0,0 +1,146 @@
1
+ # Changelog
2
+
3
+ <!-- markdownlint-disable MD024 -->
4
+
5
+ All notable changes to `pdf-to-markdown-cli` are documented here.
6
+
7
+ This project follows [Semantic Versioning](https://semver.org/spec/v2.0.0.html), and this file follows the spirit of [Keep a Changelog](https://keepachangelog.com/en/1.1.0/).
8
+
9
+ ## [Unreleased]
10
+
11
+ ## [1.0.1] - 2026-09-29
12
+
13
+ ### Added
14
+
15
+ - Multilingual sample documents in `examples/`: scanned excerpts in Russian (Tolstoy), German Fraktur (Grimm), classical Chinese with music notation (Shijing Gupu), and English (Darwin's 1859 diagram), each with its reference output.
16
+ - An opt-in live test suite (`pytest -m live`) that converts every sample with the real API and checks text, chunk merging, page numbering, images, and HTML/JSON output. A manual **Live API tests** workflow runs it in CI.
17
+ - `SECURITY.md`, issue and pull request templates, and `examples/README.md`.
18
+
19
+ ### Changed
20
+
21
+ - Rewrote `README.md` (now also the PyPI page) and `CONTRIBUTING.md`.
22
+ - Releases now publish to PyPI automatically through Trusted Publishing when a version tag is pushed.
23
+
24
+ ## [1.0.0] - 2026-09-29
25
+
26
+ First stable release: a rewrite on Datalab's current Convert API with a scriptable, re-runnable CLI. See "Upgrading from 0.x" in the README.
27
+
28
+ ### Added
29
+
30
+ - `--mode {fast,balanced,accurate}`, `--page-range`, `--no-image-captions`, `--skip-cache`, and `--api-option KEY=VALUE` for any other Convert API field.
31
+ - Multiple inputs per command, e.g. `pdf-to-md a.pdf b.docx docs/`.
32
+ - `--dry-run` previews files, pages, and chunks without calling the API or needing a key.
33
+ - `--overwrite`: existing outputs are now skipped by default, so re-runs only convert what's new.
34
+ - `-j/--concurrency`: chunks and files convert in parallel (default 5).
35
+ - `--timeout`, `-q/--quiet`, `--keep-line-breaks`, and `--api-key`.
36
+ - The `DATALAB_API_KEY` environment variable, matching Datalab's official SDK. `MARKER_PDF_KEY` still works.
37
+ - Input types: `.xlsm`, `.xltx`, `.csv`, `.htm`, `.tif`.
38
+ - A per-file result line and an end-of-run summary with page count and cost. Converted paths are printed to stdout.
39
+ - Markdown paragraphs that were hard-wrapped are joined (disable with `--keep-line-breaks`).
40
+ - Markdown no longer repeats each generated image description as a paragraph under the image. The description stays in the alt text.
41
+
42
+ ### Changed
43
+
44
+ - Uses `POST /api/v1/convert`; the `/api/v1/marker` endpoint is deprecated upstream.
45
+ - Deterministic output names: `report.md` + `report_images/` instead of UUID-keyed names. Same-name inputs become `a.pdf.md` / `a.docx.md`, and other clashes get a numeric suffix. Names are compared case-insensitively, and an output never overwrites an input file.
46
+ - `-o/--output-dir` accepts relative paths and mirrors the input folder tree.
47
+ - `--llm` now means `--mode balanced` and `--max` means `--mode accurate`. `--strip`, `--force`, and `--langs` are ignored with a warning because the API removed them. `--pages`, `--noimg`, `-cs`, and `-mp` are still accepted as aliases of the new flag names.
48
+ - Exit codes: `0` success, `1` some files failed, `2` usage/auth error, `130` interrupted.
49
+ - Quieter default output. Timestamps and debug logs only appear with `-v`.
50
+ - Dependencies trimmed to `requests`, `pikepdf`, and `tqdm`.
51
+ - Flat package layout and a pytest suite with 97% branch coverage. CI runs ruff, tests on Python 3.10-3.14, and a packaging check.
52
+
53
+ ### Fixed
54
+
55
+ - `--html` and `--json` produced no output, because the response fields were misnamed.
56
+ - Failures (including an invalid API key) exited `0` and printed "Conversion completed successfully".
57
+ - `--max-pages` was applied to every chunk instead of the whole document.
58
+ - Retries never happened because the client swallowed exceptions. 429/5xx/network errors now retry with backoff and honor `Retry-After`.
59
+ - Long conversions were marked failed after about 5 minutes of polling.
60
+ - A 30-second request timeout broke uploads of large files.
61
+ - Re-running on a directory re-converted previously extracted images, and `--html` could overwrite `.html` inputs.
62
+ - Page separators, JSON block ids, and image names restarted in every chunk. Image links broke on filenames with spaces.
63
+ - Ctrl-C printed a traceback and left temporary files behind.
64
+
65
+ ### Removed
66
+
67
+ - The on-disk request cache (`~/.docs_to_md/`), which was never reused between runs.
68
+ - The copied documentation for the deprecated Marker endpoint. The docs now link to the live Datalab API reference.
69
+
70
+ ## [0.5.2] - 2026-05-13
71
+
72
+ ### Added
73
+
74
+ - Fallback handling when the default cache or temp directories are not writable.
75
+ - Test coverage for HTML output CLI configuration.
76
+ - Mocked test coverage for processing both bundled example PDFs.
77
+ - CI and dependency automation configuration in `.github/workflows/ci.yml` and `.github/dependabot.yml`.
78
+
79
+ ### Changed
80
+
81
+ - Expanded supported input extension and MIME mappings for documents, spreadsheets, presentations, web/ebook files, and images.
82
+ - Improved result polling with per-chunk exponential backoff and cleaner progress handling.
83
+ - Modernized packaging and project metadata in `pyproject.toml`.
84
+ - Refreshed public and contributor documentation.
85
+
86
+ ### Fixed
87
+
88
+ - Avoided false "unsupported file type" failures by adding MIME detection fallbacks.
89
+ - Fixed a cache retrieval edge case where empty payloads were treated as missing.
90
+ - Improved request/chunk state tracking for completion and failure handling.
91
+
92
+ ## [0.5.1] - 2025-09-29
93
+
94
+ ### Fixed
95
+
96
+ - Prevented creation of output image directories when no images were returned.
97
+ - Corrected generated filename and image path handling for output assets.
98
+
99
+ ## [0.5.0] - 2025-05-19
100
+
101
+ ### Added
102
+
103
+ - Unit test coverage for CLI processing paths.
104
+
105
+ ### Changed
106
+
107
+ - Updated project docs for Datalab Marker API usage and contributor workflow.
108
+ - Refactored processor internals into clearer chunking and submission helper flows.
109
+ - Switched test execution to `unittest`.
110
+
111
+ ### Fixed
112
+
113
+ - Fixed mutable default list behavior in request/chunk models.
114
+
115
+ ## [0.4.0] - 2025-04-21
116
+
117
+ ### Changed
118
+
119
+ - Polished post-refactor processing behavior and output handling for the package layout.
120
+
121
+ ### Fixed
122
+
123
+ - Improved image path deduplication and related output consistency.
124
+
125
+ ## [0.3.0] - 2025-04-19 (unpublished)
126
+
127
+ ### Changed
128
+
129
+ - Refactored processing, caching, and result handling internals.
130
+
131
+ ## [0.2.1] - 2025-04-10
132
+
133
+ ### Fixed
134
+
135
+ - Fixed packaging and distribution issues from the first PyPI release.
136
+
137
+ ## [0.2.0] - 2025-04-10
138
+
139
+ ### Added
140
+
141
+ - First public PyPI release of `pdf-to-markdown-cli`.
142
+ - Package and distribution scaffolding for pip installation and the `pdf-to-md` CLI entrypoint.
143
+
144
+ ### Changed
145
+
146
+ - Reorganized the project into a publishable Python package.
@@ -0,0 +1,83 @@
1
+ # Contributing to pdf-to-md
2
+
3
+ Thanks for your interest in improving `pdf-to-md`! Bug reports, feature ideas, documentation fixes, and pull requests are all welcome.
4
+
5
+ ## Ground rules
6
+
7
+ - **Reliability over features.** A conversion tool is only useful if you can trust it with a thousand files. Changes should keep runs predictable, re-runnable, and honest about failures.
8
+ - **Keep the CLI scriptable.** Only converted file paths go to stdout, and exit codes carry meaning. Don't print anything else to stdout.
9
+ - **Stay backward-compatible** unless a breaking change is deliberate. Document breaking changes in `CHANGELOG.md` and in the README's upgrade section.
10
+ - **Never commit** API keys, private documents, or build artifacts.
11
+
12
+ ## Getting set up
13
+
14
+ ```bash
15
+ git clone https://github.com/SokolskyNikita/pdf-to-markdown-cli.git
16
+ cd pdf-to-markdown-cli
17
+ python -m venv .venv && source .venv/bin/activate
18
+ pip install -e ".[dev]"
19
+ ```
20
+
21
+ ## Checks
22
+
23
+ CI runs these on Linux, macOS, and Windows for every supported Python version. Run them before opening a pull request:
24
+
25
+ ```bash
26
+ pytest --cov # full offline suite; never calls the API
27
+ ruff check . # lint
28
+ ruff format --check . # formatting (`ruff format .` fixes it)
29
+ ```
30
+
31
+ ### Live API tests
32
+
33
+ `tests/test_live.py` converts every document in [`examples/`](examples/) with the real Datalab API. It checks text in five languages and scripts, chunk merging, page numbering, images, and the HTML and JSON outputs. These tests are skipped by default. Run them when you change anything that affects conversion output:
34
+
35
+ ```bash
36
+ DATALAB_API_KEY=... pytest -m live # ~16 pages, a few cents
37
+ ```
38
+
39
+ Maintainers can also run them from the **Live API tests** workflow in the Actions tab, which uses the repository's `DATALAB_API_KEY` secret.
40
+
41
+ ## Project layout
42
+
43
+ ```text
44
+ src/docs_to_md/
45
+ ├── cli.py arguments, entry point, exit codes, summary
46
+ ├── config.py validated runtime configuration
47
+ ├── discovery.py input discovery, deterministic output names
48
+ ├── pipeline.py split, submit concurrently, poll, report
49
+ ├── client.py API client: retries, timeouts, error types
50
+ ├── models.py API request/response types, formats
51
+ ├── pdf.py page ranges and PDF splitting (pikepdf)
52
+ ├── assemble.py merge chunks, renumber pages, write files
53
+ ├── markdown.py Markdown clean-up (reflow, captions)
54
+ ├── console.py terminal output, progress bar, logging
55
+ └── errors.py exception hierarchy
56
+ tests/ pytest suite; samples.py lists examples/
57
+ examples/ sample documents with reference outputs
58
+ ```
59
+
60
+ ## Conventions
61
+
62
+ - Python 3.10+, type-hinted, with `from __future__ import annotations`.
63
+ - Raise exceptions from `errors.py`. Use `FatalAPIError` only for problems that affect every request (bad key, no credits). It aborts the whole run.
64
+ - Keep network code in `client.py` and orchestration in `pipeline.py`. Pipeline tests use `FakeClient` from `tests/conftest.py`. HTTP behavior is tested with the fake session in `tests/test_client.py`.
65
+ - Add tests with every behavior change, in the module's test file. The suite is fast (under a second), so keep it that way.
66
+ - Update the docs in the same pull request: `README.md` for user-facing changes, `CHANGELOG.md` under **Unreleased**, and `AGENTS.md` when the module map or conventions change.
67
+
68
+ ## Reporting bugs and proposing features
69
+
70
+ Open an [issue](https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues/new/choose) using the templates. For a bug, the most useful things are the exact command, the input type, and the output of the same command with `-v`. Please don't attach confidential documents.
71
+
72
+ Report security issues privately as described in [SECURITY.md](SECURITY.md).
73
+
74
+ ## Releasing (maintainers)
75
+
76
+ 1. Move the **Unreleased** changelog notes under a `## [X.Y.Z] - YYYY-MM-DD` heading and bump `version` in `pyproject.toml`.
77
+ 2. Commit, push to `main`, and wait for CI to pass.
78
+ 3. Tag and push: `git tag -a vX.Y.Z -m "vX.Y.Z" && git push origin vX.Y.Z`.
79
+
80
+ The [release workflow](.github/workflows/release.yml) then:
81
+ - checks that the tag matches the package version, runs the tests, and builds the sdist and wheel;
82
+ - creates the GitHub release with both files attached and the changelog section as notes;
83
+ - publishes to PyPI via [Trusted Publishing](https://docs.pypi.org/trusted-publishers/), with no stored tokens.
@@ -0,0 +1,7 @@
1
+ include README.md LICENSE CHANGELOG.md CONTRIBUTING.md
2
+
3
+ # The test suite, with the sample PDFs it uses, so the sdist is testable.
4
+ recursive-include tests *.py
5
+ graft examples
6
+
7
+ global-exclude *.py[cod] __pycache__ .DS_Store
@@ -0,0 +1,305 @@
1
+ Metadata-Version: 2.4
2
+ Name: pdf-to-markdown-cli
3
+ Version: 1.0.1
4
+ Summary: Convert PDFs, Office documents, ebooks, and images to Markdown, HTML, or JSON with the Datalab API.
5
+ Author-email: Nikita Sokolsky <sokolx@gmail.com>
6
+ Maintainer-email: Nikita Sokolsky <sokolx@gmail.com>
7
+ License-Expression: MIT
8
+ Project-URL: Homepage, https://github.com/SokolskyNikita/pdf-to-markdown-cli
9
+ Project-URL: Repository, https://github.com/SokolskyNikita/pdf-to-markdown-cli
10
+ Project-URL: Documentation, https://github.com/SokolskyNikita/pdf-to-markdown-cli#readme
11
+ Project-URL: Issues, https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues
12
+ Project-URL: Changelog, https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/CHANGELOG.md
13
+ Keywords: pdf,markdown,converter,cli,document,ocr,datalab,marker
14
+ Classifier: Development Status :: 5 - Production/Stable
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Intended Audience :: End Users/Desktop
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Programming Language :: Python :: 3.14
23
+ Classifier: Operating System :: OS Independent
24
+ Classifier: Environment :: Console
25
+ Classifier: Topic :: Office/Business
26
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
27
+ Classifier: Topic :: Text Processing
28
+ Classifier: Topic :: Utilities
29
+ Requires-Python: >=3.10
30
+ Description-Content-Type: text/markdown
31
+ License-File: LICENSE
32
+ Requires-Dist: pikepdf>=8.0
33
+ Requires-Dist: requests>=2.28
34
+ Requires-Dist: tqdm>=4.60
35
+ Provides-Extra: test
36
+ Requires-Dist: pytest>=8.3.0; extra == "test"
37
+ Requires-Dist: pytest-cov>=5.0.0; extra == "test"
38
+ Provides-Extra: dev
39
+ Requires-Dist: pdf-to-markdown-cli[test]; extra == "dev"
40
+ Requires-Dist: build>=1.2.0; extra == "dev"
41
+ Requires-Dist: twine>=5.1.0; extra == "dev"
42
+ Requires-Dist: ruff>=0.11.0; extra == "dev"
43
+ Dynamic: license-file
44
+
45
+ <h1 align="center">pdf-to-md</h1>
46
+
47
+ <p align="center">
48
+ <strong>Turn PDFs, scans, and Office documents into clean Markdown with one command, even 1,000-page books.</strong>
49
+ </p>
50
+
51
+ <p align="center">
52
+ <a href="https://pypi.org/project/pdf-to-markdown-cli/"><img src="https://img.shields.io/pypi/v/pdf-to-markdown-cli.svg" alt="PyPI"></a>
53
+ <a href="https://pypi.org/project/pdf-to-markdown-cli/"><img src="https://img.shields.io/badge/python-3.10%2B-blue.svg" alt="Python 3.10+"></a>
54
+ <a href="https://github.com/SokolskyNikita/pdf-to-markdown-cli/actions/workflows/ci.yml"><img src="https://github.com/SokolskyNikita/pdf-to-markdown-cli/actions/workflows/ci.yml/badge.svg" alt="CI"></a>
55
+ <a href="https://github.com/SokolskyNikita/pdf-to-markdown-cli/releases"><img src="https://img.shields.io/github/v/release/SokolskyNikita/pdf-to-markdown-cli" alt="GitHub release"></a>
56
+ <a href="https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/LICENSE"><img src="https://img.shields.io/badge/license-MIT-green.svg" alt="License: MIT"></a>
57
+ </p>
58
+
59
+ `pdf-to-md` is a command-line tool that converts documents to Markdown, HTML, or JSON with the [Datalab](https://www.datalab.to/) Convert API, the hosted version of the open-source [Marker](https://github.com/datalab-to/marker) engine. It handles the tedious parts for you: splitting big PDFs, uploading in parallel, retrying, stitching the results back together, and fixing page numbers and image links.
60
+
61
+ ```console
62
+ $ pdf-to-md books/
63
+ ✓ darwin.pdf → darwin.md (516 pages, $1.55)
64
+ ✓ tolstoy.pdf → tolstoy.md (537 pages, $1.61)
65
+ ✓ grimm.pdf → grimm.md (510 pages, $1.53)
66
+ Done in 4m37s: 3 converted · 1563 pages · $4.69
67
+ ```
68
+
69
+ <sub>Output from a real run (file names shortened): three scanned 19th-century books in English, Russian, and German Fraktur.</sub>
70
+
71
+ ## Why pdf-to-md
72
+
73
+ - **Built for big documents.** PDFs are split into page chunks that convert in parallel and are merged back into a single file. Page separators, anchors, and image links stay correct across chunks.
74
+ - **Safe to re-run.** Output names are deterministic (`report.pdf` → `report.md`) and finished files are skipped. Re-running on a folder converts only what's new or failed, so you never pay twice by accident.
75
+ - **Honest about results.** Every file reports its pages and cost. The exit code says whether anything failed. A bad API key stops the run immediately instead of failing file by file.
76
+ - **Resilient.** Rate limits, server errors, and dropped connections are retried with backoff. Ctrl-C stops cleanly without leaving half-written files.
77
+ - **Made for scripting.** Converted paths go to stdout, progress and errors to stderr, so it drops straight into pipelines and CI jobs.
78
+ - **Any language, any era.** Tested on modern PDFs, math, and scans of Cyrillic, blackletter German, and vertical classical Chinese ([samples](#sample-conversions)).
79
+
80
+ ## Installation
81
+
82
+ Requires Python 3.10 or newer. [pipx](https://pipx.pypa.io/) installs it in its own isolated environment:
83
+
84
+ ```bash
85
+ pipx install pdf-to-markdown-cli
86
+ ```
87
+
88
+ `uv tool install pdf-to-markdown-cli` and `pip install pdf-to-markdown-cli` work too.
89
+
90
+ Then get an API key from [datalab.to/app/keys](https://www.datalab.to/app/keys) and export it:
91
+
92
+ ```bash
93
+ export DATALAB_API_KEY="your_api_key"
94
+ ```
95
+
96
+ ## Quick start
97
+
98
+ ```bash
99
+ pdf-to-md report.pdf
100
+ ```
101
+
102
+ This writes `report.md` next to the PDF, plus a `report_images/` folder if the document contains images. That's it.
103
+
104
+ ## Recipes
105
+
106
+ ```bash
107
+ # Mirror a folder tree elsewhere
108
+ pdf-to-md ~/papers -o ~/papers-md
109
+
110
+ # Preview files and pages, at no cost
111
+ pdf-to-md ~/papers --dry-run
112
+
113
+ # Best quality for hard scans and tables
114
+ pdf-to-md scan.pdf --mode accurate
115
+
116
+ # First ten pages plus page 42 (0-based)
117
+ pdf-to-md book.pdf --page-range 0-9,42
118
+
119
+ # HTML or JSON instead of Markdown
120
+ pdf-to-md slides.pptx --html
121
+ pdf-to-md form.pdf --json
122
+
123
+ # Mark page boundaries in the output
124
+ pdf-to-md contract.pdf --paginate
125
+
126
+ # Text only: no images or descriptions
127
+ pdf-to-md manual.pdf --no-images \
128
+ --no-image-captions
129
+
130
+ # Redo everything, replacing old outputs
131
+ pdf-to-md ~/papers --overwrite
132
+
133
+ # Feed converted files to other tools
134
+ pdf-to-md ~/papers -q | xargs wc -w
135
+ ```
136
+
137
+ ## How it works
138
+
139
+ 1. **Plan.** Inputs are discovered recursively. Hidden files and folders this tool generated earlier are ignored. Each input gets a deterministic output path, and inputs whose output already exists are skipped.
140
+ 2. **Split.** PDFs are cut into chunks of `--chunk-size` pages (default 25). Page selection with `--page-range` or `--max-pages` happens here, so only those pages are uploaded and billed. Other formats are uploaded whole.
141
+ 3. **Convert.** Up to `--concurrency` chunks (default 5) are processed at once. Each one is uploaded and then polled until it's done. Throttling and server errors are retried, honoring the API's `Retry-After`.
142
+ 4. **Merge.** Chunk results are joined in page order. Page separators, HTML page ids, JSON block ids, and in-document anchors are renumbered to match the original document. Images are renamed so they never collide, and links to them are URL-encoded.
143
+ 5. **Write.** The output is written to a temporary file and renamed into place, so an interrupted run never leaves a partial file behind.
144
+
145
+ ## Output
146
+
147
+ ```text
148
+ papers/
149
+ ├── report.pdf
150
+ ├── report.md ← the document
151
+ └── report_images/ ← images, if any
152
+ ```
153
+
154
+ - If two inputs would produce the same name (`a.pdf` and `a.docx`), both keep their source extension: `a.pdf.md` and `a.docx.md`. Names are compared case-insensitively, and an output never overwrites an input.
155
+ - Hard-wrapped paragraph lines are joined in Markdown output (disable with `--keep-line-breaks`). Image descriptions live in the image alt text.
156
+
157
+ ## Supported formats
158
+
159
+ | Input | Extensions |
160
+ | --- | --- |
161
+ | PDF | `.pdf` (split into chunks) |
162
+ | Word processing | `.doc` `.docx` `.odt` |
163
+ | Presentations | `.ppt` `.pptx` `.odp` |
164
+ | Spreadsheets | `.xls` `.xlsx` `.xlsm` `.xltx` `.ods` `.csv` |
165
+ | Web and ebooks | `.html` `.htm` `.epub` |
166
+ | Images | `.png` `.jpg` `.jpeg` `.webp` `.gif` `.tiff` `.tif` |
167
+
168
+ Outputs are Markdown (`.md`, the default), HTML (`.html`), or JSON (`.json`, Marker's block tree with page and position data). The API accepts files of up to 200 MB. Non-PDF files are always sent as a single request.
169
+
170
+ ## Quality, speed, and cost
171
+
172
+ `--mode` picks a Datalab processing mode:
173
+
174
+ | Mode | Use it for |
175
+ | --- | --- |
176
+ | `fast` (API default) | Clean digital PDFs and good scans. |
177
+ | `balanced` | Mixed documents with tables, forms, or harder layouts. |
178
+ | `accurate` | The hardest material: poor scans, dense tables, handwriting. |
179
+
180
+ Every result line shows what Datalab billed, and the run summary adds it up. In our September 2026 test runs, conversions cost about 0.3¢ per page ($3 per 1,000 pages). See [Datalab's pricing](https://www.datalab.to/pricing) for current rates.
181
+
182
+ Throughput depends on your plan's limits. With the default of 5 concurrent requests (the free tier's limit), the 1,563-page run above took 4 minutes 37 seconds. Paid plans allow far more concurrency, so raise `-j` to match yours.
183
+
184
+ ## Sample conversions
185
+
186
+ [`examples/`](https://github.com/SokolskyNikita/pdf-to-markdown-cli/tree/main/examples) holds sample documents next to their actual output from this tool:
187
+
188
+ | Sample (links to output) | Language | What it tests |
189
+ | --- | --- | --- |
190
+ | [Alice in Wonderland](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/examples/alice_in_wonderland_sample.md) | English | Illustrated novel, curly quotes |
191
+ | [Algebraic geometry notes](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/examples/equations.md) | English | Mathematics rendered as LaTeX |
192
+ | [*Origin of Species* (1859)](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/examples/darwin_origin_of_species_en.md) | English | Old scan, fold-out tree diagram |
193
+ | [*War and Peace* (1937 ed.)](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/examples/tolstoy_war_and_peace_ru.md) | Russian | Cyrillic mixed with French, footnotes |
194
+ | [*Grimm's Fairy Tales* (1857)](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/examples/grimm_fairy_tales_de.md) | German | Fraktur blackletter typeface |
195
+ | [*Shijing Gupu* (1908)](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/examples/shijing_gupu_zh.md) | Chinese | Vertical text, music notation |
196
+
197
+ See [examples/README.md](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/examples/README.md) for sources and how the samples are used in tests.
198
+
199
+ ## Command reference
200
+
201
+ ```text
202
+ pdf-to-md INPUT [INPUT ...] [options]
203
+ ```
204
+
205
+ `INPUT` is any mix of files and directories. The tool also runs as `python -m docs_to_md`.
206
+
207
+ **Output**
208
+
209
+ | Option | Description |
210
+ | --- | --- |
211
+ | `-f`, `--format FMT` | `markdown` (default), `html`, or `json`. `--html` and `--json` are shortcuts. |
212
+ | `-o`, `--output-dir DIR` | Write here instead of next to each input. Folder structure under directory inputs is mirrored. |
213
+ | `--overwrite` | Replace existing outputs instead of skipping those inputs. |
214
+ | `--keep-line-breaks` | Keep hard-wrapped paragraph lines in Markdown. |
215
+
216
+ **Conversion**
217
+
218
+ | Option | Description |
219
+ | --- | --- |
220
+ | `-m`, `--mode MODE` | `fast` (the API default), `balanced`, or `accurate`. |
221
+ | `--paginate` | Insert page separators. |
222
+ | `--no-images` | Don't extract images. |
223
+ | `--no-image-captions` | Don't generate image descriptions. |
224
+ | `--skip-cache` | Ignore Datalab's server-side cache of previous results. |
225
+ | `--api-option KEY=VALUE` | Send any other [Convert API](https://documentation.datalab.to/api-reference/convert-document) field. Repeatable, e.g. `--api-option extras=extract_links`. |
226
+
227
+ **Pages and chunking**
228
+
229
+ | Option | Description |
230
+ | --- | --- |
231
+ | `--page-range RANGE` | 0-based pages to convert, e.g. `0,5-10`. |
232
+ | `--max-pages N` | Convert at most `N` pages of each document. |
233
+ | `--chunk-size N` | PDF pages per request. Default: `25`. Smaller chunks mean more parallelism but more requests. |
234
+ | `--no-chunk` | Send each PDF as one request. |
235
+
236
+ **Run control**
237
+
238
+ | Option | Description |
239
+ | --- | --- |
240
+ | `--api-key KEY` | API key. Prefer the environment variable, which keeps the key out of shell history. |
241
+ | `-j`, `--concurrency N` | Requests in flight at once. Default: `5`. |
242
+ | `--timeout SECONDS` | Give up on a single request after this long. Default: `3600`. |
243
+ | `-n`, `--dry-run` | Show what would be converted, with page and chunk counts. Makes no API calls and needs no key. |
244
+ | `-q`, `--quiet` | Print only errors (converted paths still go to stdout). |
245
+ | `-v`, `--verbose` | Print debug logs, including every API call and retry. |
246
+ | `--version` | Print the version and exit. |
247
+
248
+ ## Scripting and automation
249
+
250
+ Only converted file paths are printed to stdout, one per line. Status lines, the progress bar, and errors go to stderr.
251
+
252
+ | Exit status | Meaning |
253
+ | --- | --- |
254
+ | `0` | Every input was converted or skipped. |
255
+ | `1` | At least one file failed; the rest were still converted. |
256
+ | `2` | Usage or configuration error, including a rejected API key or an exhausted account. |
257
+ | `130` | Interrupted with Ctrl-C. |
258
+
259
+ Because finished files are skipped, the simplest way to retry failures in an interrupted batch is to run the same command again.
260
+
261
+ ## Configuration
262
+
263
+ | Variable | Purpose |
264
+ | --- | --- |
265
+ | `DATALAB_API_KEY` | API key, the same variable Datalab's SDK uses. The pre-1.0 name `MARKER_PDF_KEY` also works. `--api-key` overrides both. |
266
+ | `NO_COLOR` | Disable colored output. |
267
+
268
+ The tool keeps no state between runs. Temporary chunk files live in the system temp directory and are deleted when the run ends.
269
+
270
+ ## Troubleshooting
271
+
272
+ | Message | What to do |
273
+ | --- | --- |
274
+ | `No API key found` | Set `DATALAB_API_KEY`. |
275
+ | `Authentication failed` | The key was rejected. Check it on the [Datalab dashboard](https://www.datalab.to/app/keys). |
276
+ | `Payment required` | Your Datalab account is out of credits. |
277
+ | `exists (use --overwrite)` | The output is already there. Add `--overwrite` to redo it. |
278
+ | `timed out` | A request took longer than `--timeout`. Raise it, or lower `--chunk-size`. |
279
+ | Frequent throttling | You're above your plan's limits (the free tier allows 10 requests per minute). Lower `-j`. |
280
+
281
+ Anything else: rerun with `-v` and [open an issue](https://github.com/SokolskyNikita/pdf-to-markdown-cli/issues) with the output.
282
+
283
+ ## Upgrading from 0.x
284
+
285
+ Version 1.0 moved to Datalab's current `/api/v1/convert` endpoint and fixed `--html` and `--json`, which produced no output in 0.5.x.
286
+
287
+ - Outputs are now `report.md` + `report_images/` instead of `report_<random>.md` + `images_<random>/`, and existing outputs are skipped unless you pass `--overwrite`.
288
+ - `-o` accepts relative paths and mirrors folder structure.
289
+ - `--llm` now means `--mode balanced` and `--max` means `--mode accurate`. Datalab removed the options behind `--strip`, `--force`, and `--langs`, so those flags are ignored with a warning.
290
+ - `--pages`, `--noimg`, `-cs`, and `-mp` still work as aliases of `--paginate`, `--no-images`, `--chunk-size`, and `--max-pages`.
291
+ - Failures now exit non-zero, and `~/.docs_to_md/` is no longer used and can be deleted.
292
+
293
+ Full details are in the [changelog](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/CHANGELOG.md).
294
+
295
+ ## Contributing
296
+
297
+ Bug reports, ideas, and pull requests are welcome. [CONTRIBUTING.md](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/CONTRIBUTING.md) covers setup, the test suites (including the opt-in live API tests), and the release process.
298
+
299
+ ## Acknowledgements
300
+
301
+ Conversion quality comes from [Marker](https://github.com/datalab-to/marker) and the [Datalab](https://www.datalab.to/) API. This is an independent project and isn't affiliated with Datalab.
302
+
303
+ ## License
304
+
305
+ [MIT](https://github.com/SokolskyNikita/pdf-to-markdown-cli/blob/main/LICENSE) © Nikita Sokolsky