pdf-to-markdown-cli 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {pdf_to_markdown_cli-0.2.0/pdf_to_markdown_cli.egg-info → pdf_to_markdown_cli-0.2.1}/PKG-INFO +26 -27
  2. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/README.md +25 -26
  3. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1/pdf_to_markdown_cli.egg-info}/PKG-INFO +26 -27
  4. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/setup.py +1 -1
  5. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/LICENSE +0 -0
  6. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/MANIFEST.in +0 -0
  7. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/__init__.py +0 -0
  8. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/__main__.py +0 -0
  9. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/api/__init__.py +0 -0
  10. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/api/client.py +0 -0
  11. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/api/models.py +0 -0
  12. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/config/__init__.py +0 -0
  13. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/config/cli.py +0 -0
  14. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/config/settings.py +0 -0
  15. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/core/__init__.py +0 -0
  16. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/core/processor.py +0 -0
  17. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/core/result_handler.py +0 -0
  18. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/main.py +0 -0
  19. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/pdf/__init__.py +0 -0
  20. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/pdf/splitter.py +0 -0
  21. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/storage/__init__.py +0 -0
  22. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/storage/cache.py +0 -0
  23. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/storage/models.py +0 -0
  24. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/utils/__init__.py +0 -0
  25. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/utils/exceptions.py +0 -0
  26. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/utils/file_utils.py +0 -0
  27. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/docs_to_md/utils/logging.py +0 -0
  28. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/pdf_to_markdown_cli.egg-info/SOURCES.txt +0 -0
  29. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/pdf_to_markdown_cli.egg-info/dependency_links.txt +0 -0
  30. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/pdf_to_markdown_cli.egg-info/entry_points.txt +0 -0
  31. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/pdf_to_markdown_cli.egg-info/requires.txt +0 -0
  32. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/pdf_to_markdown_cli.egg-info/top_level.txt +0 -0
  33. {pdf_to_markdown_cli-0.2.0 → pdf_to_markdown_cli-0.2.1}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pdf-to-markdown-cli
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: CLI tool to convert PDF files (and other documents) to markdown using the Marker API.
5
5
  Home-page: https://github.com/SokolskyNikita/pdf-to-markdown-cli
6
6
  Author: Nikita Sokolsky
@@ -40,9 +40,9 @@ Dynamic: requires-dist
40
40
  Dynamic: requires-python
41
41
  Dynamic: summary
42
42
 
43
- # PDF to Markdown CLI (using Marker API)
43
+ # PDF to Markdown CLI (via the Datalab Marker API)
44
44
 
45
- Convert PDF files (and other documents) to markdown using the [Marker API](https://www.marker.io/) via a command-line tool.
45
+ Convert PDF files (and other documents) to Markdown using the [Marker API](https://www.datalab.to/marker) via a convenient CLI tool.
46
46
 
47
47
  ## Overview
48
48
 
@@ -50,13 +50,12 @@ This package provides a convenient command-line interface (`pdf-to-md`) for conv
50
50
 
51
51
  ## Features
52
52
 
53
- - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to markdown
54
- - Optionally use the Marker API for enhanced PDF conversion
55
- - Handle large documents by splitting them into chunks
53
+ - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to Markdown using the best-in-class Marker API by Datalab
54
+ - Handle large documents by splitting them into chunks, which massively speeds up output speed
56
55
  - Progress tracking for long-running operations
57
- - Customizable OCR options
58
- - Local caching of results
59
- - Output in markdown, JSON, or HTML format
56
+ - Customizable OCR options, fully reflecting Marker's API as of April 2025
57
+ - Local caching of in-progress conversions, allowing for idempotence
58
+ - Output in Markdown, JSON, or HTML format
60
59
 
61
60
  ## Installation
62
61
 
@@ -79,7 +78,7 @@ pip install -e .
79
78
  ### Command-line interface
80
79
 
81
80
  ```bash
82
- # Set your Marker API key
81
+ # Obtain an API key by signing up on https://www.datalab.to/marker
83
82
  export MARKER_PDF_KEY=your_api_key_here
84
83
 
85
84
  # Basic usage
@@ -98,6 +97,23 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
98
97
  pdf-to-md /path/to/file.pdf --max
99
98
  ```
100
99
 
100
+ ### Full list of CLI options
101
+
102
+ - `input`: Input file or directory path
103
+ - `--json`: Output in JSON format (default is markdown)
104
+ - `--langs`: Comma-separated OCR languages (default: "English")
105
+ - `--llm`: Use LLM for enhanced processing
106
+ - `--strip`: Redo OCR processing
107
+ - `--noimg`: Disable image extraction
108
+ - `--force`: Force OCR on all pages
109
+ - `--pages`: Add page delimiters
110
+ - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
111
+ - `--max-pages`: Maximum number of pages to process from the start of the file
112
+ - `--no-chunk`: Disable PDF chunking
113
+ - `--chunk-size`: Set PDF chunk size in pages (default: 25)
114
+ - `--output-dir`: Output directory (default: "converted")
115
+ - `--cache-dir`: Cache directory (default: ".marker_cache")
116
+
101
117
  ### API
102
118
 
103
119
  ```python
@@ -118,23 +134,6 @@ processor = MarkerProcessor(config)
118
134
  processor.process()
119
135
  ```
120
136
 
121
- ## Command-line Options
122
-
123
- - `input`: Input file or directory path
124
- - `--json`: Output in JSON format (default is markdown)
125
- - `--langs`: Comma-separated OCR languages (default: "English")
126
- - `--llm`: Use LLM for enhanced processing
127
- - `--strip`: Redo OCR processing
128
- - `--noimg`: Disable image extraction
129
- - `--force`: Force OCR on all pages
130
- - `--pages`: Add page delimiters
131
- - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
132
- - `--max-pages`: Maximum number of pages to process from the start of the file
133
- - `--no-chunk`: Disable PDF chunking
134
- - `--chunk-size`: Set PDF chunk size in pages (default: 25)
135
- - `--output-dir`: Output directory (default: "converted")
136
- - `--cache-dir`: Cache directory (default: ".marker_cache")
137
-
138
137
  ## Project Structure
139
138
 
140
139
  The package is organized as follows:
@@ -1,6 +1,6 @@
1
- # PDF to Markdown CLI (using Marker API)
1
+ # PDF to Markdown CLI (via the Datalab Marker API)
2
2
 
3
- Convert PDF files (and other documents) to markdown using the [Marker API](https://www.marker.io/) via a command-line tool.
3
+ Convert PDF files (and other documents) to Markdown using the [Marker API](https://www.datalab.to/marker) via a convenient CLI tool.
4
4
 
5
5
  ## Overview
6
6
 
@@ -8,13 +8,12 @@ This package provides a convenient command-line interface (`pdf-to-md`) for conv
8
8
 
9
9
  ## Features
10
10
 
11
- - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to markdown
12
- - Optionally use the Marker API for enhanced PDF conversion
13
- - Handle large documents by splitting them into chunks
11
+ - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to Markdown using the best-in-class Marker API by Datalab
12
+ - Handle large documents by splitting them into chunks, which massively speeds up output speed
14
13
  - Progress tracking for long-running operations
15
- - Customizable OCR options
16
- - Local caching of results
17
- - Output in markdown, JSON, or HTML format
14
+ - Customizable OCR options, fully reflecting Marker's API as of April 2025
15
+ - Local caching of in-progress conversions, allowing for idempotence
16
+ - Output in Markdown, JSON, or HTML format
18
17
 
19
18
  ## Installation
20
19
 
@@ -37,7 +36,7 @@ pip install -e .
37
36
  ### Command-line interface
38
37
 
39
38
  ```bash
40
- # Set your Marker API key
39
+ # Obtain an API key by signing up on https://www.datalab.to/marker
41
40
  export MARKER_PDF_KEY=your_api_key_here
42
41
 
43
42
  # Basic usage
@@ -56,6 +55,23 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
56
55
  pdf-to-md /path/to/file.pdf --max
57
56
  ```
58
57
 
58
+ ### Full list of CLI options
59
+
60
+ - `input`: Input file or directory path
61
+ - `--json`: Output in JSON format (default is markdown)
62
+ - `--langs`: Comma-separated OCR languages (default: "English")
63
+ - `--llm`: Use LLM for enhanced processing
64
+ - `--strip`: Redo OCR processing
65
+ - `--noimg`: Disable image extraction
66
+ - `--force`: Force OCR on all pages
67
+ - `--pages`: Add page delimiters
68
+ - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
69
+ - `--max-pages`: Maximum number of pages to process from the start of the file
70
+ - `--no-chunk`: Disable PDF chunking
71
+ - `--chunk-size`: Set PDF chunk size in pages (default: 25)
72
+ - `--output-dir`: Output directory (default: "converted")
73
+ - `--cache-dir`: Cache directory (default: ".marker_cache")
74
+
59
75
  ### API
60
76
 
61
77
  ```python
@@ -76,23 +92,6 @@ processor = MarkerProcessor(config)
76
92
  processor.process()
77
93
  ```
78
94
 
79
- ## Command-line Options
80
-
81
- - `input`: Input file or directory path
82
- - `--json`: Output in JSON format (default is markdown)
83
- - `--langs`: Comma-separated OCR languages (default: "English")
84
- - `--llm`: Use LLM for enhanced processing
85
- - `--strip`: Redo OCR processing
86
- - `--noimg`: Disable image extraction
87
- - `--force`: Force OCR on all pages
88
- - `--pages`: Add page delimiters
89
- - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
90
- - `--max-pages`: Maximum number of pages to process from the start of the file
91
- - `--no-chunk`: Disable PDF chunking
92
- - `--chunk-size`: Set PDF chunk size in pages (default: 25)
93
- - `--output-dir`: Output directory (default: "converted")
94
- - `--cache-dir`: Cache directory (default: ".marker_cache")
95
-
96
95
  ## Project Structure
97
96
 
98
97
  The package is organized as follows:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pdf-to-markdown-cli
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: CLI tool to convert PDF files (and other documents) to markdown using the Marker API.
5
5
  Home-page: https://github.com/SokolskyNikita/pdf-to-markdown-cli
6
6
  Author: Nikita Sokolsky
@@ -40,9 +40,9 @@ Dynamic: requires-dist
40
40
  Dynamic: requires-python
41
41
  Dynamic: summary
42
42
 
43
- # PDF to Markdown CLI (using Marker API)
43
+ # PDF to Markdown CLI (via the Datalab Marker API)
44
44
 
45
- Convert PDF files (and other documents) to markdown using the [Marker API](https://www.marker.io/) via a command-line tool.
45
+ Convert PDF files (and other documents) to Markdown using the [Marker API](https://www.datalab.to/marker) via a convenient CLI tool.
46
46
 
47
47
  ## Overview
48
48
 
@@ -50,13 +50,12 @@ This package provides a convenient command-line interface (`pdf-to-md`) for conv
50
50
 
51
51
  ## Features
52
52
 
53
- - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to markdown
54
- - Optionally use the Marker API for enhanced PDF conversion
55
- - Handle large documents by splitting them into chunks
53
+ - Convert PDFs, Word documents, PowerPoint files, spreadsheets, epub, HTML, and images to Markdown using the best-in-class Marker API by Datalab
54
+ - Handle large documents by splitting them into chunks, which massively speeds up output speed
56
55
  - Progress tracking for long-running operations
57
- - Customizable OCR options
58
- - Local caching of results
59
- - Output in markdown, JSON, or HTML format
56
+ - Customizable OCR options, fully reflecting Marker's API as of April 2025
57
+ - Local caching of in-progress conversions, allowing for idempotence
58
+ - Output in Markdown, JSON, or HTML format
60
59
 
61
60
  ## Installation
62
61
 
@@ -79,7 +78,7 @@ pip install -e .
79
78
  ### Command-line interface
80
79
 
81
80
  ```bash
82
- # Set your Marker API key
81
+ # Obtain an API key by signing up on https://www.datalab.to/marker
83
82
  export MARKER_PDF_KEY=your_api_key_here
84
83
 
85
84
  # Basic usage
@@ -98,6 +97,23 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
98
97
  pdf-to-md /path/to/file.pdf --max
99
98
  ```
100
99
 
100
+ ### Full list of CLI options
101
+
102
+ - `input`: Input file or directory path
103
+ - `--json`: Output in JSON format (default is markdown)
104
+ - `--langs`: Comma-separated OCR languages (default: "English")
105
+ - `--llm`: Use LLM for enhanced processing
106
+ - `--strip`: Redo OCR processing
107
+ - `--noimg`: Disable image extraction
108
+ - `--force`: Force OCR on all pages
109
+ - `--pages`: Add page delimiters
110
+ - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
111
+ - `--max-pages`: Maximum number of pages to process from the start of the file
112
+ - `--no-chunk`: Disable PDF chunking
113
+ - `--chunk-size`: Set PDF chunk size in pages (default: 25)
114
+ - `--output-dir`: Output directory (default: "converted")
115
+ - `--cache-dir`: Cache directory (default: ".marker_cache")
116
+
101
117
  ### API
102
118
 
103
119
  ```python
@@ -118,23 +134,6 @@ processor = MarkerProcessor(config)
118
134
  processor.process()
119
135
  ```
120
136
 
121
- ## Command-line Options
122
-
123
- - `input`: Input file or directory path
124
- - `--json`: Output in JSON format (default is markdown)
125
- - `--langs`: Comma-separated OCR languages (default: "English")
126
- - `--llm`: Use LLM for enhanced processing
127
- - `--strip`: Redo OCR processing
128
- - `--noimg`: Disable image extraction
129
- - `--force`: Force OCR on all pages
130
- - `--pages`: Add page delimiters
131
- - `--max`: Enable all OCR enhancements (equivalent to --llm --strip --force)
132
- - `--max-pages`: Maximum number of pages to process from the start of the file
133
- - `--no-chunk`: Disable PDF chunking
134
- - `--chunk-size`: Set PDF chunk size in pages (default: 25)
135
- - `--output-dir`: Output directory (default: "converted")
136
- - `--cache-dir`: Cache directory (default: ".marker_cache")
137
-
138
137
  ## Project Structure
139
138
 
140
139
  The package is organized as follows:
@@ -25,7 +25,7 @@ install_requires = [
25
25
  setup(
26
26
  # Core package information
27
27
  name="pdf-to-markdown-cli",
28
- version="0.2.0",
28
+ version="0.2.1",
29
29
  author="Nikita Sokolsky",
30
30
  description="CLI tool to convert PDF files (and other documents) to markdown using the Marker API.",
31
31
  long_description=long_description,