pdf-to-markdown-cli 0.4.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pdf_to_markdown_cli-0.4.0/src/pdf_to_markdown_cli.egg-info → pdf_to_markdown_cli-0.5.0}/PKG-INFO +27 -2
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/README.md +25 -1
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/pyproject.toml +5 -2
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/core/processor.py +77 -74
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/storage/models.py +2 -1
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0/src/pdf_to_markdown_cli.egg-info}/PKG-INFO +27 -2
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/SOURCES.txt +5 -1
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/requires.txt +2 -0
- pdf_to_markdown_cli-0.5.0/tests/test_cli.py +37 -0
- pdf_to_markdown_cli-0.5.0/tests/test_paths.py +25 -0
- pdf_to_markdown_cli-0.5.0/tests/test_settings.py +65 -0
- pdf_to_markdown_cli-0.5.0/tests/test_utils.py +22 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/LICENSE +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/MANIFEST.in +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/setup.cfg +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/__init__.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/__main__.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/api/__init__.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/api/client.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/api/models.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/config/__init__.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/config/cli.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/config/settings.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/core/__init__.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/core/paths.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/core/result_handler.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/main.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/storage/__init__.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/storage/cache.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/__init__.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/exceptions.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/file_utils.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/logging.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/pdf_splitter.py +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/dependency_links.txt +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/entry_points.txt +0 -0
- {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/top_level.txt +0 -0
{pdf_to_markdown_cli-0.4.0/src/pdf_to_markdown_cli.egg-info → pdf_to_markdown_cli-0.5.0}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pdf-to-markdown-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: CLI tool to convert PDF files (and other documents) to markdown using the Marker API.
|
|
5
5
|
Author: Nikita Sokolsky
|
|
6
6
|
License: MIT License
|
|
@@ -50,6 +50,7 @@ Requires-Dist: pydantic>=2.0
|
|
|
50
50
|
Requires-Dist: ratelimit>=2.0
|
|
51
51
|
Requires-Dist: requests>=2.0
|
|
52
52
|
Requires-Dist: tqdm>=4.0
|
|
53
|
+
Provides-Extra: test
|
|
53
54
|
Dynamic: license-file
|
|
54
55
|
|
|
55
56
|
# PDF to Markdown CLI (via the Datalab Marker API)
|
|
@@ -107,6 +108,9 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
|
|
|
107
108
|
|
|
108
109
|
# Use all Marker OCR enhancements
|
|
109
110
|
pdf-to-md /path/to/file.pdf --max
|
|
111
|
+
|
|
112
|
+
# Display all available options
|
|
113
|
+
pdf-to-md --help
|
|
110
114
|
```
|
|
111
115
|
|
|
112
116
|
### Full list of CLI options
|
|
@@ -125,6 +129,7 @@ pdf-to-md /path/to/file.pdf --max
|
|
|
125
129
|
- `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
|
|
126
130
|
- `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
|
|
127
131
|
- `-v`, `--verbose`: Enable verbose (DEBUG level) logging
|
|
132
|
+
- `--version`: Show the installed `pdf-to-markdown-cli` version and exit
|
|
128
133
|
|
|
129
134
|
### Output Structure
|
|
130
135
|
|
|
@@ -134,6 +139,17 @@ If an output file with the same name already exists, the new file will be automa
|
|
|
134
139
|
|
|
135
140
|
If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
|
|
136
141
|
|
|
142
|
+
### Core Workflow
|
|
143
|
+
|
|
144
|
+
`MarkerProcessor` drives the conversion process:
|
|
145
|
+
|
|
146
|
+
1. Discover files with `FileDiscovery`.
|
|
147
|
+
2. Determine unique output paths for each file and chunk.
|
|
148
|
+
3. Submit jobs to the Marker API via `BatchProcessor`.
|
|
149
|
+
4. Poll for results and combine chunk outputs while moving extracted images.
|
|
150
|
+
|
|
151
|
+
Settings are validated in `Config.validate()` and request data is cached with `CacheManager` so interrupted runs can resume.
|
|
152
|
+
|
|
137
153
|
### API
|
|
138
154
|
|
|
139
155
|
```python
|
|
@@ -172,6 +188,8 @@ pdf-to-markdown-cli/ (Project Root)
|
|
|
172
188
|
└── ... (other config files)
|
|
173
189
|
```
|
|
174
190
|
|
|
191
|
+
`pyproject.toml` defines `pdf-to-md` as a console script pointing to `docs_to_md.main:main`, so installing the project exposes the `pdf-to-md` command.
|
|
192
|
+
|
|
175
193
|
## Requirements
|
|
176
194
|
|
|
177
195
|
- Python 3.10 or higher
|
|
@@ -188,11 +206,18 @@ To run the code directly from the source tree without installation:
|
|
|
188
206
|
# pdf-to-md /path/to/file.pdf
|
|
189
207
|
|
|
190
208
|
# Or running the module directly:
|
|
191
|
-
python -m docs_to_md /path/to/file.pdf
|
|
209
|
+
python -m docs_to_md /path/to/file.pdf
|
|
192
210
|
```
|
|
193
211
|
|
|
194
212
|
For regular use after installation, use the `pdf-to-md` command.
|
|
195
213
|
|
|
214
|
+
## Getting Started for Contributors
|
|
215
|
+
|
|
216
|
+
- Run `pdf-to-md --help` to explore all CLI flags. The `examples/` directory contains sample documents for testing.
|
|
217
|
+
- Follow `MarkerProcessor.process()` in `docs_to_md/core/processor.py` to see how jobs are prepared and results combined.
|
|
218
|
+
- Consult `datalab_marker_api_docs.md` for full details on authentication and Marker API parameters.
|
|
219
|
+
- Utilities in `docs_to_md/utils` and the caching layer in `docs_to_md/storage` are useful starting points for deeper customization.
|
|
220
|
+
|
|
196
221
|
## License
|
|
197
222
|
|
|
198
223
|
This project is licensed under the MIT License - see the LICENSE file for details.
|
|
@@ -53,6 +53,9 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
|
|
|
53
53
|
|
|
54
54
|
# Use all Marker OCR enhancements
|
|
55
55
|
pdf-to-md /path/to/file.pdf --max
|
|
56
|
+
|
|
57
|
+
# Display all available options
|
|
58
|
+
pdf-to-md --help
|
|
56
59
|
```
|
|
57
60
|
|
|
58
61
|
### Full list of CLI options
|
|
@@ -71,6 +74,7 @@ pdf-to-md /path/to/file.pdf --max
|
|
|
71
74
|
- `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
|
|
72
75
|
- `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
|
|
73
76
|
- `-v`, `--verbose`: Enable verbose (DEBUG level) logging
|
|
77
|
+
- `--version`: Show the installed `pdf-to-markdown-cli` version and exit
|
|
74
78
|
|
|
75
79
|
### Output Structure
|
|
76
80
|
|
|
@@ -80,6 +84,17 @@ If an output file with the same name already exists, the new file will be automa
|
|
|
80
84
|
|
|
81
85
|
If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
|
|
82
86
|
|
|
87
|
+
### Core Workflow
|
|
88
|
+
|
|
89
|
+
`MarkerProcessor` drives the conversion process:
|
|
90
|
+
|
|
91
|
+
1. Discover files with `FileDiscovery`.
|
|
92
|
+
2. Determine unique output paths for each file and chunk.
|
|
93
|
+
3. Submit jobs to the Marker API via `BatchProcessor`.
|
|
94
|
+
4. Poll for results and combine chunk outputs while moving extracted images.
|
|
95
|
+
|
|
96
|
+
Settings are validated in `Config.validate()` and request data is cached with `CacheManager` so interrupted runs can resume.
|
|
97
|
+
|
|
83
98
|
### API
|
|
84
99
|
|
|
85
100
|
```python
|
|
@@ -118,6 +133,8 @@ pdf-to-markdown-cli/ (Project Root)
|
|
|
118
133
|
└── ... (other config files)
|
|
119
134
|
```
|
|
120
135
|
|
|
136
|
+
`pyproject.toml` defines `pdf-to-md` as a console script pointing to `docs_to_md.main:main`, so installing the project exposes the `pdf-to-md` command.
|
|
137
|
+
|
|
121
138
|
## Requirements
|
|
122
139
|
|
|
123
140
|
- Python 3.10 or higher
|
|
@@ -134,11 +151,18 @@ To run the code directly from the source tree without installation:
|
|
|
134
151
|
# pdf-to-md /path/to/file.pdf
|
|
135
152
|
|
|
136
153
|
# Or running the module directly:
|
|
137
|
-
python -m docs_to_md /path/to/file.pdf
|
|
154
|
+
python -m docs_to_md /path/to/file.pdf
|
|
138
155
|
```
|
|
139
156
|
|
|
140
157
|
For regular use after installation, use the `pdf-to-md` command.
|
|
141
158
|
|
|
159
|
+
## Getting Started for Contributors
|
|
160
|
+
|
|
161
|
+
- Run `pdf-to-md --help` to explore all CLI flags. The `examples/` directory contains sample documents for testing.
|
|
162
|
+
- Follow `MarkerProcessor.process()` in `docs_to_md/core/processor.py` to see how jobs are prepared and results combined.
|
|
163
|
+
- Consult `datalab_marker_api_docs.md` for full details on authentication and Marker API parameters.
|
|
164
|
+
- Utilities in `docs_to_md/utils` and the caching layer in `docs_to_md/storage` are useful starting points for deeper customization.
|
|
165
|
+
|
|
142
166
|
## License
|
|
143
167
|
|
|
144
168
|
This project is licensed under the MIT License - see the LICENSE file for details.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "pdf-to-markdown-cli"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.5.0"
|
|
8
8
|
description = "CLI tool to convert PDF files (and other documents) to markdown using the Marker API."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
authors = [
|
|
@@ -48,4 +48,7 @@ pdf-to-md = "docs_to_md.main:main"
|
|
|
48
48
|
|
|
49
49
|
# Just in case: Configure setuptools to find packages in src/
|
|
50
50
|
[tool.setuptools.packages.find]
|
|
51
|
-
where = ["src"]
|
|
51
|
+
where = ["src"]
|
|
52
|
+
|
|
53
|
+
[project.optional-dependencies]
|
|
54
|
+
test = []
|
|
@@ -53,6 +53,75 @@ class BatchProcessor:
|
|
|
53
53
|
def should_chunk(self, file_path: Path) -> bool:
|
|
54
54
|
return file_path.suffix.lower() == ".pdf"
|
|
55
55
|
|
|
56
|
+
def _chunk_file(
|
|
57
|
+
self, file_path: Path, tmp_dir: Path, request: ConversionRequest
|
|
58
|
+
) -> bool:
|
|
59
|
+
"""Chunk the provided PDF and populate the request object.
|
|
60
|
+
|
|
61
|
+
Returns ``True`` if chunking fails and the request should be marked
|
|
62
|
+
as failed.
|
|
63
|
+
"""
|
|
64
|
+
try:
|
|
65
|
+
chunk_result = chunk_pdf_to_temp(str(file_path), self.chunk_size, tmp_dir)
|
|
66
|
+
if chunk_result:
|
|
67
|
+
for chunk_info in chunk_result.chunks:
|
|
68
|
+
request.add_chunk(Path(chunk_info.path), chunk_info.index)
|
|
69
|
+
logger.debug(
|
|
70
|
+
f"Created {len(request.chunks)} chunks in {tmp_dir}"
|
|
71
|
+
)
|
|
72
|
+
else:
|
|
73
|
+
logger.debug(
|
|
74
|
+
f"No chunking needed for {file_path} (<= {self.chunk_size} pages)"
|
|
75
|
+
)
|
|
76
|
+
return False
|
|
77
|
+
except (PDFProcessingError, Exception) as e:
|
|
78
|
+
logger.error(f"Error chunking PDF {file_path}: {e}", exc_info=True)
|
|
79
|
+
request.set_status(Status.FAILED, f"Error chunking PDF: {e}")
|
|
80
|
+
self.cache.save(request)
|
|
81
|
+
return True
|
|
82
|
+
|
|
83
|
+
def _submit_chunks(
|
|
84
|
+
self, request: ConversionRequest, api_params: ApiParams
|
|
85
|
+
) -> bool:
|
|
86
|
+
"""Submit all chunks to the API.
|
|
87
|
+
|
|
88
|
+
Returns ``True`` if any submission fails.
|
|
89
|
+
"""
|
|
90
|
+
submission_failed = False
|
|
91
|
+
with ProgressTracker(len(request.chunks), "Submitting to API", "chunk") as progress:
|
|
92
|
+
for chunk in request.ordered_chunks:
|
|
93
|
+
try:
|
|
94
|
+
chunk_request_id = self.client.submit_file(
|
|
95
|
+
chunk.path,
|
|
96
|
+
output_format=api_params.output_format,
|
|
97
|
+
langs=api_params.langs,
|
|
98
|
+
use_llm=api_params.use_llm,
|
|
99
|
+
strip_existing_ocr=api_params.strip_existing_ocr,
|
|
100
|
+
disable_image_extraction=api_params.disable_image_extraction,
|
|
101
|
+
force_ocr=api_params.force_ocr,
|
|
102
|
+
paginate=api_params.paginate,
|
|
103
|
+
max_pages=api_params.max_pages,
|
|
104
|
+
)
|
|
105
|
+
if chunk_request_id:
|
|
106
|
+
chunk.mark_processing(chunk_request_id)
|
|
107
|
+
else:
|
|
108
|
+
chunk.mark_failed(
|
|
109
|
+
f"API submission failed for {chunk.path.name}"
|
|
110
|
+
)
|
|
111
|
+
submission_failed = True
|
|
112
|
+
break
|
|
113
|
+
except Exception as submit_e:
|
|
114
|
+
logger.error(
|
|
115
|
+
f"Unexpected error submitting chunk {chunk.path.name}: {submit_e}",
|
|
116
|
+
exc_info=True,
|
|
117
|
+
)
|
|
118
|
+
chunk.mark_failed(f"Error submitting chunk: {submit_e}")
|
|
119
|
+
submission_failed = True
|
|
120
|
+
break
|
|
121
|
+
finally:
|
|
122
|
+
progress.update()
|
|
123
|
+
return submission_failed or request.has_failed
|
|
124
|
+
|
|
56
125
|
def process_file(
|
|
57
126
|
self,
|
|
58
127
|
file_path: Path,
|
|
@@ -60,20 +129,11 @@ class BatchProcessor:
|
|
|
60
129
|
api_params: ApiParams,
|
|
61
130
|
output_paths_obj: OutputPaths,
|
|
62
131
|
) -> Optional[str]:
|
|
63
|
-
"""
|
|
64
|
-
Process a single file: chunk if needed, submit to API.
|
|
65
|
-
|
|
66
|
-
Args:
|
|
67
|
-
file_path: Path to the input file.
|
|
68
|
-
final_output_path: Path where the final output markdown file should be saved.
|
|
69
|
-
api_params: Parameters for the Marker API call.
|
|
70
|
-
output_paths_obj: The OutputPaths object containing final markdown and image paths.
|
|
132
|
+
"""Process a single file and submit it to the Marker API.
|
|
71
133
|
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
Raises:
|
|
76
|
-
FileError: Potentially raised by underlying operations (though many are caught).
|
|
134
|
+
The method now delegates chunking and submission to smaller helpers to
|
|
135
|
+
keep the logic readable. It returns the created request ID, which can
|
|
136
|
+
later be used to poll for results.
|
|
77
137
|
"""
|
|
78
138
|
with TemporaryDirectory(self.root_tmp_dir, file_path.stem) as tmp_dir:
|
|
79
139
|
request = ConversionRequest(
|
|
@@ -93,30 +153,7 @@ class BatchProcessor:
|
|
|
93
153
|
|
|
94
154
|
try:
|
|
95
155
|
if self.should_chunk(file_path):
|
|
96
|
-
|
|
97
|
-
chunk_result = chunk_pdf_to_temp(
|
|
98
|
-
str(file_path), self.chunk_size, tmp_dir
|
|
99
|
-
)
|
|
100
|
-
if chunk_result:
|
|
101
|
-
for chunk_info in chunk_result.chunks:
|
|
102
|
-
request.add_chunk(
|
|
103
|
-
Path(chunk_info.path), chunk_info.index
|
|
104
|
-
)
|
|
105
|
-
logger.debug(
|
|
106
|
-
f"Created {len(request.chunks)} chunks in {tmp_dir}"
|
|
107
|
-
)
|
|
108
|
-
else:
|
|
109
|
-
logger.debug(
|
|
110
|
-
f"No chunking needed for {file_path} (<= {self.chunk_size} pages)"
|
|
111
|
-
)
|
|
112
|
-
except (PDFProcessingError, Exception) as e:
|
|
113
|
-
logger.error(
|
|
114
|
-
f"Error chunking PDF {file_path}: {e}", exc_info=True
|
|
115
|
-
)
|
|
116
|
-
request.set_status(
|
|
117
|
-
Status.FAILED, f"Error chunking PDF: {str(e)}"
|
|
118
|
-
)
|
|
119
|
-
self.cache.save(request)
|
|
156
|
+
if self._chunk_file(file_path, tmp_dir, request):
|
|
120
157
|
return request.request_id
|
|
121
158
|
|
|
122
159
|
if not request.chunks:
|
|
@@ -126,43 +163,9 @@ class BatchProcessor:
|
|
|
126
163
|
f"Submitting {len(request.chunks)} chunk(s) to API for {request.original_file.name}..."
|
|
127
164
|
)
|
|
128
165
|
|
|
129
|
-
submission_failed =
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
) as progress:
|
|
133
|
-
for chunk in request.ordered_chunks:
|
|
134
|
-
try:
|
|
135
|
-
chunk_request_id = self.client.submit_file(
|
|
136
|
-
chunk.path,
|
|
137
|
-
output_format=api_params.output_format,
|
|
138
|
-
langs=api_params.langs,
|
|
139
|
-
use_llm=api_params.use_llm,
|
|
140
|
-
strip_existing_ocr=api_params.strip_existing_ocr,
|
|
141
|
-
disable_image_extraction=api_params.disable_image_extraction,
|
|
142
|
-
force_ocr=api_params.force_ocr,
|
|
143
|
-
paginate=api_params.paginate,
|
|
144
|
-
max_pages=api_params.max_pages,
|
|
145
|
-
)
|
|
146
|
-
if chunk_request_id:
|
|
147
|
-
chunk.mark_processing(chunk_request_id)
|
|
148
|
-
else:
|
|
149
|
-
chunk.mark_failed(
|
|
150
|
-
f"API submission failed for {chunk.path.name}"
|
|
151
|
-
)
|
|
152
|
-
submission_failed = True
|
|
153
|
-
break
|
|
154
|
-
except Exception as submit_e:
|
|
155
|
-
logger.error(
|
|
156
|
-
f"Unexpected error submitting chunk {chunk.path.name}: {submit_e}",
|
|
157
|
-
exc_info=True,
|
|
158
|
-
)
|
|
159
|
-
chunk.mark_failed(f"Error submitting chunk: {submit_e}")
|
|
160
|
-
submission_failed = True
|
|
161
|
-
break
|
|
162
|
-
finally:
|
|
163
|
-
progress.update()
|
|
164
|
-
|
|
165
|
-
if submission_failed or request.has_failed:
|
|
166
|
+
submission_failed = self._submit_chunks(request, api_params)
|
|
167
|
+
|
|
168
|
+
if submission_failed:
|
|
166
169
|
request.set_status(
|
|
167
170
|
Status.FAILED,
|
|
168
171
|
request.error or "One or more chunk submissions failed.",
|
|
@@ -48,7 +48,8 @@ class ConversionRequest(BaseModel):
|
|
|
48
48
|
output_format: str = "markdown"
|
|
49
49
|
status: Status = Status.PENDING
|
|
50
50
|
error: Optional[str] = None
|
|
51
|
-
|
|
51
|
+
# Use default_factory to avoid shared mutable list across instances
|
|
52
|
+
chunks: List[ChunkInfo] = Field(default_factory=list)
|
|
52
53
|
chunk_size: int
|
|
53
54
|
tmp_dir: Optional[Path] = None # Directory for temporary files for this conversion
|
|
54
55
|
images_dir: Optional[Path] = None # Added to store determined image path
|
{pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0/src/pdf_to_markdown_cli.egg-info}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pdf-to-markdown-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: CLI tool to convert PDF files (and other documents) to markdown using the Marker API.
|
|
5
5
|
Author: Nikita Sokolsky
|
|
6
6
|
License: MIT License
|
|
@@ -50,6 +50,7 @@ Requires-Dist: pydantic>=2.0
|
|
|
50
50
|
Requires-Dist: ratelimit>=2.0
|
|
51
51
|
Requires-Dist: requests>=2.0
|
|
52
52
|
Requires-Dist: tqdm>=4.0
|
|
53
|
+
Provides-Extra: test
|
|
53
54
|
Dynamic: license-file
|
|
54
55
|
|
|
55
56
|
# PDF to Markdown CLI (via the Datalab Marker API)
|
|
@@ -107,6 +108,9 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
|
|
|
107
108
|
|
|
108
109
|
# Use all Marker OCR enhancements
|
|
109
110
|
pdf-to-md /path/to/file.pdf --max
|
|
111
|
+
|
|
112
|
+
# Display all available options
|
|
113
|
+
pdf-to-md --help
|
|
110
114
|
```
|
|
111
115
|
|
|
112
116
|
### Full list of CLI options
|
|
@@ -125,6 +129,7 @@ pdf-to-md /path/to/file.pdf --max
|
|
|
125
129
|
- `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
|
|
126
130
|
- `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
|
|
127
131
|
- `-v`, `--verbose`: Enable verbose (DEBUG level) logging
|
|
132
|
+
- `--version`: Show the installed `pdf-to-markdown-cli` version and exit
|
|
128
133
|
|
|
129
134
|
### Output Structure
|
|
130
135
|
|
|
@@ -134,6 +139,17 @@ If an output file with the same name already exists, the new file will be automa
|
|
|
134
139
|
|
|
135
140
|
If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
|
|
136
141
|
|
|
142
|
+
### Core Workflow
|
|
143
|
+
|
|
144
|
+
`MarkerProcessor` drives the conversion process:
|
|
145
|
+
|
|
146
|
+
1. Discover files with `FileDiscovery`.
|
|
147
|
+
2. Determine unique output paths for each file and chunk.
|
|
148
|
+
3. Submit jobs to the Marker API via `BatchProcessor`.
|
|
149
|
+
4. Poll for results and combine chunk outputs while moving extracted images.
|
|
150
|
+
|
|
151
|
+
Settings are validated in `Config.validate()` and request data is cached with `CacheManager` so interrupted runs can resume.
|
|
152
|
+
|
|
137
153
|
### API
|
|
138
154
|
|
|
139
155
|
```python
|
|
@@ -172,6 +188,8 @@ pdf-to-markdown-cli/ (Project Root)
|
|
|
172
188
|
└── ... (other config files)
|
|
173
189
|
```
|
|
174
190
|
|
|
191
|
+
`pyproject.toml` defines `pdf-to-md` as a console script pointing to `docs_to_md.main:main`, so installing the project exposes the `pdf-to-md` command.
|
|
192
|
+
|
|
175
193
|
## Requirements
|
|
176
194
|
|
|
177
195
|
- Python 3.10 or higher
|
|
@@ -188,11 +206,18 @@ To run the code directly from the source tree without installation:
|
|
|
188
206
|
# pdf-to-md /path/to/file.pdf
|
|
189
207
|
|
|
190
208
|
# Or running the module directly:
|
|
191
|
-
python -m docs_to_md /path/to/file.pdf
|
|
209
|
+
python -m docs_to_md /path/to/file.pdf
|
|
192
210
|
```
|
|
193
211
|
|
|
194
212
|
For regular use after installation, use the `pdf-to-md` command.
|
|
195
213
|
|
|
214
|
+
## Getting Started for Contributors
|
|
215
|
+
|
|
216
|
+
- Run `pdf-to-md --help` to explore all CLI flags. The `examples/` directory contains sample documents for testing.
|
|
217
|
+
- Follow `MarkerProcessor.process()` in `docs_to_md/core/processor.py` to see how jobs are prepared and results combined.
|
|
218
|
+
- Consult `datalab_marker_api_docs.md` for full details on authentication and Marker API parameters.
|
|
219
|
+
- Utilities in `docs_to_md/utils` and the caching layer in `docs_to_md/storage` are useful starting points for deeper customization.
|
|
220
|
+
|
|
196
221
|
## License
|
|
197
222
|
|
|
198
223
|
This project is licensed under the MIT License - see the LICENSE file for details.
|
{pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/SOURCES.txt
RENAMED
|
@@ -28,4 +28,8 @@ src/pdf_to_markdown_cli.egg-info/SOURCES.txt
|
|
|
28
28
|
src/pdf_to_markdown_cli.egg-info/dependency_links.txt
|
|
29
29
|
src/pdf_to_markdown_cli.egg-info/entry_points.txt
|
|
30
30
|
src/pdf_to_markdown_cli.egg-info/requires.txt
|
|
31
|
-
src/pdf_to_markdown_cli.egg-info/top_level.txt
|
|
31
|
+
src/pdf_to_markdown_cli.egg-info/top_level.txt
|
|
32
|
+
tests/test_cli.py
|
|
33
|
+
tests/test_paths.py
|
|
34
|
+
tests/test_settings.py
|
|
35
|
+
tests/test_utils.py
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import sys
|
|
3
|
+
import tempfile
|
|
4
|
+
import unittest
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from unittest import mock
|
|
7
|
+
|
|
8
|
+
from docs_to_md.config.cli import create_config_from_args
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class TestCLI(unittest.TestCase):
|
|
12
|
+
def test_create_config_from_args(self):
|
|
13
|
+
with tempfile.TemporaryDirectory() as tmp_dir:
|
|
14
|
+
input_file = Path(tmp_dir) / "input.txt"
|
|
15
|
+
input_file.write_text("data")
|
|
16
|
+
env = {"MARKER_PDF_KEY": "abc"}
|
|
17
|
+
argv = [
|
|
18
|
+
"prog",
|
|
19
|
+
str(input_file),
|
|
20
|
+
"-cs",
|
|
21
|
+
"5",
|
|
22
|
+
"--max",
|
|
23
|
+
"-o",
|
|
24
|
+
tmp_dir,
|
|
25
|
+
]
|
|
26
|
+
with mock.patch.dict(os.environ, env, clear=False):
|
|
27
|
+
with mock.patch.object(sys, "argv", argv):
|
|
28
|
+
config = create_config_from_args()
|
|
29
|
+
self.assertEqual(config.input_path, str(input_file))
|
|
30
|
+
self.assertEqual(config.chunk_size, 5)
|
|
31
|
+
self.assertTrue(config.use_llm)
|
|
32
|
+
self.assertTrue(config.force_ocr)
|
|
33
|
+
self.assertEqual(config.output_dir, Path(tmp_dir))
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
if __name__ == "__main__":
|
|
37
|
+
unittest.main()
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import tempfile
|
|
3
|
+
import unittest
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from docs_to_md.core.paths import determine_output_paths
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class TestPaths(unittest.TestCase):
|
|
10
|
+
def test_determine_output_paths_creates_base_dir(self):
|
|
11
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
12
|
+
tmp_path = Path(tmp)
|
|
13
|
+
input_file = tmp_path / "input.txt"
|
|
14
|
+
input_file.write_text("data")
|
|
15
|
+
out_dir = tmp_path / "out"
|
|
16
|
+
result = determine_output_paths(input_file, out_dir, "markdown")
|
|
17
|
+
self.assertEqual(result.markdown_path.parent, out_dir)
|
|
18
|
+
self.assertEqual(result.images_dir.parent, out_dir)
|
|
19
|
+
self.assertTrue(result.unique_key)
|
|
20
|
+
self.assertEqual(result.markdown_path.suffix, ".md")
|
|
21
|
+
self.assertTrue(out_dir.exists())
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
if __name__ == "__main__":
|
|
25
|
+
unittest.main()
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
import tempfile
|
|
2
|
+
import unittest
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from docs_to_md.config.settings import Config
|
|
6
|
+
from docs_to_md.utils.exceptions import ConfigurationError
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class TestSettings(unittest.TestCase):
|
|
10
|
+
def test_config_validation_success(self):
|
|
11
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
12
|
+
tmp_path = Path(tmp)
|
|
13
|
+
test_file = tmp_path / "input.txt"
|
|
14
|
+
test_file.write_text("data")
|
|
15
|
+
cfg = Config(
|
|
16
|
+
api_key="key",
|
|
17
|
+
input_path=str(test_file),
|
|
18
|
+
output_dir=tmp_path,
|
|
19
|
+
output_format="markdown",
|
|
20
|
+
)
|
|
21
|
+
cfg.validate()
|
|
22
|
+
|
|
23
|
+
def test_config_missing_api_key(self):
|
|
24
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
25
|
+
tmp_path = Path(tmp)
|
|
26
|
+
test_file = tmp_path / "in.txt"
|
|
27
|
+
test_file.write_text("x")
|
|
28
|
+
cfg = Config(
|
|
29
|
+
api_key="",
|
|
30
|
+
input_path=str(test_file),
|
|
31
|
+
output_dir=tmp_path,
|
|
32
|
+
output_format="markdown",
|
|
33
|
+
)
|
|
34
|
+
with self.assertRaises(ConfigurationError):
|
|
35
|
+
cfg.validate()
|
|
36
|
+
|
|
37
|
+
def test_config_nonexistent_input(self):
|
|
38
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
39
|
+
tmp_path = Path(tmp)
|
|
40
|
+
cfg = Config(
|
|
41
|
+
api_key="key",
|
|
42
|
+
input_path=str(tmp_path / "no.txt"),
|
|
43
|
+
output_dir=tmp_path,
|
|
44
|
+
output_format="markdown",
|
|
45
|
+
)
|
|
46
|
+
with self.assertRaises(ConfigurationError):
|
|
47
|
+
cfg.validate()
|
|
48
|
+
|
|
49
|
+
def test_config_relative_output_dir(self):
|
|
50
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
51
|
+
tmp_path = Path(tmp)
|
|
52
|
+
test_file = tmp_path / "file.txt"
|
|
53
|
+
test_file.write_text("x")
|
|
54
|
+
cfg = Config(
|
|
55
|
+
api_key="key",
|
|
56
|
+
input_path=str(test_file),
|
|
57
|
+
output_dir=Path("relative"),
|
|
58
|
+
output_format="markdown",
|
|
59
|
+
)
|
|
60
|
+
with self.assertRaises(ConfigurationError):
|
|
61
|
+
cfg.validate()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
if __name__ == "__main__":
|
|
65
|
+
unittest.main()
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import tempfile
|
|
2
|
+
import unittest
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
from docs_to_md.utils.file_utils import get_unique_filename
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class TestFileUtils(unittest.TestCase):
|
|
9
|
+
def test_get_unique_filename(self):
|
|
10
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
11
|
+
tmp_path = Path(tmp)
|
|
12
|
+
file_path = tmp_path / "file.txt"
|
|
13
|
+
file_path.write_text("data")
|
|
14
|
+
new_path = get_unique_filename(file_path)
|
|
15
|
+
self.assertNotEqual(new_path, file_path)
|
|
16
|
+
self.assertEqual(new_path.parent, file_path.parent)
|
|
17
|
+
self.assertTrue(new_path.stem.startswith("file_"))
|
|
18
|
+
self.assertEqual(new_path.suffix, ".txt")
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
if __name__ == "__main__":
|
|
22
|
+
unittest.main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/core/result_handler.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/pdf_splitter.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|