pdf-to-markdown-cli 0.4.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {pdf_to_markdown_cli-0.4.0/src/pdf_to_markdown_cli.egg-info → pdf_to_markdown_cli-0.5.0}/PKG-INFO +27 -2
  2. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/README.md +25 -1
  3. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/pyproject.toml +5 -2
  4. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/core/processor.py +77 -74
  5. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/storage/models.py +2 -1
  6. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0/src/pdf_to_markdown_cli.egg-info}/PKG-INFO +27 -2
  7. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/SOURCES.txt +5 -1
  8. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/requires.txt +2 -0
  9. pdf_to_markdown_cli-0.5.0/tests/test_cli.py +37 -0
  10. pdf_to_markdown_cli-0.5.0/tests/test_paths.py +25 -0
  11. pdf_to_markdown_cli-0.5.0/tests/test_settings.py +65 -0
  12. pdf_to_markdown_cli-0.5.0/tests/test_utils.py +22 -0
  13. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/LICENSE +0 -0
  14. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/MANIFEST.in +0 -0
  15. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/setup.cfg +0 -0
  16. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/__init__.py +0 -0
  17. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/__main__.py +0 -0
  18. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/api/__init__.py +0 -0
  19. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/api/client.py +0 -0
  20. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/api/models.py +0 -0
  21. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/config/__init__.py +0 -0
  22. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/config/cli.py +0 -0
  23. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/config/settings.py +0 -0
  24. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/core/__init__.py +0 -0
  25. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/core/paths.py +0 -0
  26. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/core/result_handler.py +0 -0
  27. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/main.py +0 -0
  28. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/storage/__init__.py +0 -0
  29. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/storage/cache.py +0 -0
  30. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/__init__.py +0 -0
  31. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/exceptions.py +0 -0
  32. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/file_utils.py +0 -0
  33. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/logging.py +0 -0
  34. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/docs_to_md/utils/pdf_splitter.py +0 -0
  35. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/dependency_links.txt +0 -0
  36. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/entry_points.txt +0 -0
  37. {pdf_to_markdown_cli-0.4.0 → pdf_to_markdown_cli-0.5.0}/src/pdf_to_markdown_cli.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pdf-to-markdown-cli
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: CLI tool to convert PDF files (and other documents) to markdown using the Marker API.
5
5
  Author: Nikita Sokolsky
6
6
  License: MIT License
@@ -50,6 +50,7 @@ Requires-Dist: pydantic>=2.0
50
50
  Requires-Dist: ratelimit>=2.0
51
51
  Requires-Dist: requests>=2.0
52
52
  Requires-Dist: tqdm>=4.0
53
+ Provides-Extra: test
53
54
  Dynamic: license-file
54
55
 
55
56
  # PDF to Markdown CLI (via the Datalab Marker API)
@@ -107,6 +108,9 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
107
108
 
108
109
  # Use all Marker OCR enhancements
109
110
  pdf-to-md /path/to/file.pdf --max
111
+
112
+ # Display all available options
113
+ pdf-to-md --help
110
114
  ```
111
115
 
112
116
  ### Full list of CLI options
@@ -125,6 +129,7 @@ pdf-to-md /path/to/file.pdf --max
125
129
  - `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
126
130
  - `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
127
131
  - `-v`, `--verbose`: Enable verbose (DEBUG level) logging
132
+ - `--version`: Show the installed `pdf-to-markdown-cli` version and exit
128
133
 
129
134
  ### Output Structure
130
135
 
@@ -134,6 +139,17 @@ If an output file with the same name already exists, the new file will be automa
134
139
 
135
140
  If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
136
141
 
142
+ ### Core Workflow
143
+
144
+ `MarkerProcessor` drives the conversion process:
145
+
146
+ 1. Discover files with `FileDiscovery`.
147
+ 2. Determine unique output paths for each file and chunk.
148
+ 3. Submit jobs to the Marker API via `BatchProcessor`.
149
+ 4. Poll for results and combine chunk outputs while moving extracted images.
150
+
151
+ Settings are validated in `Config.validate()` and request data is cached with `CacheManager` so interrupted runs can resume.
152
+
137
153
  ### API
138
154
 
139
155
  ```python
@@ -172,6 +188,8 @@ pdf-to-markdown-cli/ (Project Root)
172
188
  └── ... (other config files)
173
189
  ```
174
190
 
191
+ `pyproject.toml` defines `pdf-to-md` as a console script pointing to `docs_to_md.main:main`, so installing the project exposes the `pdf-to-md` command.
192
+
175
193
  ## Requirements
176
194
 
177
195
  - Python 3.10 or higher
@@ -188,11 +206,18 @@ To run the code directly from the source tree without installation:
188
206
  # pdf-to-md /path/to/file.pdf
189
207
 
190
208
  # Or running the module directly:
191
- python -m docs_to_md /path/to/file.pdf
209
+ python -m docs_to_md /path/to/file.pdf
192
210
  ```
193
211
 
194
212
  For regular use after installation, use the `pdf-to-md` command.
195
213
 
214
+ ## Getting Started for Contributors
215
+
216
+ - Run `pdf-to-md --help` to explore all CLI flags. The `examples/` directory contains sample documents for testing.
217
+ - Follow `MarkerProcessor.process()` in `docs_to_md/core/processor.py` to see how jobs are prepared and results combined.
218
+ - Consult `datalab_marker_api_docs.md` for full details on authentication and Marker API parameters.
219
+ - Utilities in `docs_to_md/utils` and the caching layer in `docs_to_md/storage` are useful starting points for deeper customization.
220
+
196
221
  ## License
197
222
 
198
223
  This project is licensed under the MIT License - see the LICENSE file for details.
@@ -53,6 +53,9 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
53
53
 
54
54
  # Use all Marker OCR enhancements
55
55
  pdf-to-md /path/to/file.pdf --max
56
+
57
+ # Display all available options
58
+ pdf-to-md --help
56
59
  ```
57
60
 
58
61
  ### Full list of CLI options
@@ -71,6 +74,7 @@ pdf-to-md /path/to/file.pdf --max
71
74
  - `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
72
75
  - `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
73
76
  - `-v`, `--verbose`: Enable verbose (DEBUG level) logging
77
+ - `--version`: Show the installed `pdf-to-markdown-cli` version and exit
74
78
 
75
79
  ### Output Structure
76
80
 
@@ -80,6 +84,17 @@ If an output file with the same name already exists, the new file will be automa
80
84
 
81
85
  If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
82
86
 
87
+ ### Core Workflow
88
+
89
+ `MarkerProcessor` drives the conversion process:
90
+
91
+ 1. Discover files with `FileDiscovery`.
92
+ 2. Determine unique output paths for each file and chunk.
93
+ 3. Submit jobs to the Marker API via `BatchProcessor`.
94
+ 4. Poll for results and combine chunk outputs while moving extracted images.
95
+
96
+ Settings are validated in `Config.validate()` and request data is cached with `CacheManager` so interrupted runs can resume.
97
+
83
98
  ### API
84
99
 
85
100
  ```python
@@ -118,6 +133,8 @@ pdf-to-markdown-cli/ (Project Root)
118
133
  └── ... (other config files)
119
134
  ```
120
135
 
136
+ `pyproject.toml` defines `pdf-to-md` as a console script pointing to `docs_to_md.main:main`, so installing the project exposes the `pdf-to-md` command.
137
+
121
138
  ## Requirements
122
139
 
123
140
  - Python 3.10 or higher
@@ -134,11 +151,18 @@ To run the code directly from the source tree without installation:
134
151
  # pdf-to-md /path/to/file.pdf
135
152
 
136
153
  # Or running the module directly:
137
- python -m docs_to_md /path/to/file.pdf
154
+ python -m docs_to_md /path/to/file.pdf
138
155
  ```
139
156
 
140
157
  For regular use after installation, use the `pdf-to-md` command.
141
158
 
159
+ ## Getting Started for Contributors
160
+
161
+ - Run `pdf-to-md --help` to explore all CLI flags. The `examples/` directory contains sample documents for testing.
162
+ - Follow `MarkerProcessor.process()` in `docs_to_md/core/processor.py` to see how jobs are prepared and results combined.
163
+ - Consult `datalab_marker_api_docs.md` for full details on authentication and Marker API parameters.
164
+ - Utilities in `docs_to_md/utils` and the caching layer in `docs_to_md/storage` are useful starting points for deeper customization.
165
+
142
166
  ## License
143
167
 
144
168
  This project is licensed under the MIT License - see the LICENSE file for details.
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "pdf-to-markdown-cli"
7
- version = "0.4.0"
7
+ version = "0.5.0"
8
8
  description = "CLI tool to convert PDF files (and other documents) to markdown using the Marker API."
9
9
  readme = "README.md"
10
10
  authors = [
@@ -48,4 +48,7 @@ pdf-to-md = "docs_to_md.main:main"
48
48
 
49
49
  # Just in case: Configure setuptools to find packages in src/
50
50
  [tool.setuptools.packages.find]
51
- where = ["src"]
51
+ where = ["src"]
52
+
53
+ [project.optional-dependencies]
54
+ test = []
@@ -53,6 +53,75 @@ class BatchProcessor:
53
53
  def should_chunk(self, file_path: Path) -> bool:
54
54
  return file_path.suffix.lower() == ".pdf"
55
55
 
56
+ def _chunk_file(
57
+ self, file_path: Path, tmp_dir: Path, request: ConversionRequest
58
+ ) -> bool:
59
+ """Chunk the provided PDF and populate the request object.
60
+
61
+ Returns ``True`` if chunking fails and the request should be marked
62
+ as failed.
63
+ """
64
+ try:
65
+ chunk_result = chunk_pdf_to_temp(str(file_path), self.chunk_size, tmp_dir)
66
+ if chunk_result:
67
+ for chunk_info in chunk_result.chunks:
68
+ request.add_chunk(Path(chunk_info.path), chunk_info.index)
69
+ logger.debug(
70
+ f"Created {len(request.chunks)} chunks in {tmp_dir}"
71
+ )
72
+ else:
73
+ logger.debug(
74
+ f"No chunking needed for {file_path} (<= {self.chunk_size} pages)"
75
+ )
76
+ return False
77
+ except (PDFProcessingError, Exception) as e:
78
+ logger.error(f"Error chunking PDF {file_path}: {e}", exc_info=True)
79
+ request.set_status(Status.FAILED, f"Error chunking PDF: {e}")
80
+ self.cache.save(request)
81
+ return True
82
+
83
+ def _submit_chunks(
84
+ self, request: ConversionRequest, api_params: ApiParams
85
+ ) -> bool:
86
+ """Submit all chunks to the API.
87
+
88
+ Returns ``True`` if any submission fails.
89
+ """
90
+ submission_failed = False
91
+ with ProgressTracker(len(request.chunks), "Submitting to API", "chunk") as progress:
92
+ for chunk in request.ordered_chunks:
93
+ try:
94
+ chunk_request_id = self.client.submit_file(
95
+ chunk.path,
96
+ output_format=api_params.output_format,
97
+ langs=api_params.langs,
98
+ use_llm=api_params.use_llm,
99
+ strip_existing_ocr=api_params.strip_existing_ocr,
100
+ disable_image_extraction=api_params.disable_image_extraction,
101
+ force_ocr=api_params.force_ocr,
102
+ paginate=api_params.paginate,
103
+ max_pages=api_params.max_pages,
104
+ )
105
+ if chunk_request_id:
106
+ chunk.mark_processing(chunk_request_id)
107
+ else:
108
+ chunk.mark_failed(
109
+ f"API submission failed for {chunk.path.name}"
110
+ )
111
+ submission_failed = True
112
+ break
113
+ except Exception as submit_e:
114
+ logger.error(
115
+ f"Unexpected error submitting chunk {chunk.path.name}: {submit_e}",
116
+ exc_info=True,
117
+ )
118
+ chunk.mark_failed(f"Error submitting chunk: {submit_e}")
119
+ submission_failed = True
120
+ break
121
+ finally:
122
+ progress.update()
123
+ return submission_failed or request.has_failed
124
+
56
125
  def process_file(
57
126
  self,
58
127
  file_path: Path,
@@ -60,20 +129,11 @@ class BatchProcessor:
60
129
  api_params: ApiParams,
61
130
  output_paths_obj: OutputPaths,
62
131
  ) -> Optional[str]:
63
- """
64
- Process a single file: chunk if needed, submit to API.
65
-
66
- Args:
67
- file_path: Path to the input file.
68
- final_output_path: Path where the final output markdown file should be saved.
69
- api_params: Parameters for the Marker API call.
70
- output_paths_obj: The OutputPaths object containing final markdown and image paths.
132
+ """Process a single file and submit it to the Marker API.
71
133
 
72
- Returns:
73
- Request ID for tracking if submission is initiated, otherwise None.
74
-
75
- Raises:
76
- FileError: Potentially raised by underlying operations (though many are caught).
134
+ The method now delegates chunking and submission to smaller helpers to
135
+ keep the logic readable. It returns the created request ID, which can
136
+ later be used to poll for results.
77
137
  """
78
138
  with TemporaryDirectory(self.root_tmp_dir, file_path.stem) as tmp_dir:
79
139
  request = ConversionRequest(
@@ -93,30 +153,7 @@ class BatchProcessor:
93
153
 
94
154
  try:
95
155
  if self.should_chunk(file_path):
96
- try:
97
- chunk_result = chunk_pdf_to_temp(
98
- str(file_path), self.chunk_size, tmp_dir
99
- )
100
- if chunk_result:
101
- for chunk_info in chunk_result.chunks:
102
- request.add_chunk(
103
- Path(chunk_info.path), chunk_info.index
104
- )
105
- logger.debug(
106
- f"Created {len(request.chunks)} chunks in {tmp_dir}"
107
- )
108
- else:
109
- logger.debug(
110
- f"No chunking needed for {file_path} (<= {self.chunk_size} pages)"
111
- )
112
- except (PDFProcessingError, Exception) as e:
113
- logger.error(
114
- f"Error chunking PDF {file_path}: {e}", exc_info=True
115
- )
116
- request.set_status(
117
- Status.FAILED, f"Error chunking PDF: {str(e)}"
118
- )
119
- self.cache.save(request)
156
+ if self._chunk_file(file_path, tmp_dir, request):
120
157
  return request.request_id
121
158
 
122
159
  if not request.chunks:
@@ -126,43 +163,9 @@ class BatchProcessor:
126
163
  f"Submitting {len(request.chunks)} chunk(s) to API for {request.original_file.name}..."
127
164
  )
128
165
 
129
- submission_failed = False
130
- with ProgressTracker(
131
- len(request.chunks), "Submitting to API", "chunk"
132
- ) as progress:
133
- for chunk in request.ordered_chunks:
134
- try:
135
- chunk_request_id = self.client.submit_file(
136
- chunk.path,
137
- output_format=api_params.output_format,
138
- langs=api_params.langs,
139
- use_llm=api_params.use_llm,
140
- strip_existing_ocr=api_params.strip_existing_ocr,
141
- disable_image_extraction=api_params.disable_image_extraction,
142
- force_ocr=api_params.force_ocr,
143
- paginate=api_params.paginate,
144
- max_pages=api_params.max_pages,
145
- )
146
- if chunk_request_id:
147
- chunk.mark_processing(chunk_request_id)
148
- else:
149
- chunk.mark_failed(
150
- f"API submission failed for {chunk.path.name}"
151
- )
152
- submission_failed = True
153
- break
154
- except Exception as submit_e:
155
- logger.error(
156
- f"Unexpected error submitting chunk {chunk.path.name}: {submit_e}",
157
- exc_info=True,
158
- )
159
- chunk.mark_failed(f"Error submitting chunk: {submit_e}")
160
- submission_failed = True
161
- break
162
- finally:
163
- progress.update()
164
-
165
- if submission_failed or request.has_failed:
166
+ submission_failed = self._submit_chunks(request, api_params)
167
+
168
+ if submission_failed:
166
169
  request.set_status(
167
170
  Status.FAILED,
168
171
  request.error or "One or more chunk submissions failed.",
@@ -48,7 +48,8 @@ class ConversionRequest(BaseModel):
48
48
  output_format: str = "markdown"
49
49
  status: Status = Status.PENDING
50
50
  error: Optional[str] = None
51
- chunks: List[ChunkInfo] = []
51
+ # Use default_factory to avoid shared mutable list across instances
52
+ chunks: List[ChunkInfo] = Field(default_factory=list)
52
53
  chunk_size: int
53
54
  tmp_dir: Optional[Path] = None # Directory for temporary files for this conversion
54
55
  images_dir: Optional[Path] = None # Added to store determined image path
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pdf-to-markdown-cli
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: CLI tool to convert PDF files (and other documents) to markdown using the Marker API.
5
5
  Author: Nikita Sokolsky
6
6
  License: MIT License
@@ -50,6 +50,7 @@ Requires-Dist: pydantic>=2.0
50
50
  Requires-Dist: ratelimit>=2.0
51
51
  Requires-Dist: requests>=2.0
52
52
  Requires-Dist: tqdm>=4.0
53
+ Provides-Extra: test
53
54
  Dynamic: license-file
54
55
 
55
56
  # PDF to Markdown CLI (via the Datalab Marker API)
@@ -107,6 +108,9 @@ pdf-to-md /path/to/file.pdf --langs "English,French,German"
107
108
 
108
109
  # Use all Marker OCR enhancements
109
110
  pdf-to-md /path/to/file.pdf --max
111
+
112
+ # Display all available options
113
+ pdf-to-md --help
110
114
  ```
111
115
 
112
116
  ### Full list of CLI options
@@ -125,6 +129,7 @@ pdf-to-md /path/to/file.pdf --max
125
129
  - `-cs`, `--chunk-size`: Set PDF chunk size in pages (default: 25)
126
130
  - `-o`, `--output-dir`: Absolute path to the output directory. If not provided, output files will be saved in the same directory as their corresponding input files.
127
131
  - `-v`, `--verbose`: Enable verbose (DEBUG level) logging
132
+ - `--version`: Show the installed `pdf-to-markdown-cli` version and exit
128
133
 
129
134
  ### Output Structure
130
135
 
@@ -134,6 +139,17 @@ If an output file with the same name already exists, the new file will be automa
134
139
 
135
140
  If you specify `--output-dir /path/to/output`, the output file will be placed in that directory (e.g., `/path/to/output/report.md`). The same automatic renaming logic applies if a file with the same name exists in the target directory.
136
141
 
142
+ ### Core Workflow
143
+
144
+ `MarkerProcessor` drives the conversion process:
145
+
146
+ 1. Discover files with `FileDiscovery`.
147
+ 2. Determine unique output paths for each file and chunk.
148
+ 3. Submit jobs to the Marker API via `BatchProcessor`.
149
+ 4. Poll for results and combine chunk outputs while moving extracted images.
150
+
151
+ Settings are validated in `Config.validate()` and request data is cached with `CacheManager` so interrupted runs can resume.
152
+
137
153
  ### API
138
154
 
139
155
  ```python
@@ -172,6 +188,8 @@ pdf-to-markdown-cli/ (Project Root)
172
188
  └── ... (other config files)
173
189
  ```
174
190
 
191
+ `pyproject.toml` defines `pdf-to-md` as a console script pointing to `docs_to_md.main:main`, so installing the project exposes the `pdf-to-md` command.
192
+
175
193
  ## Requirements
176
194
 
177
195
  - Python 3.10 or higher
@@ -188,11 +206,18 @@ To run the code directly from the source tree without installation:
188
206
  # pdf-to-md /path/to/file.pdf
189
207
 
190
208
  # Or running the module directly:
191
- python -m docs_to_md /path/to/file.pdf
209
+ python -m docs_to_md /path/to/file.pdf
192
210
  ```
193
211
 
194
212
  For regular use after installation, use the `pdf-to-md` command.
195
213
 
214
+ ## Getting Started for Contributors
215
+
216
+ - Run `pdf-to-md --help` to explore all CLI flags. The `examples/` directory contains sample documents for testing.
217
+ - Follow `MarkerProcessor.process()` in `docs_to_md/core/processor.py` to see how jobs are prepared and results combined.
218
+ - Consult `datalab_marker_api_docs.md` for full details on authentication and Marker API parameters.
219
+ - Utilities in `docs_to_md/utils` and the caching layer in `docs_to_md/storage` are useful starting points for deeper customization.
220
+
196
221
  ## License
197
222
 
198
223
  This project is licensed under the MIT License - see the LICENSE file for details.
@@ -28,4 +28,8 @@ src/pdf_to_markdown_cli.egg-info/SOURCES.txt
28
28
  src/pdf_to_markdown_cli.egg-info/dependency_links.txt
29
29
  src/pdf_to_markdown_cli.egg-info/entry_points.txt
30
30
  src/pdf_to_markdown_cli.egg-info/requires.txt
31
- src/pdf_to_markdown_cli.egg-info/top_level.txt
31
+ src/pdf_to_markdown_cli.egg-info/top_level.txt
32
+ tests/test_cli.py
33
+ tests/test_paths.py
34
+ tests/test_settings.py
35
+ tests/test_utils.py
@@ -6,3 +6,5 @@ pydantic>=2.0
6
6
  ratelimit>=2.0
7
7
  requests>=2.0
8
8
  tqdm>=4.0
9
+
10
+ [test]
@@ -0,0 +1,37 @@
1
+ import os
2
+ import sys
3
+ import tempfile
4
+ import unittest
5
+ from pathlib import Path
6
+ from unittest import mock
7
+
8
+ from docs_to_md.config.cli import create_config_from_args
9
+
10
+
11
+ class TestCLI(unittest.TestCase):
12
+ def test_create_config_from_args(self):
13
+ with tempfile.TemporaryDirectory() as tmp_dir:
14
+ input_file = Path(tmp_dir) / "input.txt"
15
+ input_file.write_text("data")
16
+ env = {"MARKER_PDF_KEY": "abc"}
17
+ argv = [
18
+ "prog",
19
+ str(input_file),
20
+ "-cs",
21
+ "5",
22
+ "--max",
23
+ "-o",
24
+ tmp_dir,
25
+ ]
26
+ with mock.patch.dict(os.environ, env, clear=False):
27
+ with mock.patch.object(sys, "argv", argv):
28
+ config = create_config_from_args()
29
+ self.assertEqual(config.input_path, str(input_file))
30
+ self.assertEqual(config.chunk_size, 5)
31
+ self.assertTrue(config.use_llm)
32
+ self.assertTrue(config.force_ocr)
33
+ self.assertEqual(config.output_dir, Path(tmp_dir))
34
+
35
+
36
+ if __name__ == "__main__":
37
+ unittest.main()
@@ -0,0 +1,25 @@
1
+ import os
2
+ import tempfile
3
+ import unittest
4
+ from pathlib import Path
5
+
6
+ from docs_to_md.core.paths import determine_output_paths
7
+
8
+
9
+ class TestPaths(unittest.TestCase):
10
+ def test_determine_output_paths_creates_base_dir(self):
11
+ with tempfile.TemporaryDirectory() as tmp:
12
+ tmp_path = Path(tmp)
13
+ input_file = tmp_path / "input.txt"
14
+ input_file.write_text("data")
15
+ out_dir = tmp_path / "out"
16
+ result = determine_output_paths(input_file, out_dir, "markdown")
17
+ self.assertEqual(result.markdown_path.parent, out_dir)
18
+ self.assertEqual(result.images_dir.parent, out_dir)
19
+ self.assertTrue(result.unique_key)
20
+ self.assertEqual(result.markdown_path.suffix, ".md")
21
+ self.assertTrue(out_dir.exists())
22
+
23
+
24
+ if __name__ == "__main__":
25
+ unittest.main()
@@ -0,0 +1,65 @@
1
+ import tempfile
2
+ import unittest
3
+ from pathlib import Path
4
+
5
+ from docs_to_md.config.settings import Config
6
+ from docs_to_md.utils.exceptions import ConfigurationError
7
+
8
+
9
+ class TestSettings(unittest.TestCase):
10
+ def test_config_validation_success(self):
11
+ with tempfile.TemporaryDirectory() as tmp:
12
+ tmp_path = Path(tmp)
13
+ test_file = tmp_path / "input.txt"
14
+ test_file.write_text("data")
15
+ cfg = Config(
16
+ api_key="key",
17
+ input_path=str(test_file),
18
+ output_dir=tmp_path,
19
+ output_format="markdown",
20
+ )
21
+ cfg.validate()
22
+
23
+ def test_config_missing_api_key(self):
24
+ with tempfile.TemporaryDirectory() as tmp:
25
+ tmp_path = Path(tmp)
26
+ test_file = tmp_path / "in.txt"
27
+ test_file.write_text("x")
28
+ cfg = Config(
29
+ api_key="",
30
+ input_path=str(test_file),
31
+ output_dir=tmp_path,
32
+ output_format="markdown",
33
+ )
34
+ with self.assertRaises(ConfigurationError):
35
+ cfg.validate()
36
+
37
+ def test_config_nonexistent_input(self):
38
+ with tempfile.TemporaryDirectory() as tmp:
39
+ tmp_path = Path(tmp)
40
+ cfg = Config(
41
+ api_key="key",
42
+ input_path=str(tmp_path / "no.txt"),
43
+ output_dir=tmp_path,
44
+ output_format="markdown",
45
+ )
46
+ with self.assertRaises(ConfigurationError):
47
+ cfg.validate()
48
+
49
+ def test_config_relative_output_dir(self):
50
+ with tempfile.TemporaryDirectory() as tmp:
51
+ tmp_path = Path(tmp)
52
+ test_file = tmp_path / "file.txt"
53
+ test_file.write_text("x")
54
+ cfg = Config(
55
+ api_key="key",
56
+ input_path=str(test_file),
57
+ output_dir=Path("relative"),
58
+ output_format="markdown",
59
+ )
60
+ with self.assertRaises(ConfigurationError):
61
+ cfg.validate()
62
+
63
+
64
+ if __name__ == "__main__":
65
+ unittest.main()
@@ -0,0 +1,22 @@
1
+ import tempfile
2
+ import unittest
3
+ from pathlib import Path
4
+
5
+ from docs_to_md.utils.file_utils import get_unique_filename
6
+
7
+
8
+ class TestFileUtils(unittest.TestCase):
9
+ def test_get_unique_filename(self):
10
+ with tempfile.TemporaryDirectory() as tmp:
11
+ tmp_path = Path(tmp)
12
+ file_path = tmp_path / "file.txt"
13
+ file_path.write_text("data")
14
+ new_path = get_unique_filename(file_path)
15
+ self.assertNotEqual(new_path, file_path)
16
+ self.assertEqual(new_path.parent, file_path.parent)
17
+ self.assertTrue(new_path.stem.startswith("file_"))
18
+ self.assertEqual(new_path.suffix, ".txt")
19
+
20
+
21
+ if __name__ == "__main__":
22
+ unittest.main()