ebook2text 2.2.0__tar.gz → 2.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. {ebook2text-2.2.0 → ebook2text-2.3.2}/PKG-INFO +67 -21
  2. {ebook2text-2.2.0 → ebook2text-2.3.2}/README.md +55 -13
  3. ebook2text-2.3.2/ebook2text/VERSION.py +1 -0
  4. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/__init__.py +3 -0
  5. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/_exceptions.py +24 -0
  6. ebook2text-2.3.2/ebook2text/ai_providers/__init__.py +15 -0
  7. ebook2text-2.3.2/ebook2text/ai_providers/_anthropic.py +50 -0
  8. ebook2text-2.3.2/ebook2text/ai_providers/_base.py +168 -0
  9. ebook2text-2.3.2/ebook2text/ai_providers/_factory.py +70 -0
  10. ebook2text-2.3.2/ebook2text/ai_providers/_gemini.py +47 -0
  11. ebook2text-2.3.2/ebook2text/ai_providers/_openai.py +55 -0
  12. ebook2text-2.3.2/ebook2text/ai_providers/_openrouter.py +19 -0
  13. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/convert_file.py +14 -5
  14. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/__init__.py +15 -4
  15. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/docx_text_extractor.py +15 -3
  16. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/epub_conversion/__init__.py +13 -4
  17. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/epub_conversion/epub_text_extractor.py +22 -3
  18. ebook2text-2.3.2/ebook2text/ocr.py +33 -0
  19. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/__init__.py +21 -4
  20. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/pdf_text_extractor.py +15 -3
  21. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text.egg-info/PKG-INFO +67 -21
  22. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text.egg-info/SOURCES.txt +8 -0
  23. ebook2text-2.3.2/ebook2text.egg-info/requires.txt +16 -0
  24. ebook2text-2.3.2/pyproject.toml +88 -0
  25. ebook2text-2.3.2/tests/test_convert_file.py +25 -0
  26. {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_docx_conversion.py +8 -4
  27. {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_epub_conversion.py +2 -7
  28. ebook2text-2.3.2/tests/test_ocr.py +55 -0
  29. {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_pdf_conversion.py +6 -12
  30. ebook2text-2.2.0/ebook2text/VERSION.py +0 -1
  31. ebook2text-2.2.0/ebook2text/ocr.py +0 -117
  32. ebook2text-2.2.0/ebook2text.egg-info/requires.txt +0 -7
  33. ebook2text-2.2.0/pyproject.toml +0 -75
  34. ebook2text-2.2.0/tests/test_ocr.py +0 -137
  35. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/_logger.py +0 -0
  36. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/_types.py +0 -0
  37. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/chapter_check.py +0 -0
  38. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/_namespaces.py +0 -0
  39. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/docx_converter.py +0 -0
  40. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/docx_image_extractor.py +0 -0
  41. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/epub_conversion/epub_converter.py +0 -0
  42. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/_enums.py +0 -0
  43. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/pdf_converter.py +0 -0
  44. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/pdf_image_extractor.py +0 -0
  45. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/pdf_line_logic.py +0 -0
  46. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/text_parser.py +0 -0
  47. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/text_utilities.py +0 -0
  48. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text.egg-info/dependency_links.txt +0 -0
  49. {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text.egg-info/top_level.txt +0 -0
  50. {ebook2text-2.2.0 → ebook2text-2.3.2}/setup.cfg +0 -0
  51. {ebook2text-2.2.0 → ebook2text-2.3.2}/setup.py +0 -0
  52. {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_chapter_check.py +0 -0
  53. {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_pdf_image_helpers.py +0 -0
  54. {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_text_conversion.py +0 -0
  55. {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_text_parser.py +0 -0
@@ -1,54 +1,58 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ebook2text
3
- Version: 2.2.0
3
+ Version: 2.3.2
4
4
  Summary: Convert common book file types to text for machine learning
5
5
  Author: Ashlynn Antrobus
6
6
  Author-email: Ashlynn Antrobus <ashlynn@prosepal.io>
7
- License: MIT
8
- Project-URL: Repository, https://github.com/ashrobertsdragon/Ebook-conversion-to-Text-for-Machine-Learning
9
- Classifier: License :: OSI Approved :: MIT License
7
+ Project-URL: Repository, https://github.com/ashrobertsdragon/ebook2text
10
8
  Classifier: Development Status :: 5 - Production/Stable
11
- Classifier: Intended Audience :: Developers
12
9
  Classifier: Environment :: Console
10
+ Classifier: Intended Audience :: Developers
13
11
  Classifier: Operating System :: OS Independent
14
12
  Classifier: Programming Language :: Python
15
13
  Classifier: Programming Language :: Python :: 3
16
14
  Classifier: Programming Language :: Python :: 3 :: Only
17
- Classifier: Programming Language :: Python :: 3.9
18
15
  Classifier: Programming Language :: Python :: 3.10
19
16
  Classifier: Programming Language :: Python :: 3.11
20
17
  Classifier: Programming Language :: Python :: 3.12
21
18
  Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Programming Language :: Python :: 3.14
22
20
  Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
21
  Classifier: Topic :: Text Processing
24
22
  Requires-Python: >=3.10
25
23
  Description-Content-Type: text/markdown
26
24
  Requires-Dist: beautifulsoup4>=4.12.3
27
- Requires-Dist: ebooklib-autoupdate>=0.18.3
25
+ Requires-Dist: ebooklib>=0.20
28
26
  Requires-Dist: openai>=1.54.3
29
27
  Requires-Dist: pdfminer-six>=20240706
30
- Requires-Dist: pillow>=10.4.0
28
+ Requires-Dist: pillow>=12.3.0
31
29
  Requires-Dist: python-docx>=1.1.2
32
30
  Requires-Dist: python-dotenv>=1.0.1
31
+ Provides-Extra: all
32
+ Requires-Dist: ebook2text[anthropic,gemini]; extra == "all"
33
+ Provides-Extra: anthropic
34
+ Requires-Dist: anthropic>=0.40.0; extra == "anthropic"
35
+ Provides-Extra: gemini
36
+ Requires-Dist: google-genai>=1.0.0; extra == "gemini"
33
37
  Dynamic: author
34
38
 
35
-
36
39
  # Ebook2Text
37
40
 
38
41
  ## Overview
39
42
 
40
- This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using GPT-4o and standardizes the text by converting smart punctuation.
43
+ This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using a vision-capable AI model and standardizes the text by converting smart punctuation.
41
44
 
42
45
  ## Features
43
46
 
44
47
  - **File Format Support**: Handles EPUB, DOCX, PDF, and TXT formats.
45
48
  - **Chapter Identification**: Detects and marks chapter breaks.
46
49
  - **OCR Capability**: Converts text from images using OCR.
50
+ - **Multiple AI Providers**: OpenAI, Google Gemini, Anthropic, and OpenRouter.
47
51
  - **Text Standardization**: Replaces smart punctuation with ASCII equivalents.
48
52
 
49
53
  ## Requirements
50
54
 
51
- To run this script, you need Python 3.9 or above and the following packages:
55
+ To run this script, you need Python 3.10 or above and the following packages:
52
56
 
53
57
  - `bs4`
54
58
  - `ebooklib-autoupdate`
@@ -58,11 +62,45 @@ To run this script, you need Python 3.9 or above and the following packages:
58
62
  - `python-dotenv`
59
63
  - `openai`
60
64
 
65
+ The OpenAI SDK is a core dependency and also serves OpenRouter. Gemini and Anthropic need optional extras:
66
+
67
+ ```bash
68
+ pip install ebook2text[gemini] # Google Gemini
69
+ pip install ebook2text[anthropic] # Anthropic Claude
70
+ pip install ebook2text[all] # both
71
+ ```
72
+
73
+ ## AI provider configuration
74
+
75
+ OCR runs through one of four providers, selected by environment variable. See `.env.sample` for a complete template.
76
+
77
+ | Variable | Purpose | Default |
78
+ | -------------------------------------------------------------------------------- | ------------------------------------------------ | ---------------------------- |
79
+ | `OCR_PROVIDER` | `openai`, `gemini`, `anthropic`, or `openrouter` | `openai` |
80
+ | `OCR_MODEL` | Model name for the chosen provider | falls back to `OPENAI_MODEL` |
81
+ | `OCR_MAX_TOKENS` | Maximum tokens in the OCR response | `1000` |
82
+ | `OPENAI_API_KEY` / `GEMINI_API_KEY` / `ANTHROPIC_API_KEY` / `OPENROUTER_API_KEY` | API key for the selected provider | — |
83
+
84
+ API keys and models are read the first time an image is actually processed, so importing the library and converting text-only books requires no configuration at all.
85
+
86
+ To choose a provider programmatically instead, pass one to `convert_file`:
87
+
88
+ ```python
89
+ from ebook2text import convert_file, get_ocr_provider
90
+
91
+ provider = get_ocr_provider(provider="anthropic", model="claude-haiku-4-5")
92
+ text = convert_file(
93
+ file_path, metadata, save_file=False, ocr_provider=provider
94
+ )
95
+ ```
96
+
97
+ Any object with a `perform_ocr(base64_images: list[str]) -> str` method satisfies the `OCRProvider` protocol, so you can supply your own implementation.
98
+
61
99
  ## Usage
62
100
 
63
101
  1. Ensure all dependencies are installed.
64
- 2. Set your environment variable for the OpenAI API key.
65
- 3. Run `convert_file` from the `convert_file` module with the path to the ebook file and a metadata dictionary with keys of 'title' and 'author' as arguments.
102
+ 1. Set your environment variables for the AI provider (see above).
103
+ 1. Run `convert_file` from the `convert_file` module with the path to the ebook file and a metadata dictionary with keys of 'title' and 'author' as arguments.
66
104
 
67
105
  - set `save_file` to False, if you want a string returned.
68
106
  - set `save_file` to True or leave blank, and provide a Path object to `save_path` to use a custom output filename.
@@ -95,7 +133,7 @@ Converts an ebook file to a standardized text format.
95
133
  `ebook2text.convert_file.py`
96
134
 
97
135
  **Signature**:
98
- `convert_file(file_path: Path, metadata: dict, *, save_file: bool = True, save_path: Optional[Path] = None) -> Union[str, None]`
136
+ `convert_file(file_path: Path, metadata: dict, *, save_file: bool = True, save_path: Path | None = None, ocr_provider: OCRProvider | None = None) -> str | None`
99
137
 
100
138
  **Arguments**:
101
139
 
@@ -103,7 +141,10 @@ Converts an ebook file to a standardized text format.
103
141
  - `metadata`: Dictionary containing the book's `title` and `author`.
104
142
  - `save_file`: Boolean flag. If `True`, saves the converted text to a file; otherwise, returns it as a string. Defaults to `True`.
105
143
  - `save_path`: Optional path to save the output file. Defaults to a generated name in the input file's directory.
144
+ - `ocr_provider`: Optional OCR provider for image text extraction. Defaults to the provider configured by environment variables.
145
+
106
146
  **Returns**:
147
+
107
148
  - If `save_file` is `True`: Returns `None`.
108
149
  - If `save_file` is `False`: Returns the converted text as a string.
109
150
 
@@ -119,12 +160,13 @@ Initializes a PDFConverter instance for handling PDF files.
119
160
  `ebook2_text.pdf_converter`
120
161
 
121
162
  **Signature**:
122
- `initialize_pdf_converter(file_path: Path, metadata: dict) -> PDFConverter`
163
+ `initialize_pdf_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> PDFConverter`
123
164
 
124
165
  **Arguments**:
125
166
 
126
167
  - `file_path`: Path to the PDF file to be processed.
127
168
  - `metadata`: Dictionary containing `title` and `author`.
169
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
128
170
 
129
171
  **Returns**:
130
172
 
@@ -139,12 +181,13 @@ Convenience function for reading and processing a PDF file, splitting its conten
139
181
 
140
182
  **Signature**:
141
183
 
142
- convert_pdf(file_path: Path, metadata: dict) -> Generator[str, None, None]
184
+ convert_pdf(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
143
185
 
144
186
  **Arguments**:
145
187
 
146
188
  - `file_path`: Path to the PDF file to be processed.
147
189
  - `metadata`: Dictionary containing `title` and `author`.
190
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
148
191
 
149
192
  **Yields**:
150
193
 
@@ -176,12 +219,13 @@ Initializes a EpubConverter instance for handling Epub files.
176
219
  `ebook2_text.epub_converter`
177
220
 
178
221
  **Signature**:
179
- `initialize_epub_converter(file_path: Path, metadata: dict) -> EpubConverter`
222
+ `initialize_epub_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> EpubConverter`
180
223
 
181
224
  **Arguments**:
182
225
 
183
226
  - `file_path`: Path to the Epub file to be processed.
184
227
  - `metadata`: Dictionary containing `title` and `author`.
228
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
185
229
 
186
230
  **Returns**:
187
231
 
@@ -196,12 +240,13 @@ Convenience function for reading and processing a Epub file, splitting its conte
196
240
 
197
241
  **Signature**:
198
242
 
199
- convert_epub(file_path: Path, metadata: dict) -> Generator[str, None, None]
243
+ convert_epub(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
200
244
 
201
245
  **Arguments**:
202
246
 
203
247
  - `file_path`: Path to the Epub file to be processed.
204
248
  - `metadata`: Dictionary containing `title` and `author`.
249
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
205
250
 
206
251
  **Yields**:
207
252
 
@@ -233,12 +278,13 @@ Initializes a DocxConverter instance for handling Docx files.
233
278
  `ebook2_text.docx_converter`
234
279
 
235
280
  **Signature**:
236
- `initialize_docx_converter(file_path: Path, metadata: dict) -> DocxConverter`
281
+ `initialize_docx_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> DocxConverter`
237
282
 
238
283
  **Arguments**:
239
284
 
240
285
  - `file_path`: Path to the Docx file to be processed.
241
286
  - `metadata`: Dictionary containing `title` and `author`.
287
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
242
288
 
243
289
  **Returns**:
244
290
 
@@ -253,12 +299,13 @@ Convenience function for reading and processing a Docx file, splitting its conte
253
299
 
254
300
  **Signature**:
255
301
 
256
- convert_docx(file_path: Path, metadata: dict) -> Generator[str, None, None]
302
+ convert_docx(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
257
303
 
258
304
  **Arguments**:
259
305
 
260
306
  - `file_path`: Path to the Docx file to be processed.
261
307
  - `metadata`: Dictionary containing `title` and `author`.
308
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
262
309
 
263
310
  **Yields**:
264
311
 
@@ -292,7 +339,6 @@ Contributions to this project are welcome. Please use Ruff for formatting to ens
292
339
  - Tests for text converter
293
340
  - More edge cases and failure states
294
341
  - Better handling of ebooklib dependency
295
- - Add additional AI models for OCR as plugins
296
342
  - Explore additional filetypes
297
343
  - Other options for determining filetype
298
344
 
@@ -1,20 +1,20 @@
1
-
2
1
  # Ebook2Text
3
2
 
4
3
  ## Overview
5
4
 
6
- This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using GPT-4o and standardizes the text by converting smart punctuation.
5
+ This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using a vision-capable AI model and standardizes the text by converting smart punctuation.
7
6
 
8
7
  ## Features
9
8
 
10
9
  - **File Format Support**: Handles EPUB, DOCX, PDF, and TXT formats.
11
10
  - **Chapter Identification**: Detects and marks chapter breaks.
12
11
  - **OCR Capability**: Converts text from images using OCR.
12
+ - **Multiple AI Providers**: OpenAI, Google Gemini, Anthropic, and OpenRouter.
13
13
  - **Text Standardization**: Replaces smart punctuation with ASCII equivalents.
14
14
 
15
15
  ## Requirements
16
16
 
17
- To run this script, you need Python 3.9 or above and the following packages:
17
+ To run this script, you need Python 3.10 or above and the following packages:
18
18
 
19
19
  - `bs4`
20
20
  - `ebooklib-autoupdate`
@@ -24,11 +24,45 @@ To run this script, you need Python 3.9 or above and the following packages:
24
24
  - `python-dotenv`
25
25
  - `openai`
26
26
 
27
+ The OpenAI SDK is a core dependency and also serves OpenRouter. Gemini and Anthropic need optional extras:
28
+
29
+ ```bash
30
+ pip install ebook2text[gemini] # Google Gemini
31
+ pip install ebook2text[anthropic] # Anthropic Claude
32
+ pip install ebook2text[all] # both
33
+ ```
34
+
35
+ ## AI provider configuration
36
+
37
+ OCR runs through one of four providers, selected by environment variable. See `.env.sample` for a complete template.
38
+
39
+ | Variable | Purpose | Default |
40
+ | -------------------------------------------------------------------------------- | ------------------------------------------------ | ---------------------------- |
41
+ | `OCR_PROVIDER` | `openai`, `gemini`, `anthropic`, or `openrouter` | `openai` |
42
+ | `OCR_MODEL` | Model name for the chosen provider | falls back to `OPENAI_MODEL` |
43
+ | `OCR_MAX_TOKENS` | Maximum tokens in the OCR response | `1000` |
44
+ | `OPENAI_API_KEY` / `GEMINI_API_KEY` / `ANTHROPIC_API_KEY` / `OPENROUTER_API_KEY` | API key for the selected provider | — |
45
+
46
+ API keys and models are read the first time an image is actually processed, so importing the library and converting text-only books requires no configuration at all.
47
+
48
+ To choose a provider programmatically instead, pass one to `convert_file`:
49
+
50
+ ```python
51
+ from ebook2text import convert_file, get_ocr_provider
52
+
53
+ provider = get_ocr_provider(provider="anthropic", model="claude-haiku-4-5")
54
+ text = convert_file(
55
+ file_path, metadata, save_file=False, ocr_provider=provider
56
+ )
57
+ ```
58
+
59
+ Any object with a `perform_ocr(base64_images: list[str]) -> str` method satisfies the `OCRProvider` protocol, so you can supply your own implementation.
60
+
27
61
  ## Usage
28
62
 
29
63
  1. Ensure all dependencies are installed.
30
- 2. Set your environment variable for the OpenAI API key.
31
- 3. Run `convert_file` from the `convert_file` module with the path to the ebook file and a metadata dictionary with keys of 'title' and 'author' as arguments.
64
+ 1. Set your environment variables for the AI provider (see above).
65
+ 1. Run `convert_file` from the `convert_file` module with the path to the ebook file and a metadata dictionary with keys of 'title' and 'author' as arguments.
32
66
 
33
67
  - set `save_file` to False, if you want a string returned.
34
68
  - set `save_file` to True or leave blank, and provide a Path object to `save_path` to use a custom output filename.
@@ -61,7 +95,7 @@ Converts an ebook file to a standardized text format.
61
95
  `ebook2text.convert_file.py`
62
96
 
63
97
  **Signature**:
64
- `convert_file(file_path: Path, metadata: dict, *, save_file: bool = True, save_path: Optional[Path] = None) -> Union[str, None]`
98
+ `convert_file(file_path: Path, metadata: dict, *, save_file: bool = True, save_path: Path | None = None, ocr_provider: OCRProvider | None = None) -> str | None`
65
99
 
66
100
  **Arguments**:
67
101
 
@@ -69,7 +103,10 @@ Converts an ebook file to a standardized text format.
69
103
  - `metadata`: Dictionary containing the book's `title` and `author`.
70
104
  - `save_file`: Boolean flag. If `True`, saves the converted text to a file; otherwise, returns it as a string. Defaults to `True`.
71
105
  - `save_path`: Optional path to save the output file. Defaults to a generated name in the input file's directory.
106
+ - `ocr_provider`: Optional OCR provider for image text extraction. Defaults to the provider configured by environment variables.
107
+
72
108
  **Returns**:
109
+
73
110
  - If `save_file` is `True`: Returns `None`.
74
111
  - If `save_file` is `False`: Returns the converted text as a string.
75
112
 
@@ -85,12 +122,13 @@ Initializes a PDFConverter instance for handling PDF files.
85
122
  `ebook2_text.pdf_converter`
86
123
 
87
124
  **Signature**:
88
- `initialize_pdf_converter(file_path: Path, metadata: dict) -> PDFConverter`
125
+ `initialize_pdf_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> PDFConverter`
89
126
 
90
127
  **Arguments**:
91
128
 
92
129
  - `file_path`: Path to the PDF file to be processed.
93
130
  - `metadata`: Dictionary containing `title` and `author`.
131
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
94
132
 
95
133
  **Returns**:
96
134
 
@@ -105,12 +143,13 @@ Convenience function for reading and processing a PDF file, splitting its conten
105
143
 
106
144
  **Signature**:
107
145
 
108
- convert_pdf(file_path: Path, metadata: dict) -> Generator[str, None, None]
146
+ convert_pdf(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
109
147
 
110
148
  **Arguments**:
111
149
 
112
150
  - `file_path`: Path to the PDF file to be processed.
113
151
  - `metadata`: Dictionary containing `title` and `author`.
152
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
114
153
 
115
154
  **Yields**:
116
155
 
@@ -142,12 +181,13 @@ Initializes a EpubConverter instance for handling Epub files.
142
181
  `ebook2_text.epub_converter`
143
182
 
144
183
  **Signature**:
145
- `initialize_epub_converter(file_path: Path, metadata: dict) -> EpubConverter`
184
+ `initialize_epub_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> EpubConverter`
146
185
 
147
186
  **Arguments**:
148
187
 
149
188
  - `file_path`: Path to the Epub file to be processed.
150
189
  - `metadata`: Dictionary containing `title` and `author`.
190
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
151
191
 
152
192
  **Returns**:
153
193
 
@@ -162,12 +202,13 @@ Convenience function for reading and processing a Epub file, splitting its conte
162
202
 
163
203
  **Signature**:
164
204
 
165
- convert_epub(file_path: Path, metadata: dict) -> Generator[str, None, None]
205
+ convert_epub(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
166
206
 
167
207
  **Arguments**:
168
208
 
169
209
  - `file_path`: Path to the Epub file to be processed.
170
210
  - `metadata`: Dictionary containing `title` and `author`.
211
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
171
212
 
172
213
  **Yields**:
173
214
 
@@ -199,12 +240,13 @@ Initializes a DocxConverter instance for handling Docx files.
199
240
  `ebook2_text.docx_converter`
200
241
 
201
242
  **Signature**:
202
- `initialize_docx_converter(file_path: Path, metadata: dict) -> DocxConverter`
243
+ `initialize_docx_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> DocxConverter`
203
244
 
204
245
  **Arguments**:
205
246
 
206
247
  - `file_path`: Path to the Docx file to be processed.
207
248
  - `metadata`: Dictionary containing `title` and `author`.
249
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
208
250
 
209
251
  **Returns**:
210
252
 
@@ -219,12 +261,13 @@ Convenience function for reading and processing a Docx file, splitting its conte
219
261
 
220
262
  **Signature**:
221
263
 
222
- convert_docx(file_path: Path, metadata: dict) -> Generator[str, None, None]
264
+ convert_docx(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
223
265
 
224
266
  **Arguments**:
225
267
 
226
268
  - `file_path`: Path to the Docx file to be processed.
227
269
  - `metadata`: Dictionary containing `title` and `author`.
270
+ - `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
228
271
 
229
272
  **Yields**:
230
273
 
@@ -258,7 +301,6 @@ Contributions to this project are welcome. Please use Ruff for formatting to ens
258
301
  - Tests for text converter
259
302
  - More edge cases and failure states
260
303
  - Better handling of ebooklib dependency
261
- - Add additional AI models for OCR as plugins
262
304
  - Explore additional filetypes
263
305
  - Other options for determining filetype
264
306
 
@@ -0,0 +1 @@
1
+ __version__ = "2.3.2"
@@ -1,10 +1,13 @@
1
1
  from ._logger import logger, set_logger
2
+ from .ai_providers import OCRProvider, get_ocr_provider
2
3
  from .convert_file import convert_file
3
4
  from .VERSION import __version__
4
5
 
5
6
  __version__ = __version__
6
7
  __all__ = [
8
+ "OCRProvider",
7
9
  "convert_file",
10
+ "get_ocr_provider",
8
11
  "logger",
9
12
  "set_logger",
10
13
  ]
@@ -46,3 +46,27 @@ class DocxConversionError(EbookConversionError):
46
46
 
47
47
  class TextConversionError(EbookConversionError):
48
48
  pass
49
+
50
+
51
+ class OCRProviderError(EbookConversionError):
52
+ """Base class for errors raised by AI OCR providers."""
53
+
54
+ pass
55
+
56
+
57
+ class UnsupportedProviderError(OCRProviderError):
58
+ """Raised when an unknown OCR provider name is requested."""
59
+
60
+ pass
61
+
62
+
63
+ class MissingDependencyError(OCRProviderError):
64
+ """Raised when a provider's SDK is not installed."""
65
+
66
+ pass
67
+
68
+
69
+ class MissingConfigurationError(OCRProviderError):
70
+ """Raised when required provider configuration is absent."""
71
+
72
+ pass
@@ -0,0 +1,15 @@
1
+ """AI provider implementations for OCR of ebook images."""
2
+
3
+ from ebook2text.ai_providers._base import (
4
+ BaseOCRProvider,
5
+ OCRProvider,
6
+ SourceImage,
7
+ )
8
+ from ebook2text.ai_providers._factory import get_ocr_provider
9
+
10
+ __all__ = [
11
+ "BaseOCRProvider",
12
+ "OCRProvider",
13
+ "SourceImage",
14
+ "get_ocr_provider",
15
+ ]
@@ -0,0 +1,50 @@
1
+ """Anthropic Claude OCR provider."""
2
+
3
+ from functools import cached_property
4
+ from typing import Any
5
+
6
+ from ebook2text.ai_providers._base import (
7
+ OCR_PROMPT,
8
+ BaseOCRProvider,
9
+ SourceImage,
10
+ import_provider_sdk,
11
+ )
12
+
13
+
14
+ class AnthropicOCR(BaseOCRProvider):
15
+ """OCR provider backed by the Anthropic Messages API."""
16
+
17
+ api_key_env = "ANTHROPIC_API_KEY"
18
+
19
+ @cached_property
20
+ def _client(self) -> Any:
21
+ """Lazily construct the Anthropic client."""
22
+ anthropic = import_provider_sdk("anthropic", "anthropic")
23
+ return anthropic.Anthropic(api_key=self._resolve_api_key())
24
+
25
+ def _request(self, images: list[SourceImage]) -> str:
26
+ """Send one messages request with image blocks and the OCR prompt."""
27
+ content = [
28
+ *(
29
+ {
30
+ "type": "image",
31
+ "source": {
32
+ "type": "base64",
33
+ "media_type": image.media_type,
34
+ "data": image.base64_data,
35
+ },
36
+ }
37
+ for image in images
38
+ ),
39
+ {"type": "text", "text": OCR_PROMPT},
40
+ ]
41
+ response = self._client.messages.create(
42
+ model=self.model,
43
+ max_tokens=self.max_tokens,
44
+ messages=[{"role": "user", "content": content}],
45
+ )
46
+ if response.stop_reason == "refusal":
47
+ return ""
48
+ return "".join(
49
+ block.text for block in response.content if block.type == "text"
50
+ )