ebook2text 2.2.0__tar.gz → 2.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ebook2text-2.2.0 → ebook2text-2.3.2}/PKG-INFO +67 -21
- {ebook2text-2.2.0 → ebook2text-2.3.2}/README.md +55 -13
- ebook2text-2.3.2/ebook2text/VERSION.py +1 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/__init__.py +3 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/_exceptions.py +24 -0
- ebook2text-2.3.2/ebook2text/ai_providers/__init__.py +15 -0
- ebook2text-2.3.2/ebook2text/ai_providers/_anthropic.py +50 -0
- ebook2text-2.3.2/ebook2text/ai_providers/_base.py +168 -0
- ebook2text-2.3.2/ebook2text/ai_providers/_factory.py +70 -0
- ebook2text-2.3.2/ebook2text/ai_providers/_gemini.py +47 -0
- ebook2text-2.3.2/ebook2text/ai_providers/_openai.py +55 -0
- ebook2text-2.3.2/ebook2text/ai_providers/_openrouter.py +19 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/convert_file.py +14 -5
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/__init__.py +15 -4
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/docx_text_extractor.py +15 -3
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/epub_conversion/__init__.py +13 -4
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/epub_conversion/epub_text_extractor.py +22 -3
- ebook2text-2.3.2/ebook2text/ocr.py +33 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/__init__.py +21 -4
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/pdf_text_extractor.py +15 -3
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text.egg-info/PKG-INFO +67 -21
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text.egg-info/SOURCES.txt +8 -0
- ebook2text-2.3.2/ebook2text.egg-info/requires.txt +16 -0
- ebook2text-2.3.2/pyproject.toml +88 -0
- ebook2text-2.3.2/tests/test_convert_file.py +25 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_docx_conversion.py +8 -4
- {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_epub_conversion.py +2 -7
- ebook2text-2.3.2/tests/test_ocr.py +55 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_pdf_conversion.py +6 -12
- ebook2text-2.2.0/ebook2text/VERSION.py +0 -1
- ebook2text-2.2.0/ebook2text/ocr.py +0 -117
- ebook2text-2.2.0/ebook2text.egg-info/requires.txt +0 -7
- ebook2text-2.2.0/pyproject.toml +0 -75
- ebook2text-2.2.0/tests/test_ocr.py +0 -137
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/_logger.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/_types.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/chapter_check.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/_namespaces.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/docx_converter.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/docx_conversion/docx_image_extractor.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/epub_conversion/epub_converter.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/_enums.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/pdf_converter.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/pdf_image_extractor.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/pdf_conversion/pdf_line_logic.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/text_parser.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text/text_utilities.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text.egg-info/dependency_links.txt +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/ebook2text.egg-info/top_level.txt +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/setup.cfg +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/setup.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_chapter_check.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_pdf_image_helpers.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_text_conversion.py +0 -0
- {ebook2text-2.2.0 → ebook2text-2.3.2}/tests/test_text_parser.py +0 -0
|
@@ -1,54 +1,58 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: ebook2text
|
|
3
|
-
Version: 2.2
|
|
3
|
+
Version: 2.3.2
|
|
4
4
|
Summary: Convert common book file types to text for machine learning
|
|
5
5
|
Author: Ashlynn Antrobus
|
|
6
6
|
Author-email: Ashlynn Antrobus <ashlynn@prosepal.io>
|
|
7
|
-
|
|
8
|
-
Project-URL: Repository, https://github.com/ashrobertsdragon/Ebook-conversion-to-Text-for-Machine-Learning
|
|
9
|
-
Classifier: License :: OSI Approved :: MIT License
|
|
7
|
+
Project-URL: Repository, https://github.com/ashrobertsdragon/ebook2text
|
|
10
8
|
Classifier: Development Status :: 5 - Production/Stable
|
|
11
|
-
Classifier: Intended Audience :: Developers
|
|
12
9
|
Classifier: Environment :: Console
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
13
11
|
Classifier: Operating System :: OS Independent
|
|
14
12
|
Classifier: Programming Language :: Python
|
|
15
13
|
Classifier: Programming Language :: Python :: 3
|
|
16
14
|
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
-
Classifier: Programming Language :: Python :: 3.9
|
|
18
15
|
Classifier: Programming Language :: Python :: 3.10
|
|
19
16
|
Classifier: Programming Language :: Python :: 3.11
|
|
20
17
|
Classifier: Programming Language :: Python :: 3.12
|
|
21
18
|
Classifier: Programming Language :: Python :: 3.13
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
22
20
|
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
21
|
Classifier: Topic :: Text Processing
|
|
24
22
|
Requires-Python: >=3.10
|
|
25
23
|
Description-Content-Type: text/markdown
|
|
26
24
|
Requires-Dist: beautifulsoup4>=4.12.3
|
|
27
|
-
Requires-Dist: ebooklib
|
|
25
|
+
Requires-Dist: ebooklib>=0.20
|
|
28
26
|
Requires-Dist: openai>=1.54.3
|
|
29
27
|
Requires-Dist: pdfminer-six>=20240706
|
|
30
|
-
Requires-Dist: pillow>=
|
|
28
|
+
Requires-Dist: pillow>=12.3.0
|
|
31
29
|
Requires-Dist: python-docx>=1.1.2
|
|
32
30
|
Requires-Dist: python-dotenv>=1.0.1
|
|
31
|
+
Provides-Extra: all
|
|
32
|
+
Requires-Dist: ebook2text[anthropic,gemini]; extra == "all"
|
|
33
|
+
Provides-Extra: anthropic
|
|
34
|
+
Requires-Dist: anthropic>=0.40.0; extra == "anthropic"
|
|
35
|
+
Provides-Extra: gemini
|
|
36
|
+
Requires-Dist: google-genai>=1.0.0; extra == "gemini"
|
|
33
37
|
Dynamic: author
|
|
34
38
|
|
|
35
|
-
|
|
36
39
|
# Ebook2Text
|
|
37
40
|
|
|
38
41
|
## Overview
|
|
39
42
|
|
|
40
|
-
This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using
|
|
43
|
+
This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using a vision-capable AI model and standardizes the text by converting smart punctuation.
|
|
41
44
|
|
|
42
45
|
## Features
|
|
43
46
|
|
|
44
47
|
- **File Format Support**: Handles EPUB, DOCX, PDF, and TXT formats.
|
|
45
48
|
- **Chapter Identification**: Detects and marks chapter breaks.
|
|
46
49
|
- **OCR Capability**: Converts text from images using OCR.
|
|
50
|
+
- **Multiple AI Providers**: OpenAI, Google Gemini, Anthropic, and OpenRouter.
|
|
47
51
|
- **Text Standardization**: Replaces smart punctuation with ASCII equivalents.
|
|
48
52
|
|
|
49
53
|
## Requirements
|
|
50
54
|
|
|
51
|
-
To run this script, you need Python 3.
|
|
55
|
+
To run this script, you need Python 3.10 or above and the following packages:
|
|
52
56
|
|
|
53
57
|
- `bs4`
|
|
54
58
|
- `ebooklib-autoupdate`
|
|
@@ -58,11 +62,45 @@ To run this script, you need Python 3.9 or above and the following packages:
|
|
|
58
62
|
- `python-dotenv`
|
|
59
63
|
- `openai`
|
|
60
64
|
|
|
65
|
+
The OpenAI SDK is a core dependency and also serves OpenRouter. Gemini and Anthropic need optional extras:
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
pip install ebook2text[gemini] # Google Gemini
|
|
69
|
+
pip install ebook2text[anthropic] # Anthropic Claude
|
|
70
|
+
pip install ebook2text[all] # both
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## AI provider configuration
|
|
74
|
+
|
|
75
|
+
OCR runs through one of four providers, selected by environment variable. See `.env.sample` for a complete template.
|
|
76
|
+
|
|
77
|
+
| Variable | Purpose | Default |
|
|
78
|
+
| -------------------------------------------------------------------------------- | ------------------------------------------------ | ---------------------------- |
|
|
79
|
+
| `OCR_PROVIDER` | `openai`, `gemini`, `anthropic`, or `openrouter` | `openai` |
|
|
80
|
+
| `OCR_MODEL` | Model name for the chosen provider | falls back to `OPENAI_MODEL` |
|
|
81
|
+
| `OCR_MAX_TOKENS` | Maximum tokens in the OCR response | `1000` |
|
|
82
|
+
| `OPENAI_API_KEY` / `GEMINI_API_KEY` / `ANTHROPIC_API_KEY` / `OPENROUTER_API_KEY` | API key for the selected provider | — |
|
|
83
|
+
|
|
84
|
+
API keys and models are read the first time an image is actually processed, so importing the library and converting text-only books requires no configuration at all.
|
|
85
|
+
|
|
86
|
+
To choose a provider programmatically instead, pass one to `convert_file`:
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
from ebook2text import convert_file, get_ocr_provider
|
|
90
|
+
|
|
91
|
+
provider = get_ocr_provider(provider="anthropic", model="claude-haiku-4-5")
|
|
92
|
+
text = convert_file(
|
|
93
|
+
file_path, metadata, save_file=False, ocr_provider=provider
|
|
94
|
+
)
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Any object with a `perform_ocr(base64_images: list[str]) -> str` method satisfies the `OCRProvider` protocol, so you can supply your own implementation.
|
|
98
|
+
|
|
61
99
|
## Usage
|
|
62
100
|
|
|
63
101
|
1. Ensure all dependencies are installed.
|
|
64
|
-
|
|
65
|
-
|
|
102
|
+
1. Set your environment variables for the AI provider (see above).
|
|
103
|
+
1. Run `convert_file` from the `convert_file` module with the path to the ebook file and a metadata dictionary with keys of 'title' and 'author' as arguments.
|
|
66
104
|
|
|
67
105
|
- set `save_file` to False, if you want a string returned.
|
|
68
106
|
- set `save_file` to True or leave blank, and provide a Path object to `save_path` to use a custom output filename.
|
|
@@ -95,7 +133,7 @@ Converts an ebook file to a standardized text format.
|
|
|
95
133
|
`ebook2text.convert_file.py`
|
|
96
134
|
|
|
97
135
|
**Signature**:
|
|
98
|
-
`convert_file(file_path: Path, metadata: dict, *, save_file: bool = True, save_path:
|
|
136
|
+
`convert_file(file_path: Path, metadata: dict, *, save_file: bool = True, save_path: Path | None = None, ocr_provider: OCRProvider | None = None) -> str | None`
|
|
99
137
|
|
|
100
138
|
**Arguments**:
|
|
101
139
|
|
|
@@ -103,7 +141,10 @@ Converts an ebook file to a standardized text format.
|
|
|
103
141
|
- `metadata`: Dictionary containing the book's `title` and `author`.
|
|
104
142
|
- `save_file`: Boolean flag. If `True`, saves the converted text to a file; otherwise, returns it as a string. Defaults to `True`.
|
|
105
143
|
- `save_path`: Optional path to save the output file. Defaults to a generated name in the input file's directory.
|
|
144
|
+
- `ocr_provider`: Optional OCR provider for image text extraction. Defaults to the provider configured by environment variables.
|
|
145
|
+
|
|
106
146
|
**Returns**:
|
|
147
|
+
|
|
107
148
|
- If `save_file` is `True`: Returns `None`.
|
|
108
149
|
- If `save_file` is `False`: Returns the converted text as a string.
|
|
109
150
|
|
|
@@ -119,12 +160,13 @@ Initializes a PDFConverter instance for handling PDF files.
|
|
|
119
160
|
`ebook2_text.pdf_converter`
|
|
120
161
|
|
|
121
162
|
**Signature**:
|
|
122
|
-
`initialize_pdf_converter(file_path: Path, metadata: dict) -> PDFConverter`
|
|
163
|
+
`initialize_pdf_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> PDFConverter`
|
|
123
164
|
|
|
124
165
|
**Arguments**:
|
|
125
166
|
|
|
126
167
|
- `file_path`: Path to the PDF file to be processed.
|
|
127
168
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
169
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
128
170
|
|
|
129
171
|
**Returns**:
|
|
130
172
|
|
|
@@ -139,12 +181,13 @@ Convenience function for reading and processing a PDF file, splitting its conten
|
|
|
139
181
|
|
|
140
182
|
**Signature**:
|
|
141
183
|
|
|
142
|
-
convert_pdf(file_path: Path, metadata: dict) -> Generator[str, None, None]
|
|
184
|
+
convert_pdf(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
|
|
143
185
|
|
|
144
186
|
**Arguments**:
|
|
145
187
|
|
|
146
188
|
- `file_path`: Path to the PDF file to be processed.
|
|
147
189
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
190
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
148
191
|
|
|
149
192
|
**Yields**:
|
|
150
193
|
|
|
@@ -176,12 +219,13 @@ Initializes a EpubConverter instance for handling Epub files.
|
|
|
176
219
|
`ebook2_text.epub_converter`
|
|
177
220
|
|
|
178
221
|
**Signature**:
|
|
179
|
-
`initialize_epub_converter(file_path: Path, metadata: dict) -> EpubConverter`
|
|
222
|
+
`initialize_epub_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> EpubConverter`
|
|
180
223
|
|
|
181
224
|
**Arguments**:
|
|
182
225
|
|
|
183
226
|
- `file_path`: Path to the Epub file to be processed.
|
|
184
227
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
228
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
185
229
|
|
|
186
230
|
**Returns**:
|
|
187
231
|
|
|
@@ -196,12 +240,13 @@ Convenience function for reading and processing a Epub file, splitting its conte
|
|
|
196
240
|
|
|
197
241
|
**Signature**:
|
|
198
242
|
|
|
199
|
-
convert_epub(file_path: Path, metadata: dict) -> Generator[str, None, None]
|
|
243
|
+
convert_epub(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
|
|
200
244
|
|
|
201
245
|
**Arguments**:
|
|
202
246
|
|
|
203
247
|
- `file_path`: Path to the Epub file to be processed.
|
|
204
248
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
249
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
205
250
|
|
|
206
251
|
**Yields**:
|
|
207
252
|
|
|
@@ -233,12 +278,13 @@ Initializes a DocxConverter instance for handling Docx files.
|
|
|
233
278
|
`ebook2_text.docx_converter`
|
|
234
279
|
|
|
235
280
|
**Signature**:
|
|
236
|
-
`initialize_docx_converter(file_path: Path, metadata: dict) -> DocxConverter`
|
|
281
|
+
`initialize_docx_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> DocxConverter`
|
|
237
282
|
|
|
238
283
|
**Arguments**:
|
|
239
284
|
|
|
240
285
|
- `file_path`: Path to the Docx file to be processed.
|
|
241
286
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
287
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
242
288
|
|
|
243
289
|
**Returns**:
|
|
244
290
|
|
|
@@ -253,12 +299,13 @@ Convenience function for reading and processing a Docx file, splitting its conte
|
|
|
253
299
|
|
|
254
300
|
**Signature**:
|
|
255
301
|
|
|
256
|
-
convert_docx(file_path: Path, metadata: dict) -> Generator[str, None, None]
|
|
302
|
+
convert_docx(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
|
|
257
303
|
|
|
258
304
|
**Arguments**:
|
|
259
305
|
|
|
260
306
|
- `file_path`: Path to the Docx file to be processed.
|
|
261
307
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
308
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
262
309
|
|
|
263
310
|
**Yields**:
|
|
264
311
|
|
|
@@ -292,7 +339,6 @@ Contributions to this project are welcome. Please use Ruff for formatting to ens
|
|
|
292
339
|
- Tests for text converter
|
|
293
340
|
- More edge cases and failure states
|
|
294
341
|
- Better handling of ebooklib dependency
|
|
295
|
-
- Add additional AI models for OCR as plugins
|
|
296
342
|
- Explore additional filetypes
|
|
297
343
|
- Other options for determining filetype
|
|
298
344
|
|
|
@@ -1,20 +1,20 @@
|
|
|
1
|
-
|
|
2
1
|
# Ebook2Text
|
|
3
2
|
|
|
4
3
|
## Overview
|
|
5
4
|
|
|
6
|
-
This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using
|
|
5
|
+
This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using a vision-capable AI model and standardizes the text by converting smart punctuation.
|
|
7
6
|
|
|
8
7
|
## Features
|
|
9
8
|
|
|
10
9
|
- **File Format Support**: Handles EPUB, DOCX, PDF, and TXT formats.
|
|
11
10
|
- **Chapter Identification**: Detects and marks chapter breaks.
|
|
12
11
|
- **OCR Capability**: Converts text from images using OCR.
|
|
12
|
+
- **Multiple AI Providers**: OpenAI, Google Gemini, Anthropic, and OpenRouter.
|
|
13
13
|
- **Text Standardization**: Replaces smart punctuation with ASCII equivalents.
|
|
14
14
|
|
|
15
15
|
## Requirements
|
|
16
16
|
|
|
17
|
-
To run this script, you need Python 3.
|
|
17
|
+
To run this script, you need Python 3.10 or above and the following packages:
|
|
18
18
|
|
|
19
19
|
- `bs4`
|
|
20
20
|
- `ebooklib-autoupdate`
|
|
@@ -24,11 +24,45 @@ To run this script, you need Python 3.9 or above and the following packages:
|
|
|
24
24
|
- `python-dotenv`
|
|
25
25
|
- `openai`
|
|
26
26
|
|
|
27
|
+
The OpenAI SDK is a core dependency and also serves OpenRouter. Gemini and Anthropic need optional extras:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pip install ebook2text[gemini] # Google Gemini
|
|
31
|
+
pip install ebook2text[anthropic] # Anthropic Claude
|
|
32
|
+
pip install ebook2text[all] # both
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## AI provider configuration
|
|
36
|
+
|
|
37
|
+
OCR runs through one of four providers, selected by environment variable. See `.env.sample` for a complete template.
|
|
38
|
+
|
|
39
|
+
| Variable | Purpose | Default |
|
|
40
|
+
| -------------------------------------------------------------------------------- | ------------------------------------------------ | ---------------------------- |
|
|
41
|
+
| `OCR_PROVIDER` | `openai`, `gemini`, `anthropic`, or `openrouter` | `openai` |
|
|
42
|
+
| `OCR_MODEL` | Model name for the chosen provider | falls back to `OPENAI_MODEL` |
|
|
43
|
+
| `OCR_MAX_TOKENS` | Maximum tokens in the OCR response | `1000` |
|
|
44
|
+
| `OPENAI_API_KEY` / `GEMINI_API_KEY` / `ANTHROPIC_API_KEY` / `OPENROUTER_API_KEY` | API key for the selected provider | — |
|
|
45
|
+
|
|
46
|
+
API keys and models are read the first time an image is actually processed, so importing the library and converting text-only books requires no configuration at all.
|
|
47
|
+
|
|
48
|
+
To choose a provider programmatically instead, pass one to `convert_file`:
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from ebook2text import convert_file, get_ocr_provider
|
|
52
|
+
|
|
53
|
+
provider = get_ocr_provider(provider="anthropic", model="claude-haiku-4-5")
|
|
54
|
+
text = convert_file(
|
|
55
|
+
file_path, metadata, save_file=False, ocr_provider=provider
|
|
56
|
+
)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Any object with a `perform_ocr(base64_images: list[str]) -> str` method satisfies the `OCRProvider` protocol, so you can supply your own implementation.
|
|
60
|
+
|
|
27
61
|
## Usage
|
|
28
62
|
|
|
29
63
|
1. Ensure all dependencies are installed.
|
|
30
|
-
|
|
31
|
-
|
|
64
|
+
1. Set your environment variables for the AI provider (see above).
|
|
65
|
+
1. Run `convert_file` from the `convert_file` module with the path to the ebook file and a metadata dictionary with keys of 'title' and 'author' as arguments.
|
|
32
66
|
|
|
33
67
|
- set `save_file` to False, if you want a string returned.
|
|
34
68
|
- set `save_file` to True or leave blank, and provide a Path object to `save_path` to use a custom output filename.
|
|
@@ -61,7 +95,7 @@ Converts an ebook file to a standardized text format.
|
|
|
61
95
|
`ebook2text.convert_file.py`
|
|
62
96
|
|
|
63
97
|
**Signature**:
|
|
64
|
-
`convert_file(file_path: Path, metadata: dict, *, save_file: bool = True, save_path:
|
|
98
|
+
`convert_file(file_path: Path, metadata: dict, *, save_file: bool = True, save_path: Path | None = None, ocr_provider: OCRProvider | None = None) -> str | None`
|
|
65
99
|
|
|
66
100
|
**Arguments**:
|
|
67
101
|
|
|
@@ -69,7 +103,10 @@ Converts an ebook file to a standardized text format.
|
|
|
69
103
|
- `metadata`: Dictionary containing the book's `title` and `author`.
|
|
70
104
|
- `save_file`: Boolean flag. If `True`, saves the converted text to a file; otherwise, returns it as a string. Defaults to `True`.
|
|
71
105
|
- `save_path`: Optional path to save the output file. Defaults to a generated name in the input file's directory.
|
|
106
|
+
- `ocr_provider`: Optional OCR provider for image text extraction. Defaults to the provider configured by environment variables.
|
|
107
|
+
|
|
72
108
|
**Returns**:
|
|
109
|
+
|
|
73
110
|
- If `save_file` is `True`: Returns `None`.
|
|
74
111
|
- If `save_file` is `False`: Returns the converted text as a string.
|
|
75
112
|
|
|
@@ -85,12 +122,13 @@ Initializes a PDFConverter instance for handling PDF files.
|
|
|
85
122
|
`ebook2_text.pdf_converter`
|
|
86
123
|
|
|
87
124
|
**Signature**:
|
|
88
|
-
`initialize_pdf_converter(file_path: Path, metadata: dict) -> PDFConverter`
|
|
125
|
+
`initialize_pdf_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> PDFConverter`
|
|
89
126
|
|
|
90
127
|
**Arguments**:
|
|
91
128
|
|
|
92
129
|
- `file_path`: Path to the PDF file to be processed.
|
|
93
130
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
131
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
94
132
|
|
|
95
133
|
**Returns**:
|
|
96
134
|
|
|
@@ -105,12 +143,13 @@ Convenience function for reading and processing a PDF file, splitting its conten
|
|
|
105
143
|
|
|
106
144
|
**Signature**:
|
|
107
145
|
|
|
108
|
-
convert_pdf(file_path: Path, metadata: dict) -> Generator[str, None, None]
|
|
146
|
+
convert_pdf(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
|
|
109
147
|
|
|
110
148
|
**Arguments**:
|
|
111
149
|
|
|
112
150
|
- `file_path`: Path to the PDF file to be processed.
|
|
113
151
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
152
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
114
153
|
|
|
115
154
|
**Yields**:
|
|
116
155
|
|
|
@@ -142,12 +181,13 @@ Initializes a EpubConverter instance for handling Epub files.
|
|
|
142
181
|
`ebook2_text.epub_converter`
|
|
143
182
|
|
|
144
183
|
**Signature**:
|
|
145
|
-
`initialize_epub_converter(file_path: Path, metadata: dict) -> EpubConverter`
|
|
184
|
+
`initialize_epub_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> EpubConverter`
|
|
146
185
|
|
|
147
186
|
**Arguments**:
|
|
148
187
|
|
|
149
188
|
- `file_path`: Path to the Epub file to be processed.
|
|
150
189
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
190
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
151
191
|
|
|
152
192
|
**Returns**:
|
|
153
193
|
|
|
@@ -162,12 +202,13 @@ Convenience function for reading and processing a Epub file, splitting its conte
|
|
|
162
202
|
|
|
163
203
|
**Signature**:
|
|
164
204
|
|
|
165
|
-
convert_epub(file_path: Path, metadata: dict) -> Generator[str, None, None]
|
|
205
|
+
convert_epub(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
|
|
166
206
|
|
|
167
207
|
**Arguments**:
|
|
168
208
|
|
|
169
209
|
- `file_path`: Path to the Epub file to be processed.
|
|
170
210
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
211
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
171
212
|
|
|
172
213
|
**Yields**:
|
|
173
214
|
|
|
@@ -199,12 +240,13 @@ Initializes a DocxConverter instance for handling Docx files.
|
|
|
199
240
|
`ebook2_text.docx_converter`
|
|
200
241
|
|
|
201
242
|
**Signature**:
|
|
202
|
-
`initialize_docx_converter(file_path: Path, metadata: dict) -> DocxConverter`
|
|
243
|
+
`initialize_docx_converter(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> DocxConverter`
|
|
203
244
|
|
|
204
245
|
**Arguments**:
|
|
205
246
|
|
|
206
247
|
- `file_path`: Path to the Docx file to be processed.
|
|
207
248
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
249
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
208
250
|
|
|
209
251
|
**Returns**:
|
|
210
252
|
|
|
@@ -219,12 +261,13 @@ Convenience function for reading and processing a Docx file, splitting its conte
|
|
|
219
261
|
|
|
220
262
|
**Signature**:
|
|
221
263
|
|
|
222
|
-
convert_docx(file_path: Path, metadata: dict) -> Generator[str, None, None]
|
|
264
|
+
convert_docx(file_path: Path, metadata: dict, ocr_provider: OCRProvider | None = None) -> Generator[str, None, None]
|
|
223
265
|
|
|
224
266
|
**Arguments**:
|
|
225
267
|
|
|
226
268
|
- `file_path`: Path to the Docx file to be processed.
|
|
227
269
|
- `metadata`: Dictionary containing `title` and `author`.
|
|
270
|
+
- `ocr_provider`: Optional OCR provider. Defaults to the environment-configured provider.
|
|
228
271
|
|
|
229
272
|
**Yields**:
|
|
230
273
|
|
|
@@ -258,7 +301,6 @@ Contributions to this project are welcome. Please use Ruff for formatting to ens
|
|
|
258
301
|
- Tests for text converter
|
|
259
302
|
- More edge cases and failure states
|
|
260
303
|
- Better handling of ebooklib dependency
|
|
261
|
-
- Add additional AI models for OCR as plugins
|
|
262
304
|
- Explore additional filetypes
|
|
263
305
|
- Other options for determining filetype
|
|
264
306
|
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "2.3.2"
|
|
@@ -1,10 +1,13 @@
|
|
|
1
1
|
from ._logger import logger, set_logger
|
|
2
|
+
from .ai_providers import OCRProvider, get_ocr_provider
|
|
2
3
|
from .convert_file import convert_file
|
|
3
4
|
from .VERSION import __version__
|
|
4
5
|
|
|
5
6
|
__version__ = __version__
|
|
6
7
|
__all__ = [
|
|
8
|
+
"OCRProvider",
|
|
7
9
|
"convert_file",
|
|
10
|
+
"get_ocr_provider",
|
|
8
11
|
"logger",
|
|
9
12
|
"set_logger",
|
|
10
13
|
]
|
|
@@ -46,3 +46,27 @@ class DocxConversionError(EbookConversionError):
|
|
|
46
46
|
|
|
47
47
|
class TextConversionError(EbookConversionError):
|
|
48
48
|
pass
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class OCRProviderError(EbookConversionError):
|
|
52
|
+
"""Base class for errors raised by AI OCR providers."""
|
|
53
|
+
|
|
54
|
+
pass
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class UnsupportedProviderError(OCRProviderError):
|
|
58
|
+
"""Raised when an unknown OCR provider name is requested."""
|
|
59
|
+
|
|
60
|
+
pass
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class MissingDependencyError(OCRProviderError):
|
|
64
|
+
"""Raised when a provider's SDK is not installed."""
|
|
65
|
+
|
|
66
|
+
pass
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class MissingConfigurationError(OCRProviderError):
|
|
70
|
+
"""Raised when required provider configuration is absent."""
|
|
71
|
+
|
|
72
|
+
pass
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""AI provider implementations for OCR of ebook images."""
|
|
2
|
+
|
|
3
|
+
from ebook2text.ai_providers._base import (
|
|
4
|
+
BaseOCRProvider,
|
|
5
|
+
OCRProvider,
|
|
6
|
+
SourceImage,
|
|
7
|
+
)
|
|
8
|
+
from ebook2text.ai_providers._factory import get_ocr_provider
|
|
9
|
+
|
|
10
|
+
__all__ = [
|
|
11
|
+
"BaseOCRProvider",
|
|
12
|
+
"OCRProvider",
|
|
13
|
+
"SourceImage",
|
|
14
|
+
"get_ocr_provider",
|
|
15
|
+
]
|
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Anthropic Claude OCR provider."""
|
|
2
|
+
|
|
3
|
+
from functools import cached_property
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from ebook2text.ai_providers._base import (
|
|
7
|
+
OCR_PROMPT,
|
|
8
|
+
BaseOCRProvider,
|
|
9
|
+
SourceImage,
|
|
10
|
+
import_provider_sdk,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class AnthropicOCR(BaseOCRProvider):
|
|
15
|
+
"""OCR provider backed by the Anthropic Messages API."""
|
|
16
|
+
|
|
17
|
+
api_key_env = "ANTHROPIC_API_KEY"
|
|
18
|
+
|
|
19
|
+
@cached_property
|
|
20
|
+
def _client(self) -> Any:
|
|
21
|
+
"""Lazily construct the Anthropic client."""
|
|
22
|
+
anthropic = import_provider_sdk("anthropic", "anthropic")
|
|
23
|
+
return anthropic.Anthropic(api_key=self._resolve_api_key())
|
|
24
|
+
|
|
25
|
+
def _request(self, images: list[SourceImage]) -> str:
|
|
26
|
+
"""Send one messages request with image blocks and the OCR prompt."""
|
|
27
|
+
content = [
|
|
28
|
+
*(
|
|
29
|
+
{
|
|
30
|
+
"type": "image",
|
|
31
|
+
"source": {
|
|
32
|
+
"type": "base64",
|
|
33
|
+
"media_type": image.media_type,
|
|
34
|
+
"data": image.base64_data,
|
|
35
|
+
},
|
|
36
|
+
}
|
|
37
|
+
for image in images
|
|
38
|
+
),
|
|
39
|
+
{"type": "text", "text": OCR_PROMPT},
|
|
40
|
+
]
|
|
41
|
+
response = self._client.messages.create(
|
|
42
|
+
model=self.model,
|
|
43
|
+
max_tokens=self.max_tokens,
|
|
44
|
+
messages=[{"role": "user", "content": content}],
|
|
45
|
+
)
|
|
46
|
+
if response.stop_reason == "refusal":
|
|
47
|
+
return ""
|
|
48
|
+
return "".join(
|
|
49
|
+
block.text for block in response.content if block.type == "text"
|
|
50
|
+
)
|