ebook2text 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,129 @@
1
+ Metadata-Version: 2.1
2
+ Name: ebook2text
3
+ Version: 1.1.0
4
+ Summary: Convert common book file types to text for machine learning
5
+ Author: Ashlynn Antrobus
6
+ Author-email: Ashlynn Antrobus <ashlynn@prosepal.io>
7
+ License: MIT
8
+ Project-URL: Respository, https://github.com/ashrobertsdragon/Ebook-conversion-to-Text-for-Machine-Learning
9
+ Classifier: License :: OSI Approved :: MIT License
10
+ Classifier: Development Status :: 5 - Production/Stable
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: Operating System :: OS Independent
13
+ Classifier: Programming Language :: Python
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3 :: Only
16
+ Classifier: Programming Language :: Python :: 3.8
17
+ Classifier: Programming Language :: Python :: 3.9
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
22
+ Classifier: Topic :: Text Processing
23
+ Requires-Python: >=3.8
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE.txt
26
+ Requires-Dist: annotated-types>=0.4.0
27
+ Requires-Dist: anyio<5,>3.5
28
+ Requires-Dist: beautifulsoup4>=4.12.3
29
+ Requires-Dist: certifi>=2024.2.2
30
+ Requires-Dist: cffi>=1.12
31
+ Requires-Dist: colorama>=0.4.6
32
+ Requires-Dist: cryptography>=42.0.7
33
+ Requires-Dist: distro<2,>=1.7
34
+ Requires-Dist: EbookLib==0.18
35
+ Requires-Dist: h11<=0.15,>=0.13
36
+ Requires-Dist: httpcore>1
37
+ Requires-Dist: httpx<1,>=0.23.0
38
+ Requires-Dist: idna<4,>2.8.0
39
+ Requires-Dist: lxml>3.1.0
40
+ Requires-Dist: openai>=1.30.1
41
+ Requires-Dist: pillow>=10.2.0
42
+ Requires-Dist: pycparser==2.22
43
+ Requires-Dist: pydantic<3,>1.9
44
+ Requires-Dist: pydantic-core==2.18.2
45
+ Requires-Dist: python-docx>=1.1.0
46
+ Requires-Dist: python-dotenv>=1.0.1
47
+ Requires-Dist: six==1.16.0
48
+ Requires-Dist: sniffio>=1.1.0
49
+ Requires-Dist: soupsieve>1.2
50
+ Requires-Dist: tqdm>=4.0.0
51
+ Requires-Dist: typing-extensions>=4.9.0
52
+
53
+
54
+ # Convert Ebook File
55
+
56
+ ## Overview
57
+
58
+ This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using GPT-4o and standardizes the text by desmartenizing punctuation.
59
+
60
+ ## Features
61
+
62
+ - **File Format Support**: Handles EPUB, DOCX, PDF, and TXT formats.
63
+ - **Chapter Identification**: Detects and marks chapter breaks.
64
+ - **OCR Capability**: Converts text from images using OCR.
65
+ - **Text Standardization**: Replaces smart punctuation with ASCII equivalents.
66
+
67
+ ## Requirements
68
+
69
+ To run this script, you need Python 3.8 or above and the following packages:
70
+
71
+ - `python-docx`
72
+ - `ebooklib`
73
+ - `openai`
74
+ - `python-dotenv`
75
+ - `bs4`
76
+ - `ebooklib`
77
+ - `pdfminer.six`
78
+ - `pillow`
79
+
80
+ ## Usage
81
+
82
+ 1. Ensure all dependencies are installed.
83
+ 2. Set your environment variable for the OpenAI API key.
84
+ 3. Place your ebook files in a known directory.
85
+ 4. Run the script with the path to the ebook file and a metadata dictionary with keys of 'title' and 'author' as arguments.
86
+
87
+ ## Functions
88
+
89
+ - `read_text_file(file: str) -> str`: Reads a text file and returns its content.
90
+ - `write_to_file(content: str, file: str)`: Writes content to a file.
91
+ - `convert_file(file_path: str, metadata: dict) -> str`: Main function to convert an ebook file to text.
92
+
93
+ ## Contributing
94
+
95
+ Contributions to this project are welcome. Please ensure that your code follows the existing style for consistency.
96
+
97
+ ## License
98
+
99
+ This project is licensed by ProsePal LLC under the MIT license
100
+
101
+ ## Version History
102
+
103
+ - **v0.1.0** (Release date: November 30, 2023)
104
+ - Initial release
105
+
106
+ - **v0.1.1** (Release date: December 2, 2023)
107
+ - fixed false positives for is_number
108
+
109
+ - **v0.2.0** (Release date: December 3, 2023)
110
+ - Conversion of docx files
111
+
112
+ - **v0.3.0** (Release date: December 8, 2023)
113
+ - Conversion of PDF files
114
+
115
+ - **v0.3.1** (Release date: Januar 23, 2024)
116
+ - fixed concantation of text in pdf conversion
117
+ - updated pillow version to secure version
118
+
119
+ - **v1.0.0** (Release date: January 23, 2024)
120
+ - created library instead of single module
121
+
122
+ - **v1.0.1** (Release date: March 13, 2024)
123
+ - setup.py and requirements.txt typo fixed
124
+
125
+ - **v1.0.2** (Release date: May 17, 2024)
126
+ - added tests, fixex minor typos
127
+
128
+ - **v1.1.0** (Release date: May 30, 2024)
129
+ - Change to abstract factory pattern
@@ -0,0 +1,77 @@
1
+
2
+ # Convert Ebook File
3
+
4
+ ## Overview
5
+
6
+ This Python script provides functionality for converting various ebook file formats (EPUB, DOCX, PDF, TXT) into a standardized text format. The script processes each file, identifying chapters, and replaces chapter headers with asterisks. It also performs OCR (Optical Character Recognition) for image-based text using GPT-4o and standardizes the text by desmartenizing punctuation.
7
+
8
+ ## Features
9
+
10
+ - **File Format Support**: Handles EPUB, DOCX, PDF, and TXT formats.
11
+ - **Chapter Identification**: Detects and marks chapter breaks.
12
+ - **OCR Capability**: Converts text from images using OCR.
13
+ - **Text Standardization**: Replaces smart punctuation with ASCII equivalents.
14
+
15
+ ## Requirements
16
+
17
+ To run this script, you need Python 3.8 or above and the following packages:
18
+
19
+ - `python-docx`
20
+ - `ebooklib`
21
+ - `openai`
22
+ - `python-dotenv`
23
+ - `bs4`
24
+ - `ebooklib`
25
+ - `pdfminer.six`
26
+ - `pillow`
27
+
28
+ ## Usage
29
+
30
+ 1. Ensure all dependencies are installed.
31
+ 2. Set your environment variable for the OpenAI API key.
32
+ 3. Place your ebook files in a known directory.
33
+ 4. Run the script with the path to the ebook file and a metadata dictionary with keys of 'title' and 'author' as arguments.
34
+
35
+ ## Functions
36
+
37
+ - `read_text_file(file: str) -> str`: Reads a text file and returns its content.
38
+ - `write_to_file(content: str, file: str)`: Writes content to a file.
39
+ - `convert_file(file_path: str, metadata: dict) -> str`: Main function to convert an ebook file to text.
40
+
41
+ ## Contributing
42
+
43
+ Contributions to this project are welcome. Please ensure that your code follows the existing style for consistency.
44
+
45
+ ## License
46
+
47
+ This project is licensed by ProsePal LLC under the MIT license
48
+
49
+ ## Version History
50
+
51
+ - **v0.1.0** (Release date: November 30, 2023)
52
+ - Initial release
53
+
54
+ - **v0.1.1** (Release date: December 2, 2023)
55
+ - fixed false positives for is_number
56
+
57
+ - **v0.2.0** (Release date: December 3, 2023)
58
+ - Conversion of docx files
59
+
60
+ - **v0.3.0** (Release date: December 8, 2023)
61
+ - Conversion of PDF files
62
+
63
+ - **v0.3.1** (Release date: Januar 23, 2024)
64
+ - fixed concantation of text in pdf conversion
65
+ - updated pillow version to secure version
66
+
67
+ - **v1.0.0** (Release date: January 23, 2024)
68
+ - created library instead of single module
69
+
70
+ - **v1.0.1** (Release date: March 13, 2024)
71
+ - setup.py and requirements.txt typo fixed
72
+
73
+ - **v1.0.2** (Release date: May 17, 2024)
74
+ - added tests, fixex minor typos
75
+
76
+ - **v1.1.0** (Release date: May 30, 2024)
77
+ - Change to abstract factory pattern
@@ -0,0 +1 @@
1
+ __version__ = "1.1.0"
@@ -0,0 +1,6 @@
1
+ from .convert_file import convert_file
2
+ from .file_handling import read_text_file, write_to_file
3
+ from .VERSION import __version__
4
+
5
+ __version__ = __version__
6
+ __all__ = ["convert_file", "read_text_file", "write_to_file"]
@@ -0,0 +1,24 @@
1
+ class ImageSizeError(Exception):
2
+ """
3
+ Custom exception class for image size-related errors.
4
+ """
5
+
6
+ pass
7
+
8
+
9
+ class ImageTooSmallError(ImageSizeError):
10
+ """
11
+ Custom exception class for handling errors related to an image being too
12
+ small.
13
+ """
14
+
15
+ pass
16
+
17
+
18
+ class ImageTooLargeError(ImageSizeError):
19
+ """
20
+ Custom exception class for handling errors related to an image being too
21
+ large.
22
+ """
23
+
24
+ pass
@@ -0,0 +1,6 @@
1
+ docx_ns_map = {
2
+ "a": "http://schemas.openxmlformats.org/drawingml/2006/main",
3
+ "pic": "http://schemas.openxmlformats.org/drawingml/2006/picture",
4
+ "w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main",
5
+ "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships",
6
+ }
@@ -0,0 +1,20 @@
1
+ from bs4.element import ResultSet, Tag
2
+ from docx.document import Document
3
+ from docx.text.paragraph import Paragraph
4
+ from ebooklib.epub import EpubBook, EpubItem
5
+ from pdfminer.layout import LTChar, LTContainer, LTPage, LTText
6
+ from pdfminer.pdftypes import PDFStream
7
+
8
+ __all__ = [
9
+ "Document",
10
+ "EpubBook",
11
+ "EpubItem",
12
+ "LTChar",
13
+ "LTContainer",
14
+ "LTText",
15
+ "LTPage",
16
+ "Paragraph",
17
+ "PDFStream",
18
+ "ResultSet",
19
+ "Tag",
20
+ ]
@@ -0,0 +1,68 @@
1
+ from abc import ABC, abstractmethod
2
+ from typing import Generic, List, TypeVar
3
+
4
+ from .text_conversion import desmarten_text
5
+
6
+ T = TypeVar("T")
7
+
8
+
9
+ class BookConversion(ABC, Generic[T]):
10
+ def __init__(self, file_path: str, metadata: dict):
11
+ self.file_path: str = file_path
12
+ self.metadata: dict = metadata
13
+ self.book = self._read_file(file_path)
14
+
15
+ @abstractmethod
16
+ def _read_file(self, file_path: str) -> T:
17
+ raise NotImplementedError("Must be implemented in child class")
18
+
19
+ @abstractmethod
20
+ def split_chapters(self) -> str:
21
+ raise NotImplementedError("Must be implemented in child class")
22
+
23
+ @abstractmethod
24
+ def extract_text(self, T) -> str:
25
+ raise NotImplementedError("Must be implemented in child class")
26
+
27
+ @abstractmethod
28
+ def extract_images(self, T) -> List[str]:
29
+ raise NotImplementedError("Must be implemented in child class")
30
+
31
+
32
+ class ImageExtraction(ABC, Generic[T]):
33
+ @abstractmethod
34
+ def extract_images(self, T) -> List[str]:
35
+ raise NotImplementedError("Must be implemented in child class")
36
+
37
+
38
+ class TextExtraction(ABC, Generic[T]):
39
+ @abstractmethod
40
+ def extract_text(self, T) -> str:
41
+ raise NotImplementedError("Must be implemented in child class")
42
+
43
+ @abstractmethod
44
+ def _extract_image_text(self, T) -> str:
45
+ raise NotImplementedError("Must be implemented in child class")
46
+
47
+
48
+ class ChapterSplit(ABC, Generic[T]):
49
+ @abstractmethod
50
+ def __init__(
51
+ self, text_obj: T, metadata: dict, converter: BookConversion
52
+ ) -> None:
53
+ self.text_obj = text_obj
54
+ self.metadata = metadata
55
+ self.converter = converter
56
+
57
+ self.MAX_LINES_TO_CHECK: int = 3
58
+ self.CHAPTER_SEPARATOR: str = "***"
59
+
60
+ @abstractmethod
61
+ def split_chapters(self) -> str:
62
+ raise NotImplementedError("Must be implemented in child class")
63
+
64
+ def clean_text(self, text: str) -> str:
65
+ """
66
+ Removes smart punctuation from text
67
+ """
68
+ return desmarten_text(text)
@@ -0,0 +1,229 @@
1
+ NOT_CHAPTER = {
2
+ "about",
3
+ "acknowledgements",
4
+ "afterward",
5
+ "annotation",
6
+ "appendix",
7
+ "assessment",
8
+ "backmatter",
9
+ "bibliography",
10
+ "colophon",
11
+ "conclusion",
12
+ "contents",
13
+ "contributors",
14
+ "copyright",
15
+ "cover",
16
+ "credits",
17
+ "dedication",
18
+ "division",
19
+ "endnotes",
20
+ "epigraph",
21
+ "errata",
22
+ "footnotes",
23
+ "forward",
24
+ "frontmatter",
25
+ "glossary",
26
+ "imprintur",
27
+ "imprint",
28
+ "index",
29
+ "introduction",
30
+ "landmarks",
31
+ "list",
32
+ "notice",
33
+ "page",
34
+ "preamble",
35
+ "preface",
36
+ "prologue",
37
+ "question",
38
+ "rear",
39
+ "revision",
40
+ "sign up",
41
+ "table",
42
+ "toc",
43
+ "volume",
44
+ "warning",
45
+ }
46
+
47
+
48
+ def roman_to_int(roman_num: str) -> int:
49
+ """
50
+ Convert a Roman numeral to an integer.
51
+ Arguments:
52
+ roman_num (str): A string representing the Roman numeral.
53
+ Returns int: The integer value of the Roman numeral.
54
+ """
55
+ if not isinstance(roman_num, str):
56
+ raise TypeError("Input must be a string")
57
+
58
+ roman: str = roman_num.upper()
59
+
60
+ for numeral in ("V", "L", "D"):
61
+ if roman.count(numeral) > 1:
62
+ raise ValueError("Not roman numeral")
63
+
64
+ ROMAN_NUMERALS = {
65
+ "I": 1,
66
+ "V": 5,
67
+ "X": 10,
68
+ "L": 50,
69
+ "C": 100,
70
+ "D": 500,
71
+ "M": 1000,
72
+ }
73
+
74
+ total: int = 0
75
+ prev_value: int = 0
76
+ consecutive_count: int = 1
77
+ previous_char: str = ""
78
+
79
+ for char in reversed(roman):
80
+ if char not in ROMAN_NUMERALS:
81
+ raise ValueError("Not roman numeral")
82
+
83
+ consecutive_count = (
84
+ consecutive_count + 1 if char == previous_char else 1
85
+ )
86
+ if consecutive_count > 3:
87
+ raise ValueError("Not roman numeral")
88
+
89
+ value = ROMAN_NUMERALS[char]
90
+ if value >= prev_value:
91
+ total += value
92
+ else:
93
+ if previous_char == "":
94
+ raise ValueError("Not roman numeral")
95
+ elif previous_char in ("V", "X") and char != "I":
96
+ raise ValueError("Not roman numeral")
97
+ elif previous_char in ("L", "C") and char != "X":
98
+ raise ValueError("Not roman numeral")
99
+ elif previous_char in ("D", "M") and char != "C":
100
+ raise ValueError("Not roman numeral")
101
+ total -= value
102
+ prev_value = value
103
+ previous_char = char
104
+ if not total:
105
+ raise ValueError("Not roman numeral")
106
+ return total
107
+
108
+
109
+ def word_to_num(number_string: str) -> int:
110
+ """
111
+ Convert a spelled-out number without spaces or hyphens to a string of the
112
+ integer. The function supports numbers from 0 to 99 and is case
113
+ insensitive.
114
+ Args:
115
+ number_string (str): A string representing the spelled-out number.
116
+ Returns:
117
+ int: The integer value of the spelled-out number.
118
+ """
119
+ if not isinstance(number_string, str):
120
+ raise TypeError("Must be a string")
121
+ if len(number_string) == 0:
122
+ raise ValueError("Must have a value")
123
+
124
+ num_words = {
125
+ "zero": 0,
126
+ "one": 1,
127
+ "two": 2,
128
+ "three": 3,
129
+ "four": 4,
130
+ "five": 5,
131
+ "six": 6,
132
+ "seven": 7,
133
+ "eight": 8,
134
+ "nine": 9,
135
+ "ten": 10,
136
+ "teen": 10,
137
+ "eleven": 11,
138
+ "twelve": 12,
139
+ "thirteen": 13,
140
+ "twenty": 20,
141
+ "thirty": 30,
142
+ "forty": 40,
143
+ "fifty": 50,
144
+ "sixty": 60,
145
+ "seventy": 70,
146
+ "eighty": 80,
147
+ "ninety": 90,
148
+ }
149
+
150
+ num_str_lower: str = number_string.lower()
151
+ cleaned_num_str: str = num_str_lower.replace("-", "").replace(" ", "")
152
+ total: int = 0
153
+ temp_word: str = ""
154
+
155
+ for char in reversed(cleaned_num_str):
156
+ temp_word = char + temp_word
157
+ if temp_word in num_words:
158
+ total += num_words[temp_word]
159
+ temp_word = ""
160
+
161
+ if temp_word:
162
+ raise ValueError(f"Unknown number word: {temp_word}")
163
+ return total
164
+
165
+
166
+ def is_spelled_out_number(s: str) -> bool:
167
+ """
168
+ Try to convert the word to an integer. If it's not a valid spelled-out
169
+ number, it will raise a ValueError
170
+ """
171
+
172
+ try:
173
+ word_to_num(s)
174
+ return True
175
+ except ValueError:
176
+ return False
177
+
178
+
179
+ def is_roman_numeral(word: str) -> bool:
180
+ """
181
+ Try to convert the word to an integer. If it"s not a valid roman numeral,
182
+ it will raise a ValueError
183
+ """
184
+
185
+ try:
186
+ roman_to_int(word)
187
+ return True
188
+ except ValueError:
189
+ return False
190
+ except TypeError:
191
+ return False
192
+
193
+
194
+ def is_number(s: str) -> bool:
195
+ """
196
+ Check if a string contains a number value either as digits,
197
+ roman numerals, or spelled-out numbers.
198
+ """
199
+
200
+ return s.isdigit() or is_roman_numeral(s) or is_spelled_out_number(s)
201
+
202
+
203
+ def is_chapter(s: str) -> bool:
204
+ """
205
+ Check if a string contains the word "chapter", a Roman numeral, a
206
+ spelled-out number, or a digit.
207
+ Arguments:
208
+ s (str): The string to check.
209
+ Returns bool: True if the string meets the criteria, False otherwise.
210
+ """
211
+ lower_s = s.lower().strip()
212
+ return lower_s.startswith("chapter") or (
213
+ len(lower_s.split()) == 1 and is_number(lower_s)
214
+ )
215
+
216
+
217
+ def is_not_chapter(paragraph: str, metadata: dict) -> bool:
218
+ """
219
+ Checks if the given line is not a chapter.
220
+ """
221
+ title = metadata.get("title", "no title found")
222
+ author = metadata.get("author", "no author found")
223
+ paragraph = paragraph.lower()
224
+ return any(
225
+ paragraph.startswith(title.lower())
226
+ or paragraph.startswith(author.lower())
227
+ or paragraph.startswith(not_chapter_word)
228
+ for not_chapter_word in NOT_CHAPTER
229
+ )
@@ -0,0 +1,46 @@
1
+ import logging
2
+ import os
3
+
4
+ from .docx_conversion import read_docx
5
+ from .epub_conversion import read_epub
6
+ from .file_handling import read_text_file, write_to_file
7
+ from .pdf_conversion import read_pdf
8
+ from .text_conversion import parse_text_file
9
+
10
+
11
+ def convert_file(file_path: str, metadata: dict) -> None:
12
+ """
13
+ Converts a book to a text file with 3 asterisks for chapter breaks
14
+ Args:
15
+ book_name: Name of the book.
16
+ folder_name: Name of the folder containing the book.
17
+ """
18
+
19
+ book_content = ""
20
+ folder, book_file = os.path.split(file_path)
21
+
22
+ book_file = book_file.replace(" ", "_")
23
+ book_file = book_file.replace("-", "_")
24
+ filename_list = book_file.split(".")
25
+
26
+ if len(filename_list) > 1:
27
+ base_name = "_".join(filename_list[:-1])
28
+ else:
29
+ base_name = filename_list[0]
30
+ extension = filename_list[-1].lower()
31
+
32
+ if extension == "epub":
33
+ book_content = read_epub(file_path, metadata)
34
+ elif extension == "docx":
35
+ book_content = read_docx(file_path, metadata)
36
+ elif extension == "pdf":
37
+ book_content = read_pdf(file_path, metadata)
38
+ elif extension == "txt" or extension == "text":
39
+ book_content = read_text_file(file_path)
40
+ book_content = parse_text_file(book_content)
41
+ else:
42
+ logging.error(f"Invalid file type {extension} for file {file_path}")
43
+
44
+ book_name = f"{base_name}.txt"
45
+ book_path = os.path.join(folder, book_name)
46
+ write_to_file(book_content, book_path)