ebook2text 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ebook2text/VERSION.py +1 -0
- ebook2text/__init__.py +6 -0
- ebook2text/_exceptions.py +24 -0
- ebook2text/_namespaces.py +6 -0
- ebook2text/_types.py +20 -0
- ebook2text/abstract_book.py +68 -0
- ebook2text/chapter_check.py +229 -0
- ebook2text/convert_file.py +46 -0
- ebook2text/docx_conversion.py +322 -0
- ebook2text/epub_conversion.py +179 -0
- ebook2text/file_handling.py +9 -0
- ebook2text/ocr.py +85 -0
- ebook2text/pdf_conversion.py +467 -0
- ebook2text/text_conversion.py +55 -0
- ebook2text-1.1.0.dist-info/LICENSE.txt +7 -0
- ebook2text-1.1.0.dist-info/METADATA +129 -0
- ebook2text-1.1.0.dist-info/RECORD +19 -0
- ebook2text-1.1.0.dist-info/WHEEL +5 -0
- ebook2text-1.1.0.dist-info/top_level.txt +1 -0
ebook2text/VERSION.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "1.1.0"
|
ebook2text/__init__.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
class ImageSizeError(Exception):
|
|
2
|
+
"""
|
|
3
|
+
Custom exception class for image size-related errors.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
pass
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class ImageTooSmallError(ImageSizeError):
|
|
10
|
+
"""
|
|
11
|
+
Custom exception class for handling errors related to an image being too
|
|
12
|
+
small.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
pass
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class ImageTooLargeError(ImageSizeError):
|
|
19
|
+
"""
|
|
20
|
+
Custom exception class for handling errors related to an image being too
|
|
21
|
+
large.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
pass
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
docx_ns_map = {
|
|
2
|
+
"a": "http://schemas.openxmlformats.org/drawingml/2006/main",
|
|
3
|
+
"pic": "http://schemas.openxmlformats.org/drawingml/2006/picture",
|
|
4
|
+
"w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main",
|
|
5
|
+
"r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships",
|
|
6
|
+
}
|
ebook2text/_types.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
from bs4.element import ResultSet, Tag
|
|
2
|
+
from docx.document import Document
|
|
3
|
+
from docx.text.paragraph import Paragraph
|
|
4
|
+
from ebooklib.epub import EpubBook, EpubItem
|
|
5
|
+
from pdfminer.layout import LTChar, LTContainer, LTPage, LTText
|
|
6
|
+
from pdfminer.pdftypes import PDFStream
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"Document",
|
|
10
|
+
"EpubBook",
|
|
11
|
+
"EpubItem",
|
|
12
|
+
"LTChar",
|
|
13
|
+
"LTContainer",
|
|
14
|
+
"LTText",
|
|
15
|
+
"LTPage",
|
|
16
|
+
"Paragraph",
|
|
17
|
+
"PDFStream",
|
|
18
|
+
"ResultSet",
|
|
19
|
+
"Tag",
|
|
20
|
+
]
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
from typing import Generic, List, TypeVar
|
|
3
|
+
|
|
4
|
+
from .text_conversion import desmarten_text
|
|
5
|
+
|
|
6
|
+
T = TypeVar("T")
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class BookConversion(ABC, Generic[T]):
|
|
10
|
+
def __init__(self, file_path: str, metadata: dict):
|
|
11
|
+
self.file_path: str = file_path
|
|
12
|
+
self.metadata: dict = metadata
|
|
13
|
+
self.book = self._read_file(file_path)
|
|
14
|
+
|
|
15
|
+
@abstractmethod
|
|
16
|
+
def _read_file(self, file_path: str) -> T:
|
|
17
|
+
raise NotImplementedError("Must be implemented in child class")
|
|
18
|
+
|
|
19
|
+
@abstractmethod
|
|
20
|
+
def split_chapters(self) -> str:
|
|
21
|
+
raise NotImplementedError("Must be implemented in child class")
|
|
22
|
+
|
|
23
|
+
@abstractmethod
|
|
24
|
+
def extract_text(self, T) -> str:
|
|
25
|
+
raise NotImplementedError("Must be implemented in child class")
|
|
26
|
+
|
|
27
|
+
@abstractmethod
|
|
28
|
+
def extract_images(self, T) -> List[str]:
|
|
29
|
+
raise NotImplementedError("Must be implemented in child class")
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class ImageExtraction(ABC, Generic[T]):
|
|
33
|
+
@abstractmethod
|
|
34
|
+
def extract_images(self, T) -> List[str]:
|
|
35
|
+
raise NotImplementedError("Must be implemented in child class")
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class TextExtraction(ABC, Generic[T]):
|
|
39
|
+
@abstractmethod
|
|
40
|
+
def extract_text(self, T) -> str:
|
|
41
|
+
raise NotImplementedError("Must be implemented in child class")
|
|
42
|
+
|
|
43
|
+
@abstractmethod
|
|
44
|
+
def _extract_image_text(self, T) -> str:
|
|
45
|
+
raise NotImplementedError("Must be implemented in child class")
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class ChapterSplit(ABC, Generic[T]):
|
|
49
|
+
@abstractmethod
|
|
50
|
+
def __init__(
|
|
51
|
+
self, text_obj: T, metadata: dict, converter: BookConversion
|
|
52
|
+
) -> None:
|
|
53
|
+
self.text_obj = text_obj
|
|
54
|
+
self.metadata = metadata
|
|
55
|
+
self.converter = converter
|
|
56
|
+
|
|
57
|
+
self.MAX_LINES_TO_CHECK: int = 3
|
|
58
|
+
self.CHAPTER_SEPARATOR: str = "***"
|
|
59
|
+
|
|
60
|
+
@abstractmethod
|
|
61
|
+
def split_chapters(self) -> str:
|
|
62
|
+
raise NotImplementedError("Must be implemented in child class")
|
|
63
|
+
|
|
64
|
+
def clean_text(self, text: str) -> str:
|
|
65
|
+
"""
|
|
66
|
+
Removes smart punctuation from text
|
|
67
|
+
"""
|
|
68
|
+
return desmarten_text(text)
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
NOT_CHAPTER = {
|
|
2
|
+
"about",
|
|
3
|
+
"acknowledgements",
|
|
4
|
+
"afterward",
|
|
5
|
+
"annotation",
|
|
6
|
+
"appendix",
|
|
7
|
+
"assessment",
|
|
8
|
+
"backmatter",
|
|
9
|
+
"bibliography",
|
|
10
|
+
"colophon",
|
|
11
|
+
"conclusion",
|
|
12
|
+
"contents",
|
|
13
|
+
"contributors",
|
|
14
|
+
"copyright",
|
|
15
|
+
"cover",
|
|
16
|
+
"credits",
|
|
17
|
+
"dedication",
|
|
18
|
+
"division",
|
|
19
|
+
"endnotes",
|
|
20
|
+
"epigraph",
|
|
21
|
+
"errata",
|
|
22
|
+
"footnotes",
|
|
23
|
+
"forward",
|
|
24
|
+
"frontmatter",
|
|
25
|
+
"glossary",
|
|
26
|
+
"imprintur",
|
|
27
|
+
"imprint",
|
|
28
|
+
"index",
|
|
29
|
+
"introduction",
|
|
30
|
+
"landmarks",
|
|
31
|
+
"list",
|
|
32
|
+
"notice",
|
|
33
|
+
"page",
|
|
34
|
+
"preamble",
|
|
35
|
+
"preface",
|
|
36
|
+
"prologue",
|
|
37
|
+
"question",
|
|
38
|
+
"rear",
|
|
39
|
+
"revision",
|
|
40
|
+
"sign up",
|
|
41
|
+
"table",
|
|
42
|
+
"toc",
|
|
43
|
+
"volume",
|
|
44
|
+
"warning",
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def roman_to_int(roman_num: str) -> int:
|
|
49
|
+
"""
|
|
50
|
+
Convert a Roman numeral to an integer.
|
|
51
|
+
Arguments:
|
|
52
|
+
roman_num (str): A string representing the Roman numeral.
|
|
53
|
+
Returns int: The integer value of the Roman numeral.
|
|
54
|
+
"""
|
|
55
|
+
if not isinstance(roman_num, str):
|
|
56
|
+
raise TypeError("Input must be a string")
|
|
57
|
+
|
|
58
|
+
roman: str = roman_num.upper()
|
|
59
|
+
|
|
60
|
+
for numeral in ("V", "L", "D"):
|
|
61
|
+
if roman.count(numeral) > 1:
|
|
62
|
+
raise ValueError("Not roman numeral")
|
|
63
|
+
|
|
64
|
+
ROMAN_NUMERALS = {
|
|
65
|
+
"I": 1,
|
|
66
|
+
"V": 5,
|
|
67
|
+
"X": 10,
|
|
68
|
+
"L": 50,
|
|
69
|
+
"C": 100,
|
|
70
|
+
"D": 500,
|
|
71
|
+
"M": 1000,
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
total: int = 0
|
|
75
|
+
prev_value: int = 0
|
|
76
|
+
consecutive_count: int = 1
|
|
77
|
+
previous_char: str = ""
|
|
78
|
+
|
|
79
|
+
for char in reversed(roman):
|
|
80
|
+
if char not in ROMAN_NUMERALS:
|
|
81
|
+
raise ValueError("Not roman numeral")
|
|
82
|
+
|
|
83
|
+
consecutive_count = (
|
|
84
|
+
consecutive_count + 1 if char == previous_char else 1
|
|
85
|
+
)
|
|
86
|
+
if consecutive_count > 3:
|
|
87
|
+
raise ValueError("Not roman numeral")
|
|
88
|
+
|
|
89
|
+
value = ROMAN_NUMERALS[char]
|
|
90
|
+
if value >= prev_value:
|
|
91
|
+
total += value
|
|
92
|
+
else:
|
|
93
|
+
if previous_char == "":
|
|
94
|
+
raise ValueError("Not roman numeral")
|
|
95
|
+
elif previous_char in ("V", "X") and char != "I":
|
|
96
|
+
raise ValueError("Not roman numeral")
|
|
97
|
+
elif previous_char in ("L", "C") and char != "X":
|
|
98
|
+
raise ValueError("Not roman numeral")
|
|
99
|
+
elif previous_char in ("D", "M") and char != "C":
|
|
100
|
+
raise ValueError("Not roman numeral")
|
|
101
|
+
total -= value
|
|
102
|
+
prev_value = value
|
|
103
|
+
previous_char = char
|
|
104
|
+
if not total:
|
|
105
|
+
raise ValueError("Not roman numeral")
|
|
106
|
+
return total
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def word_to_num(number_string: str) -> int:
|
|
110
|
+
"""
|
|
111
|
+
Convert a spelled-out number without spaces or hyphens to a string of the
|
|
112
|
+
integer. The function supports numbers from 0 to 99 and is case
|
|
113
|
+
insensitive.
|
|
114
|
+
Args:
|
|
115
|
+
number_string (str): A string representing the spelled-out number.
|
|
116
|
+
Returns:
|
|
117
|
+
int: The integer value of the spelled-out number.
|
|
118
|
+
"""
|
|
119
|
+
if not isinstance(number_string, str):
|
|
120
|
+
raise TypeError("Must be a string")
|
|
121
|
+
if len(number_string) == 0:
|
|
122
|
+
raise ValueError("Must have a value")
|
|
123
|
+
|
|
124
|
+
num_words = {
|
|
125
|
+
"zero": 0,
|
|
126
|
+
"one": 1,
|
|
127
|
+
"two": 2,
|
|
128
|
+
"three": 3,
|
|
129
|
+
"four": 4,
|
|
130
|
+
"five": 5,
|
|
131
|
+
"six": 6,
|
|
132
|
+
"seven": 7,
|
|
133
|
+
"eight": 8,
|
|
134
|
+
"nine": 9,
|
|
135
|
+
"ten": 10,
|
|
136
|
+
"teen": 10,
|
|
137
|
+
"eleven": 11,
|
|
138
|
+
"twelve": 12,
|
|
139
|
+
"thirteen": 13,
|
|
140
|
+
"twenty": 20,
|
|
141
|
+
"thirty": 30,
|
|
142
|
+
"forty": 40,
|
|
143
|
+
"fifty": 50,
|
|
144
|
+
"sixty": 60,
|
|
145
|
+
"seventy": 70,
|
|
146
|
+
"eighty": 80,
|
|
147
|
+
"ninety": 90,
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
num_str_lower: str = number_string.lower()
|
|
151
|
+
cleaned_num_str: str = num_str_lower.replace("-", "").replace(" ", "")
|
|
152
|
+
total: int = 0
|
|
153
|
+
temp_word: str = ""
|
|
154
|
+
|
|
155
|
+
for char in reversed(cleaned_num_str):
|
|
156
|
+
temp_word = char + temp_word
|
|
157
|
+
if temp_word in num_words:
|
|
158
|
+
total += num_words[temp_word]
|
|
159
|
+
temp_word = ""
|
|
160
|
+
|
|
161
|
+
if temp_word:
|
|
162
|
+
raise ValueError(f"Unknown number word: {temp_word}")
|
|
163
|
+
return total
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def is_spelled_out_number(s: str) -> bool:
|
|
167
|
+
"""
|
|
168
|
+
Try to convert the word to an integer. If it's not a valid spelled-out
|
|
169
|
+
number, it will raise a ValueError
|
|
170
|
+
"""
|
|
171
|
+
|
|
172
|
+
try:
|
|
173
|
+
word_to_num(s)
|
|
174
|
+
return True
|
|
175
|
+
except ValueError:
|
|
176
|
+
return False
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def is_roman_numeral(word: str) -> bool:
|
|
180
|
+
"""
|
|
181
|
+
Try to convert the word to an integer. If it"s not a valid roman numeral,
|
|
182
|
+
it will raise a ValueError
|
|
183
|
+
"""
|
|
184
|
+
|
|
185
|
+
try:
|
|
186
|
+
roman_to_int(word)
|
|
187
|
+
return True
|
|
188
|
+
except ValueError:
|
|
189
|
+
return False
|
|
190
|
+
except TypeError:
|
|
191
|
+
return False
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def is_number(s: str) -> bool:
|
|
195
|
+
"""
|
|
196
|
+
Check if a string contains a number value either as digits,
|
|
197
|
+
roman numerals, or spelled-out numbers.
|
|
198
|
+
"""
|
|
199
|
+
|
|
200
|
+
return s.isdigit() or is_roman_numeral(s) or is_spelled_out_number(s)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def is_chapter(s: str) -> bool:
|
|
204
|
+
"""
|
|
205
|
+
Check if a string contains the word "chapter", a Roman numeral, a
|
|
206
|
+
spelled-out number, or a digit.
|
|
207
|
+
Arguments:
|
|
208
|
+
s (str): The string to check.
|
|
209
|
+
Returns bool: True if the string meets the criteria, False otherwise.
|
|
210
|
+
"""
|
|
211
|
+
lower_s = s.lower().strip()
|
|
212
|
+
return lower_s.startswith("chapter") or (
|
|
213
|
+
len(lower_s.split()) == 1 and is_number(lower_s)
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def is_not_chapter(paragraph: str, metadata: dict) -> bool:
|
|
218
|
+
"""
|
|
219
|
+
Checks if the given line is not a chapter.
|
|
220
|
+
"""
|
|
221
|
+
title = metadata.get("title", "no title found")
|
|
222
|
+
author = metadata.get("author", "no author found")
|
|
223
|
+
paragraph = paragraph.lower()
|
|
224
|
+
return any(
|
|
225
|
+
paragraph.startswith(title.lower())
|
|
226
|
+
or paragraph.startswith(author.lower())
|
|
227
|
+
or paragraph.startswith(not_chapter_word)
|
|
228
|
+
for not_chapter_word in NOT_CHAPTER
|
|
229
|
+
)
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import logging
|
|
2
|
+
import os
|
|
3
|
+
|
|
4
|
+
from .docx_conversion import read_docx
|
|
5
|
+
from .epub_conversion import read_epub
|
|
6
|
+
from .file_handling import read_text_file, write_to_file
|
|
7
|
+
from .pdf_conversion import read_pdf
|
|
8
|
+
from .text_conversion import parse_text_file
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def convert_file(file_path: str, metadata: dict) -> None:
|
|
12
|
+
"""
|
|
13
|
+
Converts a book to a text file with 3 asterisks for chapter breaks
|
|
14
|
+
Args:
|
|
15
|
+
book_name: Name of the book.
|
|
16
|
+
folder_name: Name of the folder containing the book.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
book_content = ""
|
|
20
|
+
folder, book_file = os.path.split(file_path)
|
|
21
|
+
|
|
22
|
+
book_file = book_file.replace(" ", "_")
|
|
23
|
+
book_file = book_file.replace("-", "_")
|
|
24
|
+
filename_list = book_file.split(".")
|
|
25
|
+
|
|
26
|
+
if len(filename_list) > 1:
|
|
27
|
+
base_name = "_".join(filename_list[:-1])
|
|
28
|
+
else:
|
|
29
|
+
base_name = filename_list[0]
|
|
30
|
+
extension = filename_list[-1].lower()
|
|
31
|
+
|
|
32
|
+
if extension == "epub":
|
|
33
|
+
book_content = read_epub(file_path, metadata)
|
|
34
|
+
elif extension == "docx":
|
|
35
|
+
book_content = read_docx(file_path, metadata)
|
|
36
|
+
elif extension == "pdf":
|
|
37
|
+
book_content = read_pdf(file_path, metadata)
|
|
38
|
+
elif extension == "txt" or extension == "text":
|
|
39
|
+
book_content = read_text_file(file_path)
|
|
40
|
+
book_content = parse_text_file(book_content)
|
|
41
|
+
else:
|
|
42
|
+
logging.error(f"Invalid file type {extension} for file {file_path}")
|
|
43
|
+
|
|
44
|
+
book_name = f"{base_name}.txt"
|
|
45
|
+
book_path = os.path.join(folder, book_name)
|
|
46
|
+
write_to_file(book_content, book_path)
|