ebook2text 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
ebook2text/VERSION.py ADDED
@@ -0,0 +1 @@
1
+ __version__ = "1.1.0"
ebook2text/__init__.py ADDED
@@ -0,0 +1,6 @@
1
+ from .convert_file import convert_file
2
+ from .file_handling import read_text_file, write_to_file
3
+ from .VERSION import __version__
4
+
5
+ __version__ = __version__
6
+ __all__ = ["convert_file", "read_text_file", "write_to_file"]
@@ -0,0 +1,24 @@
1
+ class ImageSizeError(Exception):
2
+ """
3
+ Custom exception class for image size-related errors.
4
+ """
5
+
6
+ pass
7
+
8
+
9
+ class ImageTooSmallError(ImageSizeError):
10
+ """
11
+ Custom exception class for handling errors related to an image being too
12
+ small.
13
+ """
14
+
15
+ pass
16
+
17
+
18
+ class ImageTooLargeError(ImageSizeError):
19
+ """
20
+ Custom exception class for handling errors related to an image being too
21
+ large.
22
+ """
23
+
24
+ pass
@@ -0,0 +1,6 @@
1
+ docx_ns_map = {
2
+ "a": "http://schemas.openxmlformats.org/drawingml/2006/main",
3
+ "pic": "http://schemas.openxmlformats.org/drawingml/2006/picture",
4
+ "w": "http://schemas.openxmlformats.org/wordprocessingml/2006/main",
5
+ "r": "http://schemas.openxmlformats.org/officeDocument/2006/relationships",
6
+ }
ebook2text/_types.py ADDED
@@ -0,0 +1,20 @@
1
+ from bs4.element import ResultSet, Tag
2
+ from docx.document import Document
3
+ from docx.text.paragraph import Paragraph
4
+ from ebooklib.epub import EpubBook, EpubItem
5
+ from pdfminer.layout import LTChar, LTContainer, LTPage, LTText
6
+ from pdfminer.pdftypes import PDFStream
7
+
8
+ __all__ = [
9
+ "Document",
10
+ "EpubBook",
11
+ "EpubItem",
12
+ "LTChar",
13
+ "LTContainer",
14
+ "LTText",
15
+ "LTPage",
16
+ "Paragraph",
17
+ "PDFStream",
18
+ "ResultSet",
19
+ "Tag",
20
+ ]
@@ -0,0 +1,68 @@
1
+ from abc import ABC, abstractmethod
2
+ from typing import Generic, List, TypeVar
3
+
4
+ from .text_conversion import desmarten_text
5
+
6
+ T = TypeVar("T")
7
+
8
+
9
+ class BookConversion(ABC, Generic[T]):
10
+ def __init__(self, file_path: str, metadata: dict):
11
+ self.file_path: str = file_path
12
+ self.metadata: dict = metadata
13
+ self.book = self._read_file(file_path)
14
+
15
+ @abstractmethod
16
+ def _read_file(self, file_path: str) -> T:
17
+ raise NotImplementedError("Must be implemented in child class")
18
+
19
+ @abstractmethod
20
+ def split_chapters(self) -> str:
21
+ raise NotImplementedError("Must be implemented in child class")
22
+
23
+ @abstractmethod
24
+ def extract_text(self, T) -> str:
25
+ raise NotImplementedError("Must be implemented in child class")
26
+
27
+ @abstractmethod
28
+ def extract_images(self, T) -> List[str]:
29
+ raise NotImplementedError("Must be implemented in child class")
30
+
31
+
32
+ class ImageExtraction(ABC, Generic[T]):
33
+ @abstractmethod
34
+ def extract_images(self, T) -> List[str]:
35
+ raise NotImplementedError("Must be implemented in child class")
36
+
37
+
38
+ class TextExtraction(ABC, Generic[T]):
39
+ @abstractmethod
40
+ def extract_text(self, T) -> str:
41
+ raise NotImplementedError("Must be implemented in child class")
42
+
43
+ @abstractmethod
44
+ def _extract_image_text(self, T) -> str:
45
+ raise NotImplementedError("Must be implemented in child class")
46
+
47
+
48
+ class ChapterSplit(ABC, Generic[T]):
49
+ @abstractmethod
50
+ def __init__(
51
+ self, text_obj: T, metadata: dict, converter: BookConversion
52
+ ) -> None:
53
+ self.text_obj = text_obj
54
+ self.metadata = metadata
55
+ self.converter = converter
56
+
57
+ self.MAX_LINES_TO_CHECK: int = 3
58
+ self.CHAPTER_SEPARATOR: str = "***"
59
+
60
+ @abstractmethod
61
+ def split_chapters(self) -> str:
62
+ raise NotImplementedError("Must be implemented in child class")
63
+
64
+ def clean_text(self, text: str) -> str:
65
+ """
66
+ Removes smart punctuation from text
67
+ """
68
+ return desmarten_text(text)
@@ -0,0 +1,229 @@
1
+ NOT_CHAPTER = {
2
+ "about",
3
+ "acknowledgements",
4
+ "afterward",
5
+ "annotation",
6
+ "appendix",
7
+ "assessment",
8
+ "backmatter",
9
+ "bibliography",
10
+ "colophon",
11
+ "conclusion",
12
+ "contents",
13
+ "contributors",
14
+ "copyright",
15
+ "cover",
16
+ "credits",
17
+ "dedication",
18
+ "division",
19
+ "endnotes",
20
+ "epigraph",
21
+ "errata",
22
+ "footnotes",
23
+ "forward",
24
+ "frontmatter",
25
+ "glossary",
26
+ "imprintur",
27
+ "imprint",
28
+ "index",
29
+ "introduction",
30
+ "landmarks",
31
+ "list",
32
+ "notice",
33
+ "page",
34
+ "preamble",
35
+ "preface",
36
+ "prologue",
37
+ "question",
38
+ "rear",
39
+ "revision",
40
+ "sign up",
41
+ "table",
42
+ "toc",
43
+ "volume",
44
+ "warning",
45
+ }
46
+
47
+
48
+ def roman_to_int(roman_num: str) -> int:
49
+ """
50
+ Convert a Roman numeral to an integer.
51
+ Arguments:
52
+ roman_num (str): A string representing the Roman numeral.
53
+ Returns int: The integer value of the Roman numeral.
54
+ """
55
+ if not isinstance(roman_num, str):
56
+ raise TypeError("Input must be a string")
57
+
58
+ roman: str = roman_num.upper()
59
+
60
+ for numeral in ("V", "L", "D"):
61
+ if roman.count(numeral) > 1:
62
+ raise ValueError("Not roman numeral")
63
+
64
+ ROMAN_NUMERALS = {
65
+ "I": 1,
66
+ "V": 5,
67
+ "X": 10,
68
+ "L": 50,
69
+ "C": 100,
70
+ "D": 500,
71
+ "M": 1000,
72
+ }
73
+
74
+ total: int = 0
75
+ prev_value: int = 0
76
+ consecutive_count: int = 1
77
+ previous_char: str = ""
78
+
79
+ for char in reversed(roman):
80
+ if char not in ROMAN_NUMERALS:
81
+ raise ValueError("Not roman numeral")
82
+
83
+ consecutive_count = (
84
+ consecutive_count + 1 if char == previous_char else 1
85
+ )
86
+ if consecutive_count > 3:
87
+ raise ValueError("Not roman numeral")
88
+
89
+ value = ROMAN_NUMERALS[char]
90
+ if value >= prev_value:
91
+ total += value
92
+ else:
93
+ if previous_char == "":
94
+ raise ValueError("Not roman numeral")
95
+ elif previous_char in ("V", "X") and char != "I":
96
+ raise ValueError("Not roman numeral")
97
+ elif previous_char in ("L", "C") and char != "X":
98
+ raise ValueError("Not roman numeral")
99
+ elif previous_char in ("D", "M") and char != "C":
100
+ raise ValueError("Not roman numeral")
101
+ total -= value
102
+ prev_value = value
103
+ previous_char = char
104
+ if not total:
105
+ raise ValueError("Not roman numeral")
106
+ return total
107
+
108
+
109
+ def word_to_num(number_string: str) -> int:
110
+ """
111
+ Convert a spelled-out number without spaces or hyphens to a string of the
112
+ integer. The function supports numbers from 0 to 99 and is case
113
+ insensitive.
114
+ Args:
115
+ number_string (str): A string representing the spelled-out number.
116
+ Returns:
117
+ int: The integer value of the spelled-out number.
118
+ """
119
+ if not isinstance(number_string, str):
120
+ raise TypeError("Must be a string")
121
+ if len(number_string) == 0:
122
+ raise ValueError("Must have a value")
123
+
124
+ num_words = {
125
+ "zero": 0,
126
+ "one": 1,
127
+ "two": 2,
128
+ "three": 3,
129
+ "four": 4,
130
+ "five": 5,
131
+ "six": 6,
132
+ "seven": 7,
133
+ "eight": 8,
134
+ "nine": 9,
135
+ "ten": 10,
136
+ "teen": 10,
137
+ "eleven": 11,
138
+ "twelve": 12,
139
+ "thirteen": 13,
140
+ "twenty": 20,
141
+ "thirty": 30,
142
+ "forty": 40,
143
+ "fifty": 50,
144
+ "sixty": 60,
145
+ "seventy": 70,
146
+ "eighty": 80,
147
+ "ninety": 90,
148
+ }
149
+
150
+ num_str_lower: str = number_string.lower()
151
+ cleaned_num_str: str = num_str_lower.replace("-", "").replace(" ", "")
152
+ total: int = 0
153
+ temp_word: str = ""
154
+
155
+ for char in reversed(cleaned_num_str):
156
+ temp_word = char + temp_word
157
+ if temp_word in num_words:
158
+ total += num_words[temp_word]
159
+ temp_word = ""
160
+
161
+ if temp_word:
162
+ raise ValueError(f"Unknown number word: {temp_word}")
163
+ return total
164
+
165
+
166
+ def is_spelled_out_number(s: str) -> bool:
167
+ """
168
+ Try to convert the word to an integer. If it's not a valid spelled-out
169
+ number, it will raise a ValueError
170
+ """
171
+
172
+ try:
173
+ word_to_num(s)
174
+ return True
175
+ except ValueError:
176
+ return False
177
+
178
+
179
+ def is_roman_numeral(word: str) -> bool:
180
+ """
181
+ Try to convert the word to an integer. If it"s not a valid roman numeral,
182
+ it will raise a ValueError
183
+ """
184
+
185
+ try:
186
+ roman_to_int(word)
187
+ return True
188
+ except ValueError:
189
+ return False
190
+ except TypeError:
191
+ return False
192
+
193
+
194
+ def is_number(s: str) -> bool:
195
+ """
196
+ Check if a string contains a number value either as digits,
197
+ roman numerals, or spelled-out numbers.
198
+ """
199
+
200
+ return s.isdigit() or is_roman_numeral(s) or is_spelled_out_number(s)
201
+
202
+
203
+ def is_chapter(s: str) -> bool:
204
+ """
205
+ Check if a string contains the word "chapter", a Roman numeral, a
206
+ spelled-out number, or a digit.
207
+ Arguments:
208
+ s (str): The string to check.
209
+ Returns bool: True if the string meets the criteria, False otherwise.
210
+ """
211
+ lower_s = s.lower().strip()
212
+ return lower_s.startswith("chapter") or (
213
+ len(lower_s.split()) == 1 and is_number(lower_s)
214
+ )
215
+
216
+
217
+ def is_not_chapter(paragraph: str, metadata: dict) -> bool:
218
+ """
219
+ Checks if the given line is not a chapter.
220
+ """
221
+ title = metadata.get("title", "no title found")
222
+ author = metadata.get("author", "no author found")
223
+ paragraph = paragraph.lower()
224
+ return any(
225
+ paragraph.startswith(title.lower())
226
+ or paragraph.startswith(author.lower())
227
+ or paragraph.startswith(not_chapter_word)
228
+ for not_chapter_word in NOT_CHAPTER
229
+ )
@@ -0,0 +1,46 @@
1
+ import logging
2
+ import os
3
+
4
+ from .docx_conversion import read_docx
5
+ from .epub_conversion import read_epub
6
+ from .file_handling import read_text_file, write_to_file
7
+ from .pdf_conversion import read_pdf
8
+ from .text_conversion import parse_text_file
9
+
10
+
11
+ def convert_file(file_path: str, metadata: dict) -> None:
12
+ """
13
+ Converts a book to a text file with 3 asterisks for chapter breaks
14
+ Args:
15
+ book_name: Name of the book.
16
+ folder_name: Name of the folder containing the book.
17
+ """
18
+
19
+ book_content = ""
20
+ folder, book_file = os.path.split(file_path)
21
+
22
+ book_file = book_file.replace(" ", "_")
23
+ book_file = book_file.replace("-", "_")
24
+ filename_list = book_file.split(".")
25
+
26
+ if len(filename_list) > 1:
27
+ base_name = "_".join(filename_list[:-1])
28
+ else:
29
+ base_name = filename_list[0]
30
+ extension = filename_list[-1].lower()
31
+
32
+ if extension == "epub":
33
+ book_content = read_epub(file_path, metadata)
34
+ elif extension == "docx":
35
+ book_content = read_docx(file_path, metadata)
36
+ elif extension == "pdf":
37
+ book_content = read_pdf(file_path, metadata)
38
+ elif extension == "txt" or extension == "text":
39
+ book_content = read_text_file(file_path)
40
+ book_content = parse_text_file(book_content)
41
+ else:
42
+ logging.error(f"Invalid file type {extension} for file {file_path}")
43
+
44
+ book_name = f"{base_name}.txt"
45
+ book_path = os.path.join(folder, book_name)
46
+ write_to_file(book_content, book_path)