pyxtxt 0.2__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyxtxt-0.2.1/MANIFEST.in +3 -0
- {pyxtxt-0.2/src/pyxtxt.egg-info → pyxtxt-0.2.1}/PKG-INFO +18 -3
- {pyxtxt-0.2 → pyxtxt-0.2.1}/README.md +17 -2
- pyxtxt-0.2.1/examples.py +179 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/pyproject.toml +7 -1
- pyxtxt-0.2.1/src/pyxtxt/estrattori/eml.py +44 -0
- pyxtxt-0.2.1/src/pyxtxt/estrattori/epub.py +28 -0
- pyxtxt-0.2.1/src/pyxtxt/estrattori/md.py +27 -0
- pyxtxt-0.2.1/src/pyxtxt/estrattori/msg.py +41 -0
- pyxtxt-0.2.1/src/pyxtxt/estrattori/rtf.py +20 -0
- pyxtxt-0.2.1/src/pyxtxt/estrattori/tex.py +21 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1/src/pyxtxt.egg-info}/PKG-INFO +18 -3
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt.egg-info/SOURCES.txt +8 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/LICENCSE +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/setup.cfg +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/__init__.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/core.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/doc.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/docx.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/odt.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/pptx.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/svg.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/xlsx.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt/pyxtxt.py +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt.egg-info/requires.txt +0 -0
- {pyxtxt-0.2 → pyxtxt-0.2.1}/src/pyxtxt.egg-info/top_level.txt +0 -0
pyxtxt-0.2.1/MANIFEST.in
ADDED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.2
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -65,7 +65,7 @@ Requires-Dist: lxml; extra == "all"
|
|
|
65
65
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
66
66
|
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
67
67
|
|
|
68
|
-
**NEW in v0.1
|
|
68
|
+
**NEW in v0.2.1+**: Enhanced support for web content, byte streams, and requests integration!
|
|
69
69
|
|
|
70
70
|
---
|
|
71
71
|
|
|
@@ -205,13 +205,28 @@ text = xtxt(attachment_bytes)
|
|
|
205
205
|
|
|
206
206
|
## 📖 Full Examples
|
|
207
207
|
|
|
208
|
-
See [examples.py](
|
|
208
|
+
See [examples.py](./examples.py) for comprehensive usage examples including:
|
|
209
209
|
- Local file processing
|
|
210
210
|
- Memory buffer handling
|
|
211
211
|
- Web content extraction
|
|
212
212
|
- Error handling patterns
|
|
213
213
|
- All supported formats demonstration
|
|
214
214
|
|
|
215
|
+
### Accessing Examples After Installation
|
|
216
|
+
After installing PyxTxt from PyPI, you can access the examples file:
|
|
217
|
+
|
|
218
|
+
```python
|
|
219
|
+
import pkg_resources
|
|
220
|
+
|
|
221
|
+
# Get path to examples file
|
|
222
|
+
examples_path = pkg_resources.resource_filename('pyxtxt', 'examples.py')
|
|
223
|
+
print(f"Examples file location: {examples_path}")
|
|
224
|
+
|
|
225
|
+
# Or read the content directly
|
|
226
|
+
examples_content = pkg_resources.resource_string('pyxtxt', 'examples.py').decode('utf-8')
|
|
227
|
+
print(examples_content)
|
|
228
|
+
```
|
|
229
|
+
|
|
215
230
|
## 🔒 License
|
|
216
231
|
|
|
217
232
|
Distributed under the MIT License. See LICENSE file for details.
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
8
8
|
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
9
9
|
|
|
10
|
-
**NEW in v0.1
|
|
10
|
+
**NEW in v0.2.1+**: Enhanced support for web content, byte streams, and requests integration!
|
|
11
11
|
|
|
12
12
|
---
|
|
13
13
|
|
|
@@ -147,13 +147,28 @@ text = xtxt(attachment_bytes)
|
|
|
147
147
|
|
|
148
148
|
## 📖 Full Examples
|
|
149
149
|
|
|
150
|
-
See [examples.py](
|
|
150
|
+
See [examples.py](./examples.py) for comprehensive usage examples including:
|
|
151
151
|
- Local file processing
|
|
152
152
|
- Memory buffer handling
|
|
153
153
|
- Web content extraction
|
|
154
154
|
- Error handling patterns
|
|
155
155
|
- All supported formats demonstration
|
|
156
156
|
|
|
157
|
+
### Accessing Examples After Installation
|
|
158
|
+
After installing PyxTxt from PyPI, you can access the examples file:
|
|
159
|
+
|
|
160
|
+
```python
|
|
161
|
+
import pkg_resources
|
|
162
|
+
|
|
163
|
+
# Get path to examples file
|
|
164
|
+
examples_path = pkg_resources.resource_filename('pyxtxt', 'examples.py')
|
|
165
|
+
print(f"Examples file location: {examples_path}")
|
|
166
|
+
|
|
167
|
+
# Or read the content directly
|
|
168
|
+
examples_content = pkg_resources.resource_string('pyxtxt', 'examples.py').decode('utf-8')
|
|
169
|
+
print(examples_content)
|
|
170
|
+
```
|
|
171
|
+
|
|
157
172
|
## 🔒 License
|
|
158
173
|
|
|
159
174
|
Distributed under the MIT License. See LICENSE file for details.
|
pyxtxt-0.2.1/examples.py
ADDED
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
PyxTxt - Usage Examples
|
|
4
|
+
=======================
|
|
5
|
+
|
|
6
|
+
This file shows how to use the PyxTxt library to extract text
|
|
7
|
+
from different file types and data streams.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import io
|
|
11
|
+
from pyxtxt import xtxt, extxt_available_formats, xtxt_from_url
|
|
12
|
+
|
|
13
|
+
def example_basic():
|
|
14
|
+
"""Example 1: Extraction from local file"""
|
|
15
|
+
print("=== EXAMPLE 1: Local file ====")
|
|
16
|
+
|
|
17
|
+
# Extract text from a local PDF file
|
|
18
|
+
try:
|
|
19
|
+
text = xtxt("test.pdf")
|
|
20
|
+
if text:
|
|
21
|
+
print(f"Extracted text: {text[:100]}...")
|
|
22
|
+
else:
|
|
23
|
+
print("No text extracted or file not found")
|
|
24
|
+
except Exception as e:
|
|
25
|
+
print(f"Error: {e}")
|
|
26
|
+
|
|
27
|
+
def example_buffer():
|
|
28
|
+
"""Example 2: Extraction from memory buffer"""
|
|
29
|
+
print("\n=== EXAMPLE 2: Memory buffer ====")
|
|
30
|
+
|
|
31
|
+
# Read file into memory and process
|
|
32
|
+
try:
|
|
33
|
+
with open("test.txt", "rb") as f:
|
|
34
|
+
buffer = io.BytesIO(f.read())
|
|
35
|
+
text = xtxt(buffer)
|
|
36
|
+
print(f"From buffer: {text}")
|
|
37
|
+
except FileNotFoundError:
|
|
38
|
+
print("File test.txt not found - creating example...")
|
|
39
|
+
# Create a sample text buffer
|
|
40
|
+
sample_text = "This is sample text for PyxTxt!"
|
|
41
|
+
buffer = io.BytesIO(sample_text.encode('utf-8'))
|
|
42
|
+
text = xtxt(buffer)
|
|
43
|
+
print(f"From sample buffer: {text}")
|
|
44
|
+
|
|
45
|
+
def example_bytes():
|
|
46
|
+
"""Example 3: Extraction from bytes object (NEW!)"""
|
|
47
|
+
print("\n=== EXAMPLE 3: Bytes object ====")
|
|
48
|
+
|
|
49
|
+
# Simulate downloaded content
|
|
50
|
+
sample_content = b"This is sample text content downloaded from web"
|
|
51
|
+
text = xtxt(sample_content)
|
|
52
|
+
print(f"From bytes: {text}")
|
|
53
|
+
|
|
54
|
+
# Example with simulated PDF content (won't work but shows usage)
|
|
55
|
+
try:
|
|
56
|
+
with open("test.pdf", "rb") as f:
|
|
57
|
+
pdf_bytes = f.read()
|
|
58
|
+
text = xtxt(pdf_bytes)
|
|
59
|
+
if text:
|
|
60
|
+
print(f"PDF from bytes: {text[:100]}...")
|
|
61
|
+
except FileNotFoundError:
|
|
62
|
+
print("File test.pdf not found for bytes example")
|
|
63
|
+
|
|
64
|
+
def example_requests():
|
|
65
|
+
"""Example 4: Extraction from requests.Response (NEW!)"""
|
|
66
|
+
print("\n=== EXAMPLE 4: requests.Response ====")
|
|
67
|
+
|
|
68
|
+
try:
|
|
69
|
+
import requests
|
|
70
|
+
|
|
71
|
+
# Download a text file
|
|
72
|
+
url = "https://raw.githubusercontent.com/python/cpython/main/README.rst"
|
|
73
|
+
response = requests.get(url)
|
|
74
|
+
|
|
75
|
+
if response.status_code == 200:
|
|
76
|
+
# Method 1: Pass response object directly
|
|
77
|
+
text1 = xtxt(response)
|
|
78
|
+
if text1:
|
|
79
|
+
print(f"From Response object: {text1[:100]}...")
|
|
80
|
+
|
|
81
|
+
# Method 2: Pass response.content (bytes)
|
|
82
|
+
text2 = xtxt(response.content)
|
|
83
|
+
if text2:
|
|
84
|
+
print(f"From response.content: {text2[:100]}...")
|
|
85
|
+
|
|
86
|
+
except ImportError:
|
|
87
|
+
print("requests not installed. Install with: pip install requests")
|
|
88
|
+
except Exception as e:
|
|
89
|
+
print(f"Download error: {e}")
|
|
90
|
+
|
|
91
|
+
def example_url_helper():
|
|
92
|
+
"""Example 5: URL helper function (NEW!)"""
|
|
93
|
+
print("\n=== EXAMPLE 5: xtxt_from_url helper ====")
|
|
94
|
+
|
|
95
|
+
# Download directly from URL
|
|
96
|
+
url = "https://raw.githubusercontent.com/python/cpython/main/README.rst"
|
|
97
|
+
text = xtxt_from_url(url)
|
|
98
|
+
|
|
99
|
+
if text:
|
|
100
|
+
print(f"Downloaded from URL: {text[:100]}...")
|
|
101
|
+
else:
|
|
102
|
+
print("Download or parsing error")
|
|
103
|
+
|
|
104
|
+
# With additional parameters for requests
|
|
105
|
+
text_with_headers = xtxt_from_url(
|
|
106
|
+
url,
|
|
107
|
+
headers={'User-Agent': 'PyxTxt-Example/1.0'},
|
|
108
|
+
timeout=10
|
|
109
|
+
)
|
|
110
|
+
if text_with_headers:
|
|
111
|
+
print("Download with custom headers successful")
|
|
112
|
+
|
|
113
|
+
def example_supported_formats():
|
|
114
|
+
"""Example 6: Display supported formats"""
|
|
115
|
+
print("\n=== EXAMPLE 6: Supported formats ====")
|
|
116
|
+
|
|
117
|
+
print("Supported MIME types:")
|
|
118
|
+
formats = extxt_available_formats()
|
|
119
|
+
for fmt in formats:
|
|
120
|
+
print(f" - {fmt}")
|
|
121
|
+
|
|
122
|
+
print("\nPretty format names:")
|
|
123
|
+
pretty_formats = extxt_available_formats(pretty=True)
|
|
124
|
+
for fmt in pretty_formats:
|
|
125
|
+
print(f" - {fmt}")
|
|
126
|
+
|
|
127
|
+
def example_web_use_cases():
|
|
128
|
+
"""Example 7: Common use cases for web content"""
|
|
129
|
+
print("\n=== EXAMPLE 7: Web use cases =====")
|
|
130
|
+
|
|
131
|
+
# Case 1: API that returns a document
|
|
132
|
+
try:
|
|
133
|
+
import requests
|
|
134
|
+
|
|
135
|
+
# Simulate API call that returns PDF
|
|
136
|
+
print("Example API call...")
|
|
137
|
+
# api_response = requests.post("https://api.example.com/generate-pdf",
|
|
138
|
+
# json={"type": "report"})
|
|
139
|
+
# text = xtxt(api_response.content)
|
|
140
|
+
|
|
141
|
+
# Case 2: File download from form upload
|
|
142
|
+
print("Example file upload processing...")
|
|
143
|
+
# uploaded_file_content = request.files['document'].read() # Flask example
|
|
144
|
+
# text = xtxt(uploaded_file_content)
|
|
145
|
+
|
|
146
|
+
# Case 3: Email attachments processing
|
|
147
|
+
print("Example email attachment...")
|
|
148
|
+
# attachment_bytes = email_message.get_payload(decode=True)
|
|
149
|
+
# text = xtxt(attachment_bytes)
|
|
150
|
+
|
|
151
|
+
print("See code comments for detailed examples")
|
|
152
|
+
|
|
153
|
+
except ImportError:
|
|
154
|
+
print("requests not available - examples commented out")
|
|
155
|
+
|
|
156
|
+
def main():
|
|
157
|
+
"""Run all examples"""
|
|
158
|
+
print("PyxTxt - Usage Examples")
|
|
159
|
+
print("=" * 40)
|
|
160
|
+
|
|
161
|
+
example_basic()
|
|
162
|
+
example_buffer()
|
|
163
|
+
example_bytes()
|
|
164
|
+
example_requests()
|
|
165
|
+
example_url_helper()
|
|
166
|
+
example_supported_formats()
|
|
167
|
+
example_web_use_cases()
|
|
168
|
+
|
|
169
|
+
print("\n" + "=" * 40)
|
|
170
|
+
print("Examples completed!")
|
|
171
|
+
print("\nNew features added:")
|
|
172
|
+
print("✅ bytes object support")
|
|
173
|
+
print("✅ requests.Response support")
|
|
174
|
+
print("✅ xtxt_from_url() helper function")
|
|
175
|
+
print("✅ Better error handling")
|
|
176
|
+
print("✅ Type hints")
|
|
177
|
+
|
|
178
|
+
if __name__ == "__main__":
|
|
179
|
+
main()
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "pyxtxt"
|
|
3
|
-
version = "0.2"
|
|
3
|
+
version = "0.2.1"
|
|
4
4
|
description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.7"
|
|
@@ -56,3 +56,9 @@ build-backend = "setuptools.build_meta"
|
|
|
56
56
|
[tool.setuptools.packages.find]
|
|
57
57
|
where = ["src"]
|
|
58
58
|
|
|
59
|
+
[tool.setuptools]
|
|
60
|
+
include-package-data = true
|
|
61
|
+
|
|
62
|
+
[tool.setuptools.package-data]
|
|
63
|
+
"*" = ["examples.py"]
|
|
64
|
+
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
import email
|
|
5
|
+
from email import policy
|
|
6
|
+
from bs4 import BeautifulSoup
|
|
7
|
+
except ImportError:
|
|
8
|
+
email = None
|
|
9
|
+
|
|
10
|
+
if email:
|
|
11
|
+
def xtxt_eml(file_buffer):
|
|
12
|
+
content = file_buffer.read()
|
|
13
|
+
if isinstance(content, bytes):
|
|
14
|
+
msg = email.message_from_bytes(content, policy=policy.default)
|
|
15
|
+
else:
|
|
16
|
+
msg = email.message_from_string(content, policy=policy.default)
|
|
17
|
+
|
|
18
|
+
parts = []
|
|
19
|
+
|
|
20
|
+
if msg.is_multipart():
|
|
21
|
+
for part in msg.walk():
|
|
22
|
+
content_type = part.get_content_type()
|
|
23
|
+
if content_type == "text/plain":
|
|
24
|
+
parts.append(part.get_content())
|
|
25
|
+
elif content_type == "text/html":
|
|
26
|
+
html = part.get_content()
|
|
27
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
28
|
+
parts.append(soup.get_text(separator="\n"))
|
|
29
|
+
else:
|
|
30
|
+
content_type = msg.get_content_type()
|
|
31
|
+
payload = msg.get_content()
|
|
32
|
+
if content_type == "text/html":
|
|
33
|
+
soup = BeautifulSoup(payload, "html.parser")
|
|
34
|
+
parts.append(soup.get_text(separator="\n"))
|
|
35
|
+
else:
|
|
36
|
+
parts.append(payload)
|
|
37
|
+
|
|
38
|
+
return "\n\n".join(part.strip() for part in parts if part)
|
|
39
|
+
|
|
40
|
+
register_extractor(
|
|
41
|
+
"message/rfc822",
|
|
42
|
+
xtxt_eml,
|
|
43
|
+
name="EML"
|
|
44
|
+
)
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
from ebooklib import epub
|
|
5
|
+
from bs4 import BeautifulSoup
|
|
6
|
+
except ImportError:
|
|
7
|
+
epub = None
|
|
8
|
+
|
|
9
|
+
if epub:
|
|
10
|
+
def xtxt_epub(file_buffer):
|
|
11
|
+
book = epub.read_epub(file_buffer)
|
|
12
|
+
|
|
13
|
+
testo = []
|
|
14
|
+
|
|
15
|
+
for item in book.get_items():
|
|
16
|
+
if item.get_type() == epub.ITEM_DOCUMENT:
|
|
17
|
+
soup = BeautifulSoup(item.get_body_content(), 'html.parser')
|
|
18
|
+
estratto = soup.get_text(separator='\n', strip=True)
|
|
19
|
+
if estratto:
|
|
20
|
+
testo.append(estratto)
|
|
21
|
+
|
|
22
|
+
return "\n\n".join(testo)
|
|
23
|
+
|
|
24
|
+
register_extractor(
|
|
25
|
+
"application/epub+zip",
|
|
26
|
+
xtxt_epub,
|
|
27
|
+
name="EPUB"
|
|
28
|
+
)
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
import markdown
|
|
5
|
+
from bs4 import BeautifulSoup
|
|
6
|
+
except ImportError:
|
|
7
|
+
markdown = None
|
|
8
|
+
|
|
9
|
+
if markdown:
|
|
10
|
+
def xtxt_md(file_buffer):
|
|
11
|
+
# Legge il file come testo
|
|
12
|
+
content = file_buffer.read()
|
|
13
|
+
if isinstance(content, bytes):
|
|
14
|
+
content = content.decode("utf-8", errors="ignore")
|
|
15
|
+
|
|
16
|
+
# Converte Markdown in HTML
|
|
17
|
+
html = markdown.markdown(content)
|
|
18
|
+
|
|
19
|
+
# Estrae il testo dall'HTML
|
|
20
|
+
soup = BeautifulSoup(html, "html.parser")
|
|
21
|
+
return soup.get_text(separator="\n")
|
|
22
|
+
|
|
23
|
+
register_extractor(
|
|
24
|
+
"text/markdown",
|
|
25
|
+
xtxt_md,
|
|
26
|
+
name="Markdown"
|
|
27
|
+
)
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
import tempfile
|
|
3
|
+
|
|
4
|
+
try:
|
|
5
|
+
import extract_msg
|
|
6
|
+
from bs4 import BeautifulSoup
|
|
7
|
+
except ImportError:
|
|
8
|
+
extract_msg = None
|
|
9
|
+
|
|
10
|
+
if extract_msg:
|
|
11
|
+
def xtxt_msg(file_buffer):
|
|
12
|
+
# Salva su file temporaneo perché extract_msg lavora su path
|
|
13
|
+
content = file_buffer.read()
|
|
14
|
+
|
|
15
|
+
with tempfile.NamedTemporaryFile(suffix=".msg", delete=False) as tmp:
|
|
16
|
+
tmp.write(content)
|
|
17
|
+
tmp.flush()
|
|
18
|
+
|
|
19
|
+
try:
|
|
20
|
+
msg = extract_msg.Message(tmp.name)
|
|
21
|
+
msg.extract() # Decodifica i contenuti
|
|
22
|
+
|
|
23
|
+
parts = []
|
|
24
|
+
|
|
25
|
+
if msg.body:
|
|
26
|
+
parts.append(msg.body)
|
|
27
|
+
|
|
28
|
+
if msg.htmlBody:
|
|
29
|
+
soup = BeautifulSoup(msg.htmlBody, "html.parser")
|
|
30
|
+
parts.append(soup.get_text(separator="\n"))
|
|
31
|
+
|
|
32
|
+
return "\n\n".join(part.strip() for part in parts if part)
|
|
33
|
+
finally:
|
|
34
|
+
import os
|
|
35
|
+
os.unlink(tmp.name)
|
|
36
|
+
|
|
37
|
+
register_extractor(
|
|
38
|
+
"application/vnd.ms-outlook",
|
|
39
|
+
xtxt_msg,
|
|
40
|
+
name="MSG"
|
|
41
|
+
)
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
from striprtf.striprtf import rtf_to_text
|
|
5
|
+
except ImportError:
|
|
6
|
+
rtf_to_text = None
|
|
7
|
+
|
|
8
|
+
if rtf_to_text:
|
|
9
|
+
def xtxt_rtf(file_buffer):
|
|
10
|
+
content = file_buffer.read()
|
|
11
|
+
if isinstance(content, bytes):
|
|
12
|
+
content = content.decode("utf-8", errors="ignore")
|
|
13
|
+
|
|
14
|
+
return rtf_to_text(content)
|
|
15
|
+
|
|
16
|
+
register_extractor(
|
|
17
|
+
"application/rtf",
|
|
18
|
+
xtxt_rtf,
|
|
19
|
+
name="RTF"
|
|
20
|
+
)
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
from pylatexenc.latex2text import LatexNodes2Text
|
|
5
|
+
except ImportError:
|
|
6
|
+
LatexNodes2Text = None
|
|
7
|
+
|
|
8
|
+
if LatexNodes2Text:
|
|
9
|
+
def xtxt_tex(file_buffer):
|
|
10
|
+
content = file_buffer.read()
|
|
11
|
+
if isinstance(content, bytes):
|
|
12
|
+
content = content.decode("utf-8", errors="ignore")
|
|
13
|
+
|
|
14
|
+
text = LatexNodes2Text().latex_to_text(content)
|
|
15
|
+
return text.strip()
|
|
16
|
+
|
|
17
|
+
register_extractor(
|
|
18
|
+
"application/x-tex",
|
|
19
|
+
xtxt_tex,
|
|
20
|
+
name="LaTeX"
|
|
21
|
+
)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.2
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -65,7 +65,7 @@ Requires-Dist: lxml; extra == "all"
|
|
|
65
65
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
66
66
|
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
67
67
|
|
|
68
|
-
**NEW in v0.1
|
|
68
|
+
**NEW in v0.2.1+**: Enhanced support for web content, byte streams, and requests integration!
|
|
69
69
|
|
|
70
70
|
---
|
|
71
71
|
|
|
@@ -205,13 +205,28 @@ text = xtxt(attachment_bytes)
|
|
|
205
205
|
|
|
206
206
|
## 📖 Full Examples
|
|
207
207
|
|
|
208
|
-
See [examples.py](
|
|
208
|
+
See [examples.py](./examples.py) for comprehensive usage examples including:
|
|
209
209
|
- Local file processing
|
|
210
210
|
- Memory buffer handling
|
|
211
211
|
- Web content extraction
|
|
212
212
|
- Error handling patterns
|
|
213
213
|
- All supported formats demonstration
|
|
214
214
|
|
|
215
|
+
### Accessing Examples After Installation
|
|
216
|
+
After installing PyxTxt from PyPI, you can access the examples file:
|
|
217
|
+
|
|
218
|
+
```python
|
|
219
|
+
import pkg_resources
|
|
220
|
+
|
|
221
|
+
# Get path to examples file
|
|
222
|
+
examples_path = pkg_resources.resource_filename('pyxtxt', 'examples.py')
|
|
223
|
+
print(f"Examples file location: {examples_path}")
|
|
224
|
+
|
|
225
|
+
# Or read the content directly
|
|
226
|
+
examples_content = pkg_resources.resource_string('pyxtxt', 'examples.py').decode('utf-8')
|
|
227
|
+
print(examples_content)
|
|
228
|
+
```
|
|
229
|
+
|
|
215
230
|
## 🔒 License
|
|
216
231
|
|
|
217
232
|
Distributed under the MIT License. See LICENSE file for details.
|
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
LICENCSE
|
|
2
|
+
MANIFEST.in
|
|
2
3
|
README.md
|
|
4
|
+
examples.py
|
|
3
5
|
pyproject.toml
|
|
4
6
|
src/pyxtxt/__init__.py
|
|
5
7
|
src/pyxtxt/core.py
|
|
@@ -12,11 +14,17 @@ src/pyxtxt.egg-info/top_level.txt
|
|
|
12
14
|
src/pyxtxt/estrattori/__init__.py
|
|
13
15
|
src/pyxtxt/estrattori/doc.py
|
|
14
16
|
src/pyxtxt/estrattori/docx.py
|
|
17
|
+
src/pyxtxt/estrattori/eml.py
|
|
18
|
+
src/pyxtxt/estrattori/epub.py
|
|
15
19
|
src/pyxtxt/estrattori/html.py
|
|
20
|
+
src/pyxtxt/estrattori/md.py
|
|
21
|
+
src/pyxtxt/estrattori/msg.py
|
|
16
22
|
src/pyxtxt/estrattori/odt.py
|
|
17
23
|
src/pyxtxt/estrattori/pdf.py
|
|
18
24
|
src/pyxtxt/estrattori/pptx.py
|
|
25
|
+
src/pyxtxt/estrattori/rtf.py
|
|
19
26
|
src/pyxtxt/estrattori/svg.py
|
|
27
|
+
src/pyxtxt/estrattori/tex.py
|
|
20
28
|
src/pyxtxt/estrattori/txt.py
|
|
21
29
|
src/pyxtxt/estrattori/xls.py
|
|
22
30
|
src/pyxtxt/estrattori/xlsx.py
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|