pyxtxt 0.1.23__tar.gz → 0.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyxtxt-0.1.23/src/pyxtxt.egg-info → pyxtxt-0.2}/PKG-INFO +89 -36
- pyxtxt-0.2/README.md +178 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/pyproject.toml +1 -1
- pyxtxt-0.2/src/pyxtxt/__init__.py +1 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/core.py +57 -3
- pyxtxt-0.2/src/pyxtxt/estrattori/doc.py +26 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/xlsx.py +4 -3
- {pyxtxt-0.1.23 → pyxtxt-0.2/src/pyxtxt.egg-info}/PKG-INFO +89 -36
- pyxtxt-0.1.23/README.md +0 -125
- pyxtxt-0.1.23/src/pyxtxt/__init__.py +0 -1
- pyxtxt-0.1.23/src/pyxtxt/estrattori/doc.py +0 -41
- {pyxtxt-0.1.23 → pyxtxt-0.2}/LICENCSE +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/setup.cfg +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/docx.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/odt.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/pptx.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/svg.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt/pyxtxt.py +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt.egg-info/requires.txt +0 -0
- {pyxtxt-0.1.23 → pyxtxt-0.2}/src/pyxtxt.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -63,23 +63,26 @@ Requires-Dist: lxml; extra == "all"
|
|
|
63
63
|
[](https://opensource.org/licenses/MIT)
|
|
64
64
|
|
|
65
65
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
66
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy
|
|
66
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
67
|
+
|
|
68
|
+
**NEW in v0.1.24+**: Enhanced support for web content, byte streams, and requests integration!
|
|
67
69
|
|
|
68
70
|
---
|
|
69
71
|
|
|
70
72
|
## ✨ Features
|
|
71
73
|
|
|
72
|
-
-
|
|
73
|
-
-
|
|
74
|
-
-
|
|
75
|
-
-
|
|
76
|
-
-
|
|
74
|
+
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
75
|
+
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
|
|
76
|
+
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
77
|
+
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
78
|
+
- **Memory efficient**: Process files without saving to disk
|
|
79
|
+
- **Modern Python**: Full type hints and clean API design
|
|
77
80
|
|
|
78
81
|
---
|
|
79
82
|
|
|
80
83
|
## 📦 Installation
|
|
81
84
|
|
|
82
|
-
The library
|
|
85
|
+
The library is modular so you can install all modules:
|
|
83
86
|
|
|
84
87
|
```bash
|
|
85
88
|
pip install pyxtxt[all]
|
|
@@ -88,8 +91,8 @@ or just the modules you need:
|
|
|
88
91
|
```bash
|
|
89
92
|
pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
|
|
90
93
|
```
|
|
91
|
-
|
|
92
|
-
The architecture is designed to
|
|
94
|
+
Because needed libraries are common, installing the html module will also enable SVG and XML support.
|
|
95
|
+
The architecture is designed to grow with new modules for additional formats.
|
|
93
96
|
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
94
97
|
The pyproject.toml file should select the correct version for your system. But if you have any problem you can install it manually.
|
|
95
98
|
|
|
@@ -129,55 +132,105 @@ Use python-magic-bin instead of python-magic for easier installation.
|
|
|
129
132
|
|
|
130
133
|
Dependencies are automatically installed from pyproject.toml.
|
|
131
134
|
|
|
132
|
-
## 📚 Usage
|
|
133
|
-
Extract text from a file path:
|
|
135
|
+
## 📚 Usage Examples
|
|
134
136
|
|
|
137
|
+
### Basic Usage
|
|
135
138
|
```python
|
|
136
139
|
from pyxtxt import xtxt
|
|
137
140
|
|
|
141
|
+
# Extract from file path
|
|
138
142
|
text = xtxt("document.pdf")
|
|
139
143
|
print(text)
|
|
140
|
-
```
|
|
141
|
-
Extract text from a file-like buffer:
|
|
142
144
|
|
|
143
|
-
|
|
145
|
+
# Extract from BytesIO buffer
|
|
144
146
|
import io
|
|
145
|
-
|
|
146
147
|
with open("document.docx", "rb") as f:
|
|
147
148
|
buffer = io.BytesIO(f.read())
|
|
148
|
-
|
|
149
|
-
from pyxtxt import xtxt
|
|
150
149
|
text = xtxt(buffer)
|
|
151
150
|
print(text)
|
|
152
151
|
```
|
|
153
|
-
|
|
154
|
-
|
|
152
|
+
|
|
153
|
+
### NEW: Web Content Support
|
|
155
154
|
```python
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
#
|
|
160
|
-
|
|
161
|
-
|
|
155
|
+
import requests
|
|
156
|
+
from pyxtxt import xtxt, xtxt_from_url
|
|
157
|
+
|
|
158
|
+
# Method 1: Direct from bytes
|
|
159
|
+
response = requests.get("https://example.com/document.pdf")
|
|
160
|
+
text = xtxt(response.content)
|
|
161
|
+
|
|
162
|
+
# Method 2: Direct from Response object
|
|
163
|
+
text = xtxt(response)
|
|
164
|
+
|
|
165
|
+
# Method 3: URL helper function
|
|
166
|
+
text = xtxt_from_url("https://example.com/document.pdf")
|
|
162
167
|
```
|
|
163
|
-
## ⚠️ Known Limitations
|
|
164
|
-
When passing a raw stream (io.BytesIO) without a filename, legacy files (.doc, .xls, .ppt) may not be correctly detected.
|
|
165
168
|
|
|
166
|
-
|
|
167
|
-
|
|
169
|
+
### Show Available Formats
|
|
170
|
+
```python
|
|
171
|
+
from pyxtxt import extxt_available_formats
|
|
172
|
+
|
|
173
|
+
# List supported MIME types
|
|
174
|
+
formats = extxt_available_formats()
|
|
175
|
+
print(formats)
|
|
176
|
+
|
|
177
|
+
# Pretty format names
|
|
178
|
+
formats = extxt_available_formats(pretty=True)
|
|
179
|
+
print(formats)
|
|
180
|
+
```
|
|
181
|
+
## 🌐 Common Web Use Cases
|
|
168
182
|
|
|
169
|
-
|
|
183
|
+
```python
|
|
184
|
+
# API responses
|
|
185
|
+
api_response = requests.post("https://api.example.com/generate-pdf")
|
|
186
|
+
text = xtxt(api_response.content)
|
|
170
187
|
|
|
171
|
-
|
|
188
|
+
# File uploads (Flask/Django)
|
|
189
|
+
uploaded_bytes = request.files['document'].read()
|
|
190
|
+
text = xtxt(uploaded_bytes)
|
|
172
191
|
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
192
|
+
# Email attachments
|
|
193
|
+
attachment_bytes = email_msg.get_payload(decode=True)
|
|
194
|
+
text = xtxt(attachment_bytes)
|
|
176
195
|
```
|
|
177
196
|
|
|
197
|
+
## ⚠️ Known Limitations
|
|
198
|
+
|
|
199
|
+
- **Legacy file detection**: When using raw streams without filenames, legacy files (.doc, .xls, .ppt) may not be correctly detected due to identical file signatures in libmagic
|
|
200
|
+
- **Filename hints recommended**: When available, providing original filenames improves detection accuracy
|
|
201
|
+
- **MSWrite .doc files**: Require `antiword` installation:
|
|
202
|
+
```bash
|
|
203
|
+
sudo apt-get update && sudo apt-get install antiword
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
## 📖 Full Examples
|
|
207
|
+
|
|
208
|
+
See [examples.py](https://github.com/dede-amdp/pyxtxt/blob/main/examples.py) for comprehensive usage examples including:
|
|
209
|
+
- Local file processing
|
|
210
|
+
- Memory buffer handling
|
|
211
|
+
- Web content extraction
|
|
212
|
+
- Error handling patterns
|
|
213
|
+
- All supported formats demonstration
|
|
214
|
+
|
|
178
215
|
## 🔒 License
|
|
179
|
-
|
|
216
|
+
|
|
217
|
+
Distributed under the MIT License. See LICENSE file for details.
|
|
180
218
|
|
|
181
219
|
The software is provided "as is" without any warranty of any kind.
|
|
182
220
|
|
|
221
|
+
## 🤝 Contributing
|
|
222
|
+
|
|
183
223
|
Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
224
|
+
|
|
225
|
+
- **Bug reports**: Please include file samples and error details
|
|
226
|
+
- **Feature requests**: Describe your use case and expected behavior
|
|
227
|
+
- **Code contributions**: Follow existing patterns and add tests
|
|
228
|
+
|
|
229
|
+
## 📊 Changelog
|
|
230
|
+
|
|
231
|
+
### v0.1.24+
|
|
232
|
+
- ✅ Added support for `bytes` objects
|
|
233
|
+
- ✅ Added support for `requests.Response` objects
|
|
234
|
+
- ✅ Added `xtxt_from_url()` helper function
|
|
235
|
+
- ✅ Improved type hints and error handling
|
|
236
|
+
- ✅ Enhanced web content processing capabilities
|
pyxtxt-0.2/README.md
ADDED
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
# PyxTxt
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/pyxtxt/)
|
|
4
|
+
[](https://pypi.org/project/pyxtxt/)
|
|
5
|
+
[](https://opensource.org/licenses/MIT)
|
|
6
|
+
|
|
7
|
+
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
8
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
9
|
+
|
|
10
|
+
**NEW in v0.1.24+**: Enhanced support for web content, byte streams, and requests integration!
|
|
11
|
+
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
## ✨ Features
|
|
15
|
+
|
|
16
|
+
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
17
|
+
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
|
|
18
|
+
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
19
|
+
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
20
|
+
- **Memory efficient**: Process files without saving to disk
|
|
21
|
+
- **Modern Python**: Full type hints and clean API design
|
|
22
|
+
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
## 📦 Installation
|
|
26
|
+
|
|
27
|
+
The library is modular so you can install all modules:
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pip install pyxtxt[all]
|
|
31
|
+
```
|
|
32
|
+
or just the modules you need:
|
|
33
|
+
```bash
|
|
34
|
+
pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
|
|
35
|
+
```
|
|
36
|
+
Because needed libraries are common, installing the html module will also enable SVG and XML support.
|
|
37
|
+
The architecture is designed to grow with new modules for additional formats.
|
|
38
|
+
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
39
|
+
The pyproject.toml file should select the correct version for your system. But if you have any problem you can install it manually.
|
|
40
|
+
|
|
41
|
+
**On Ubuntu/Debian:**
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
sudo apt install libmagic1
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
**On Mac (Homebrew):**
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
brew install libmagic
|
|
51
|
+
```
|
|
52
|
+
**On Windows:**
|
|
53
|
+
|
|
54
|
+
Use python-magic-bin instead of python-magic for easier installation.
|
|
55
|
+
|
|
56
|
+
## 🛠️ Dependencies
|
|
57
|
+
- PyMuPDF (fitz)
|
|
58
|
+
|
|
59
|
+
- beautifulsoup4
|
|
60
|
+
|
|
61
|
+
- python-docx
|
|
62
|
+
|
|
63
|
+
- python-pptx
|
|
64
|
+
|
|
65
|
+
- odfpy
|
|
66
|
+
|
|
67
|
+
- openpyxl
|
|
68
|
+
|
|
69
|
+
- lxml
|
|
70
|
+
|
|
71
|
+
- xlrd (<2.0.0)
|
|
72
|
+
|
|
73
|
+
- python-magic
|
|
74
|
+
|
|
75
|
+
Dependencies are automatically installed from pyproject.toml.
|
|
76
|
+
|
|
77
|
+
## 📚 Usage Examples
|
|
78
|
+
|
|
79
|
+
### Basic Usage
|
|
80
|
+
```python
|
|
81
|
+
from pyxtxt import xtxt
|
|
82
|
+
|
|
83
|
+
# Extract from file path
|
|
84
|
+
text = xtxt("document.pdf")
|
|
85
|
+
print(text)
|
|
86
|
+
|
|
87
|
+
# Extract from BytesIO buffer
|
|
88
|
+
import io
|
|
89
|
+
with open("document.docx", "rb") as f:
|
|
90
|
+
buffer = io.BytesIO(f.read())
|
|
91
|
+
text = xtxt(buffer)
|
|
92
|
+
print(text)
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
### NEW: Web Content Support
|
|
96
|
+
```python
|
|
97
|
+
import requests
|
|
98
|
+
from pyxtxt import xtxt, xtxt_from_url
|
|
99
|
+
|
|
100
|
+
# Method 1: Direct from bytes
|
|
101
|
+
response = requests.get("https://example.com/document.pdf")
|
|
102
|
+
text = xtxt(response.content)
|
|
103
|
+
|
|
104
|
+
# Method 2: Direct from Response object
|
|
105
|
+
text = xtxt(response)
|
|
106
|
+
|
|
107
|
+
# Method 3: URL helper function
|
|
108
|
+
text = xtxt_from_url("https://example.com/document.pdf")
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
### Show Available Formats
|
|
112
|
+
```python
|
|
113
|
+
from pyxtxt import extxt_available_formats
|
|
114
|
+
|
|
115
|
+
# List supported MIME types
|
|
116
|
+
formats = extxt_available_formats()
|
|
117
|
+
print(formats)
|
|
118
|
+
|
|
119
|
+
# Pretty format names
|
|
120
|
+
formats = extxt_available_formats(pretty=True)
|
|
121
|
+
print(formats)
|
|
122
|
+
```
|
|
123
|
+
## 🌐 Common Web Use Cases
|
|
124
|
+
|
|
125
|
+
```python
|
|
126
|
+
# API responses
|
|
127
|
+
api_response = requests.post("https://api.example.com/generate-pdf")
|
|
128
|
+
text = xtxt(api_response.content)
|
|
129
|
+
|
|
130
|
+
# File uploads (Flask/Django)
|
|
131
|
+
uploaded_bytes = request.files['document'].read()
|
|
132
|
+
text = xtxt(uploaded_bytes)
|
|
133
|
+
|
|
134
|
+
# Email attachments
|
|
135
|
+
attachment_bytes = email_msg.get_payload(decode=True)
|
|
136
|
+
text = xtxt(attachment_bytes)
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
## ⚠️ Known Limitations
|
|
140
|
+
|
|
141
|
+
- **Legacy file detection**: When using raw streams without filenames, legacy files (.doc, .xls, .ppt) may not be correctly detected due to identical file signatures in libmagic
|
|
142
|
+
- **Filename hints recommended**: When available, providing original filenames improves detection accuracy
|
|
143
|
+
- **MSWrite .doc files**: Require `antiword` installation:
|
|
144
|
+
```bash
|
|
145
|
+
sudo apt-get update && sudo apt-get install antiword
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
## 📖 Full Examples
|
|
149
|
+
|
|
150
|
+
See [examples.py](https://github.com/dede-amdp/pyxtxt/blob/main/examples.py) for comprehensive usage examples including:
|
|
151
|
+
- Local file processing
|
|
152
|
+
- Memory buffer handling
|
|
153
|
+
- Web content extraction
|
|
154
|
+
- Error handling patterns
|
|
155
|
+
- All supported formats demonstration
|
|
156
|
+
|
|
157
|
+
## 🔒 License
|
|
158
|
+
|
|
159
|
+
Distributed under the MIT License. See LICENSE file for details.
|
|
160
|
+
|
|
161
|
+
The software is provided "as is" without any warranty of any kind.
|
|
162
|
+
|
|
163
|
+
## 🤝 Contributing
|
|
164
|
+
|
|
165
|
+
Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
166
|
+
|
|
167
|
+
- **Bug reports**: Please include file samples and error details
|
|
168
|
+
- **Feature requests**: Describe your use case and expected behavior
|
|
169
|
+
- **Code contributions**: Follow existing patterns and add tests
|
|
170
|
+
|
|
171
|
+
## 📊 Changelog
|
|
172
|
+
|
|
173
|
+
### v0.1.24+
|
|
174
|
+
- ✅ Added support for `bytes` objects
|
|
175
|
+
- ✅ Added support for `requests.Response` objects
|
|
176
|
+
- ✅ Added `xtxt_from_url()` helper function
|
|
177
|
+
- ✅ Improved type hints and error handling
|
|
178
|
+
- ✅ Enhanced web content processing capabilities
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from .core import xtxt, extxt_available_formats, xtxt_from_url
|
|
@@ -2,6 +2,8 @@ from .estrattori import estrattori
|
|
|
2
2
|
from functools import singledispatch
|
|
3
3
|
import io
|
|
4
4
|
import magic
|
|
5
|
+
from typing import Union, Optional
|
|
6
|
+
from urllib.parse import urlparse
|
|
5
7
|
|
|
6
8
|
@singledispatch
|
|
7
9
|
def xtxt(file_input):
|
|
@@ -9,7 +11,7 @@ def xtxt(file_input):
|
|
|
9
11
|
|
|
10
12
|
# Caso 1: file path (str)
|
|
11
13
|
@xtxt.register
|
|
12
|
-
def _(file_input: str):
|
|
14
|
+
def _(file_input: str) -> Optional[str]:
|
|
13
15
|
try:
|
|
14
16
|
with open(file_input, "rb") as f:
|
|
15
17
|
data = f.read()
|
|
@@ -23,7 +25,7 @@ def _(file_input: str):
|
|
|
23
25
|
|
|
24
26
|
# Caso 2: buffer (BytesIO)
|
|
25
27
|
@xtxt.register
|
|
26
|
-
def _(file_input: io.BytesIO):
|
|
28
|
+
def _(file_input: io.BytesIO) -> Optional[str]:
|
|
27
29
|
try:
|
|
28
30
|
|
|
29
31
|
|
|
@@ -47,7 +49,6 @@ def _(file_input: io.BytesIO):
|
|
|
47
49
|
mime_type = magic.Magic(mime=True).from_buffer(file_input.read(2048))
|
|
48
50
|
file_input.name='IO_buffer'
|
|
49
51
|
file_input.seek(0)
|
|
50
|
-
print(mime_type)
|
|
51
52
|
if mime_type.startswith("text/"):
|
|
52
53
|
if (mime_type != "text/html") and (mime_type != "text/xml") and (mime_type != "text/plain"):
|
|
53
54
|
print(f"📄 File recognized as text type: {mime_type}, treated as text/plain")
|
|
@@ -63,6 +64,59 @@ def _(file_input: io.BytesIO):
|
|
|
63
64
|
except Exception as e:
|
|
64
65
|
print(f"❌ Error while reading: {e}")
|
|
65
66
|
return None
|
|
67
|
+
|
|
68
|
+
# Caso 3: bytes object
|
|
69
|
+
@xtxt.register
|
|
70
|
+
def _(file_input: bytes) -> Optional[str]:
|
|
71
|
+
"""Estrae testo da oggetto bytes (es. da download web)"""
|
|
72
|
+
try:
|
|
73
|
+
buffer = io.BytesIO(file_input)
|
|
74
|
+
buffer.name = 'bytes_input'
|
|
75
|
+
mime_type = magic.Magic(mime=True).from_buffer(file_input[:2048])
|
|
76
|
+
buffer.mimeType = mime_type
|
|
77
|
+
return xtxt(buffer)
|
|
78
|
+
except Exception as e:
|
|
79
|
+
print(f"❌ Error processing bytes: {e}")
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
# Supporto per requests.Response (se disponibile)
|
|
83
|
+
try:
|
|
84
|
+
import requests
|
|
85
|
+
|
|
86
|
+
@xtxt.register
|
|
87
|
+
def _(file_input: requests.Response) -> Optional[str]:
|
|
88
|
+
"""Estrae testo da requests.Response object"""
|
|
89
|
+
try:
|
|
90
|
+
return xtxt(file_input.content)
|
|
91
|
+
except Exception as e:
|
|
92
|
+
print(f"❌ Error processing Response: {e}")
|
|
93
|
+
return None
|
|
94
|
+
except ImportError:
|
|
95
|
+
# requests non installato, skip registrazione
|
|
96
|
+
pass
|
|
97
|
+
|
|
98
|
+
def xtxt_from_url(url: str, **kwargs) -> Optional[str]:
|
|
99
|
+
"""Scarica contenuto da URL e ne estrae il testo
|
|
100
|
+
|
|
101
|
+
Args:
|
|
102
|
+
url: URL del documento da processare
|
|
103
|
+
**kwargs: Parametri aggiuntivi per requests.get()
|
|
104
|
+
|
|
105
|
+
Returns:
|
|
106
|
+
str: Testo estratto o None se errore
|
|
107
|
+
"""
|
|
108
|
+
try:
|
|
109
|
+
import requests
|
|
110
|
+
response = requests.get(url, **kwargs)
|
|
111
|
+
response.raise_for_status()
|
|
112
|
+
return xtxt(response.content)
|
|
113
|
+
except ImportError:
|
|
114
|
+
print("❌ requests library not installed. Install with: pip install requests")
|
|
115
|
+
return None
|
|
116
|
+
except Exception as e:
|
|
117
|
+
print(f"❌ Error downloading from URL {url}: {e}")
|
|
118
|
+
return None
|
|
119
|
+
|
|
66
120
|
def extxt_available_formats(pretty=False):
|
|
67
121
|
if pretty:
|
|
68
122
|
from .estrattori import pretty_names
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
|
|
2
|
+
from . import register_extractor
|
|
3
|
+
import shutil
|
|
4
|
+
import tempfile
|
|
5
|
+
import subprocess
|
|
6
|
+
def xtxt_doc(file_buffer):
|
|
7
|
+
if shutil.which("antiword") is None:
|
|
8
|
+
print("⚠️ 'antiword' is not installed or is not in the system PATH.")
|
|
9
|
+
return None
|
|
10
|
+
try:
|
|
11
|
+
file_buffer.seek(0)
|
|
12
|
+
data = file_buffer.read()
|
|
13
|
+
with tempfile.NamedTemporaryFile(suffix=".doc") as temp_file:
|
|
14
|
+
temp_file.write(data)
|
|
15
|
+
temp_file.flush()
|
|
16
|
+
temp_path = temp_file.name
|
|
17
|
+
result = subprocess.run(["antiword", temp_path],capture_output=True,text=True)
|
|
18
|
+
if result.returncode != 0:
|
|
19
|
+
print(f"⚠️ antiword failed: {result.stderr}")
|
|
20
|
+
return None
|
|
21
|
+
return result.stdout.strip()
|
|
22
|
+
except Exception as e:
|
|
23
|
+
print(f"⚠️ Error during extraction from DOC: {e}")
|
|
24
|
+
return None
|
|
25
|
+
|
|
26
|
+
register_extractor("application/msword",xtxt_doc,name="DOC")
|
|
@@ -41,10 +41,11 @@ if openpyxl:
|
|
|
41
41
|
count += 1
|
|
42
42
|
|
|
43
43
|
return "\n".join(testo)
|
|
44
|
-
|
|
45
|
-
|
|
44
|
+
|
|
45
|
+
register_extractor(
|
|
46
|
+
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
|
46
47
|
xtxt_xlsx,
|
|
47
48
|
name="XLSX"
|
|
48
|
-
)
|
|
49
|
+
)
|
|
49
50
|
|
|
50
51
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -63,23 +63,26 @@ Requires-Dist: lxml; extra == "all"
|
|
|
63
63
|
[](https://opensource.org/licenses/MIT)
|
|
64
64
|
|
|
65
65
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
66
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy
|
|
66
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
67
|
+
|
|
68
|
+
**NEW in v0.1.24+**: Enhanced support for web content, byte streams, and requests integration!
|
|
67
69
|
|
|
68
70
|
---
|
|
69
71
|
|
|
70
72
|
## ✨ Features
|
|
71
73
|
|
|
72
|
-
-
|
|
73
|
-
-
|
|
74
|
-
-
|
|
75
|
-
-
|
|
76
|
-
-
|
|
74
|
+
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
75
|
+
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
|
|
76
|
+
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
77
|
+
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
78
|
+
- **Memory efficient**: Process files without saving to disk
|
|
79
|
+
- **Modern Python**: Full type hints and clean API design
|
|
77
80
|
|
|
78
81
|
---
|
|
79
82
|
|
|
80
83
|
## 📦 Installation
|
|
81
84
|
|
|
82
|
-
The library
|
|
85
|
+
The library is modular so you can install all modules:
|
|
83
86
|
|
|
84
87
|
```bash
|
|
85
88
|
pip install pyxtxt[all]
|
|
@@ -88,8 +91,8 @@ or just the modules you need:
|
|
|
88
91
|
```bash
|
|
89
92
|
pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
|
|
90
93
|
```
|
|
91
|
-
|
|
92
|
-
The architecture is designed to
|
|
94
|
+
Because needed libraries are common, installing the html module will also enable SVG and XML support.
|
|
95
|
+
The architecture is designed to grow with new modules for additional formats.
|
|
93
96
|
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
94
97
|
The pyproject.toml file should select the correct version for your system. But if you have any problem you can install it manually.
|
|
95
98
|
|
|
@@ -129,55 +132,105 @@ Use python-magic-bin instead of python-magic for easier installation.
|
|
|
129
132
|
|
|
130
133
|
Dependencies are automatically installed from pyproject.toml.
|
|
131
134
|
|
|
132
|
-
## 📚 Usage
|
|
133
|
-
Extract text from a file path:
|
|
135
|
+
## 📚 Usage Examples
|
|
134
136
|
|
|
137
|
+
### Basic Usage
|
|
135
138
|
```python
|
|
136
139
|
from pyxtxt import xtxt
|
|
137
140
|
|
|
141
|
+
# Extract from file path
|
|
138
142
|
text = xtxt("document.pdf")
|
|
139
143
|
print(text)
|
|
140
|
-
```
|
|
141
|
-
Extract text from a file-like buffer:
|
|
142
144
|
|
|
143
|
-
|
|
145
|
+
# Extract from BytesIO buffer
|
|
144
146
|
import io
|
|
145
|
-
|
|
146
147
|
with open("document.docx", "rb") as f:
|
|
147
148
|
buffer = io.BytesIO(f.read())
|
|
148
|
-
|
|
149
|
-
from pyxtxt import xtxt
|
|
150
149
|
text = xtxt(buffer)
|
|
151
150
|
print(text)
|
|
152
151
|
```
|
|
153
|
-
|
|
154
|
-
|
|
152
|
+
|
|
153
|
+
### NEW: Web Content Support
|
|
155
154
|
```python
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
#
|
|
160
|
-
|
|
161
|
-
|
|
155
|
+
import requests
|
|
156
|
+
from pyxtxt import xtxt, xtxt_from_url
|
|
157
|
+
|
|
158
|
+
# Method 1: Direct from bytes
|
|
159
|
+
response = requests.get("https://example.com/document.pdf")
|
|
160
|
+
text = xtxt(response.content)
|
|
161
|
+
|
|
162
|
+
# Method 2: Direct from Response object
|
|
163
|
+
text = xtxt(response)
|
|
164
|
+
|
|
165
|
+
# Method 3: URL helper function
|
|
166
|
+
text = xtxt_from_url("https://example.com/document.pdf")
|
|
162
167
|
```
|
|
163
|
-
## ⚠️ Known Limitations
|
|
164
|
-
When passing a raw stream (io.BytesIO) without a filename, legacy files (.doc, .xls, .ppt) may not be correctly detected.
|
|
165
168
|
|
|
166
|
-
|
|
167
|
-
|
|
169
|
+
### Show Available Formats
|
|
170
|
+
```python
|
|
171
|
+
from pyxtxt import extxt_available_formats
|
|
172
|
+
|
|
173
|
+
# List supported MIME types
|
|
174
|
+
formats = extxt_available_formats()
|
|
175
|
+
print(formats)
|
|
176
|
+
|
|
177
|
+
# Pretty format names
|
|
178
|
+
formats = extxt_available_formats(pretty=True)
|
|
179
|
+
print(formats)
|
|
180
|
+
```
|
|
181
|
+
## 🌐 Common Web Use Cases
|
|
168
182
|
|
|
169
|
-
|
|
183
|
+
```python
|
|
184
|
+
# API responses
|
|
185
|
+
api_response = requests.post("https://api.example.com/generate-pdf")
|
|
186
|
+
text = xtxt(api_response.content)
|
|
170
187
|
|
|
171
|
-
|
|
188
|
+
# File uploads (Flask/Django)
|
|
189
|
+
uploaded_bytes = request.files['document'].read()
|
|
190
|
+
text = xtxt(uploaded_bytes)
|
|
172
191
|
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
192
|
+
# Email attachments
|
|
193
|
+
attachment_bytes = email_msg.get_payload(decode=True)
|
|
194
|
+
text = xtxt(attachment_bytes)
|
|
176
195
|
```
|
|
177
196
|
|
|
197
|
+
## ⚠️ Known Limitations
|
|
198
|
+
|
|
199
|
+
- **Legacy file detection**: When using raw streams without filenames, legacy files (.doc, .xls, .ppt) may not be correctly detected due to identical file signatures in libmagic
|
|
200
|
+
- **Filename hints recommended**: When available, providing original filenames improves detection accuracy
|
|
201
|
+
- **MSWrite .doc files**: Require `antiword` installation:
|
|
202
|
+
```bash
|
|
203
|
+
sudo apt-get update && sudo apt-get install antiword
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
## 📖 Full Examples
|
|
207
|
+
|
|
208
|
+
See [examples.py](https://github.com/dede-amdp/pyxtxt/blob/main/examples.py) for comprehensive usage examples including:
|
|
209
|
+
- Local file processing
|
|
210
|
+
- Memory buffer handling
|
|
211
|
+
- Web content extraction
|
|
212
|
+
- Error handling patterns
|
|
213
|
+
- All supported formats demonstration
|
|
214
|
+
|
|
178
215
|
## 🔒 License
|
|
179
|
-
|
|
216
|
+
|
|
217
|
+
Distributed under the MIT License. See LICENSE file for details.
|
|
180
218
|
|
|
181
219
|
The software is provided "as is" without any warranty of any kind.
|
|
182
220
|
|
|
221
|
+
## 🤝 Contributing
|
|
222
|
+
|
|
183
223
|
Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
224
|
+
|
|
225
|
+
- **Bug reports**: Please include file samples and error details
|
|
226
|
+
- **Feature requests**: Describe your use case and expected behavior
|
|
227
|
+
- **Code contributions**: Follow existing patterns and add tests
|
|
228
|
+
|
|
229
|
+
## 📊 Changelog
|
|
230
|
+
|
|
231
|
+
### v0.1.24+
|
|
232
|
+
- ✅ Added support for `bytes` objects
|
|
233
|
+
- ✅ Added support for `requests.Response` objects
|
|
234
|
+
- ✅ Added `xtxt_from_url()` helper function
|
|
235
|
+
- ✅ Improved type hints and error handling
|
|
236
|
+
- ✅ Enhanced web content processing capabilities
|
pyxtxt-0.1.23/README.md
DELETED
|
@@ -1,125 +0,0 @@
|
|
|
1
|
-
# PyxTxt
|
|
2
|
-
|
|
3
|
-
[](https://pypi.org/project/pyxtxt/)
|
|
4
|
-
[](https://pypi.org/project/pyxtxt/)
|
|
5
|
-
[](https://opensource.org/licenses/MIT)
|
|
6
|
-
|
|
7
|
-
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
8
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy XLS files, and more.
|
|
9
|
-
|
|
10
|
-
---
|
|
11
|
-
|
|
12
|
-
## ✨ Features
|
|
13
|
-
|
|
14
|
-
- Extracts text from both file paths and in-memory buffers (`io.BytesIO`).
|
|
15
|
-
- Supports multiple formats: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls,.ppt).
|
|
16
|
-
- Automatically detects MIME type using `python-magic`.
|
|
17
|
-
- Compatible with modern and legacy formats.
|
|
18
|
-
- Can handle streamed content without saving to disk (with some limitations).
|
|
19
|
-
|
|
20
|
-
---
|
|
21
|
-
|
|
22
|
-
## 📦 Installation
|
|
23
|
-
|
|
24
|
-
The library i modular so you can install all modules:
|
|
25
|
-
|
|
26
|
-
```bash
|
|
27
|
-
pip install pyxtxt[all]
|
|
28
|
-
```
|
|
29
|
-
or just the modules you need:
|
|
30
|
-
```bash
|
|
31
|
-
pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
|
|
32
|
-
```
|
|
33
|
-
Beause needed libraries are common installing the html module will enable also SVG and XML.
|
|
34
|
-
The architecture is designed to be able to grow with new modules to work with other formats as well.
|
|
35
|
-
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
36
|
-
The pyproject.toml file should select the correct version for your system. But if you have any problem you can install it manually.
|
|
37
|
-
|
|
38
|
-
**On Ubuntu/Debian:**
|
|
39
|
-
|
|
40
|
-
```bash
|
|
41
|
-
sudo apt install libmagic1
|
|
42
|
-
```
|
|
43
|
-
|
|
44
|
-
**On Mac (Homebrew):**
|
|
45
|
-
|
|
46
|
-
```bash
|
|
47
|
-
brew install libmagic
|
|
48
|
-
```
|
|
49
|
-
**On Windows:**
|
|
50
|
-
|
|
51
|
-
Use python-magic-bin instead of python-magic for easier installation.
|
|
52
|
-
|
|
53
|
-
## 🛠️ Dependencies
|
|
54
|
-
- PyMuPDF (fitz)
|
|
55
|
-
|
|
56
|
-
- beautifulsoup4
|
|
57
|
-
|
|
58
|
-
- python-docx
|
|
59
|
-
|
|
60
|
-
- python-pptx
|
|
61
|
-
|
|
62
|
-
- odfpy
|
|
63
|
-
|
|
64
|
-
- openpyxl
|
|
65
|
-
|
|
66
|
-
- lxml
|
|
67
|
-
|
|
68
|
-
- xlrd (<2.0.0)
|
|
69
|
-
|
|
70
|
-
- python-magic
|
|
71
|
-
|
|
72
|
-
Dependencies are automatically installed from pyproject.toml.
|
|
73
|
-
|
|
74
|
-
## 📚 Usage Example
|
|
75
|
-
Extract text from a file path:
|
|
76
|
-
|
|
77
|
-
```python
|
|
78
|
-
from pyxtxt import xtxt
|
|
79
|
-
|
|
80
|
-
text = xtxt("document.pdf")
|
|
81
|
-
print(text)
|
|
82
|
-
```
|
|
83
|
-
Extract text from a file-like buffer:
|
|
84
|
-
|
|
85
|
-
```python
|
|
86
|
-
import io
|
|
87
|
-
|
|
88
|
-
with open("document.docx", "rb") as f:
|
|
89
|
-
buffer = io.BytesIO(f.read())
|
|
90
|
-
|
|
91
|
-
from pyxtxt import xtxt
|
|
92
|
-
text = xtxt(buffer)
|
|
93
|
-
print(text)
|
|
94
|
-
```
|
|
95
|
-
Show available formats:
|
|
96
|
-
from pyxtxt import extxt_available_formats
|
|
97
|
-
```python
|
|
98
|
-
from pyxtxt import extxt_available_formats
|
|
99
|
-
text = extxt_available_formats()
|
|
100
|
-
print(text)
|
|
101
|
-
# For a pretty printing
|
|
102
|
-
text = extxt_available_formats(True)
|
|
103
|
-
print(text)
|
|
104
|
-
```
|
|
105
|
-
## ⚠️ Known Limitations
|
|
106
|
-
When passing a raw stream (io.BytesIO) without a filename, legacy files (.doc, .xls, .ppt) may not be correctly detected.
|
|
107
|
-
|
|
108
|
-
This is a limitation of libmagic beacuse the signature byte sequence at the start of doc/xls/ppt is exactly the same (b'\xD0\xCF\x11\xE0\xA1\xB1\x1A\xE1'),
|
|
109
|
-
not of pyxtxt.
|
|
110
|
-
|
|
111
|
-
If available, using the original filename is highly recommended.
|
|
112
|
-
|
|
113
|
-
To extract text from documents in MSWrite's old .doc format, it is necessary to install antiword.
|
|
114
|
-
|
|
115
|
-
```bash
|
|
116
|
-
sudo apt-get update
|
|
117
|
-
sudo apt-get -y install antiword
|
|
118
|
-
```
|
|
119
|
-
|
|
120
|
-
## 🔒 License
|
|
121
|
-
Distributed under the MIT License.
|
|
122
|
-
|
|
123
|
-
The software is provided "as is" without any warranty of any kind.
|
|
124
|
-
|
|
125
|
-
Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
from .core import xtxt, extxt_available_formats
|
|
@@ -1,41 +0,0 @@
|
|
|
1
|
-
|
|
2
|
-
from . import register_extractor
|
|
3
|
-
import shutil
|
|
4
|
-
import tempfile
|
|
5
|
-
try:
|
|
6
|
-
import textract
|
|
7
|
-
except ImportError:
|
|
8
|
-
textract = None
|
|
9
|
-
if textract:
|
|
10
|
-
def xtxt_doc(file_buffer):
|
|
11
|
-
if shutil.which("antiword") is None:
|
|
12
|
-
print("⚠️ 'antiword' is not installed or is not in the system PATH.")
|
|
13
|
-
return None
|
|
14
|
-
try:
|
|
15
|
-
file_buffer.seek(0)
|
|
16
|
-
data = file_buffer.read()
|
|
17
|
-
with tempfile.NamedTemporaryFile(suffix=".doc") as temp_file:
|
|
18
|
-
temp_file.write(data)
|
|
19
|
-
temp_file.flush()
|
|
20
|
-
testo = textract.process(temp_file.name)
|
|
21
|
-
return testo.decode("utf-8").strip()
|
|
22
|
-
except Exception as e:
|
|
23
|
-
print(f"⚠️ Error during extraction from DOC: {e}")
|
|
24
|
-
return None
|
|
25
|
-
|
|
26
|
-
register_extractor("application/msword",xtxt_doc,name="DOC")
|
|
27
|
-
'''
|
|
28
|
-
def xtxt_doc(file_buffer):
|
|
29
|
-
|
|
30
|
-
if shutil.which("antiword") is None:
|
|
31
|
-
print("⚠️ 'antiword' is not installed or is not in the system PATH.")
|
|
32
|
-
return None
|
|
33
|
-
try:
|
|
34
|
-
file_buffer.seek(0)
|
|
35
|
-
data = file_buffer.read()
|
|
36
|
-
testo = textract.process("temp.doc", input_data=data)
|
|
37
|
-
return testo.decode("utf-8").strip()
|
|
38
|
-
except Exception as e:
|
|
39
|
-
print(f"⚠️ Error during extraction from DOC: {e}")
|
|
40
|
-
return None
|
|
41
|
-
'''
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|