pyxtxt 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyxtxt-0.2.2/src/pyxtxt.egg-info → pyxtxt-0.2.3}/PKG-INFO +123 -21
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/README.md +80 -20
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/pyproject.toml +54 -2
- pyxtxt-0.2.3/src/pyxtxt/estrattori/audio.py +52 -0
- pyxtxt-0.2.3/src/pyxtxt/estrattori/ocr.py +52 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3/src/pyxtxt.egg-info}/PKG-INFO +123 -21
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/SOURCES.txt +3 -1
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/requires.txt +36 -0
- /pyxtxt-0.2.2/LICENCSE → /pyxtxt-0.2.3/LICENSE +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/MANIFEST.in +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/examples.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/setup.cfg +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/__init__.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/core.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/doc.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/docx.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/eml.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/epub.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/md.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/msg.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/odt.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/pptx.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/rtf.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/svg.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/tex.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/xlsx.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/pyxtxt.py +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -25,8 +25,21 @@ License: MIT License
|
|
|
25
25
|
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
26
|
SOFTWARE.
|
|
27
27
|
|
|
28
|
+
Classifier: Development Status :: 4 - Beta
|
|
29
|
+
Classifier: Intended Audience :: Developers
|
|
30
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
31
|
+
Classifier: Operating System :: OS Independent
|
|
32
|
+
Classifier: Programming Language :: Python :: 3
|
|
33
|
+
Classifier: Programming Language :: Python :: 3.7
|
|
34
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
35
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
36
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
38
|
+
Classifier: Topic :: Text Processing
|
|
39
|
+
Classifier: Topic :: Utilities
|
|
28
40
|
Requires-Python: >=3.7
|
|
29
41
|
Description-Content-Type: text/markdown
|
|
42
|
+
License-File: LICENSE
|
|
30
43
|
Requires-Dist: python-magic; sys_platform != "win32"
|
|
31
44
|
Requires-Dist: python-magic-bin; sys_platform == "win32"
|
|
32
45
|
Provides-Extra: pdf
|
|
@@ -45,6 +58,26 @@ Requires-Dist: beautifulsoup4; extra == "html"
|
|
|
45
58
|
Requires-Dist: lxml; extra == "html"
|
|
46
59
|
Provides-Extra: doc
|
|
47
60
|
Requires-Dist: textract; extra == "doc"
|
|
61
|
+
Provides-Extra: markdown
|
|
62
|
+
Requires-Dist: markdown; extra == "markdown"
|
|
63
|
+
Requires-Dist: beautifulsoup4; extra == "markdown"
|
|
64
|
+
Provides-Extra: epub
|
|
65
|
+
Requires-Dist: ebooklib; extra == "epub"
|
|
66
|
+
Requires-Dist: beautifulsoup4; extra == "epub"
|
|
67
|
+
Provides-Extra: rtf
|
|
68
|
+
Requires-Dist: striprtf; extra == "rtf"
|
|
69
|
+
Provides-Extra: email
|
|
70
|
+
Requires-Dist: beautifulsoup4; extra == "email"
|
|
71
|
+
Provides-Extra: outlook
|
|
72
|
+
Requires-Dist: extract-msg; extra == "outlook"
|
|
73
|
+
Requires-Dist: beautifulsoup4; extra == "outlook"
|
|
74
|
+
Provides-Extra: latex
|
|
75
|
+
Requires-Dist: pylatexenc; extra == "latex"
|
|
76
|
+
Provides-Extra: audio
|
|
77
|
+
Requires-Dist: openai-whisper; extra == "audio"
|
|
78
|
+
Provides-Extra: ocr
|
|
79
|
+
Requires-Dist: easyocr; extra == "ocr"
|
|
80
|
+
Requires-Dist: pillow; extra == "ocr"
|
|
48
81
|
Provides-Extra: all
|
|
49
82
|
Requires-Dist: textract; extra == "all"
|
|
50
83
|
Requires-Dist: PyMuPDF; extra == "all"
|
|
@@ -55,6 +88,15 @@ Requires-Dist: xlrd; extra == "all"
|
|
|
55
88
|
Requires-Dist: odfpy; extra == "all"
|
|
56
89
|
Requires-Dist: beautifulsoup4; extra == "all"
|
|
57
90
|
Requires-Dist: lxml; extra == "all"
|
|
91
|
+
Requires-Dist: markdown; extra == "all"
|
|
92
|
+
Requires-Dist: ebooklib; extra == "all"
|
|
93
|
+
Requires-Dist: striprtf; extra == "all"
|
|
94
|
+
Requires-Dist: extract-msg; extra == "all"
|
|
95
|
+
Requires-Dist: pylatexenc; extra == "all"
|
|
96
|
+
Requires-Dist: openai-whisper; extra == "all"
|
|
97
|
+
Requires-Dist: easyocr; extra == "all"
|
|
98
|
+
Requires-Dist: pillow; extra == "all"
|
|
99
|
+
Dynamic: license-file
|
|
58
100
|
|
|
59
101
|
# PyxTxt
|
|
60
102
|
|
|
@@ -63,16 +105,18 @@ Requires-Dist: lxml; extra == "all"
|
|
|
63
105
|
[](https://opensource.org/licenses/MIT)
|
|
64
106
|
|
|
65
107
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
66
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
108
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
|
|
67
109
|
|
|
68
|
-
**NEW in v0.2.
|
|
110
|
+
**NEW in v0.2.3+**: Added audio transcription (Whisper) and OCR from images (EasyOCR)!
|
|
69
111
|
|
|
70
112
|
---
|
|
71
113
|
|
|
72
114
|
## ✨ Features
|
|
73
115
|
|
|
74
116
|
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
75
|
-
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
|
|
117
|
+
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
|
|
118
|
+
- **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
|
|
119
|
+
- **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
|
|
76
120
|
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
77
121
|
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
78
122
|
- **Memory efficient**: Process files without saving to disk
|
|
@@ -89,8 +133,21 @@ pip install pyxtxt[all]
|
|
|
89
133
|
```
|
|
90
134
|
or just the modules you need:
|
|
91
135
|
```bash
|
|
92
|
-
pip install pyxtxt[pdf,
|
|
136
|
+
pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
|
|
93
137
|
```
|
|
138
|
+
|
|
139
|
+
### Audio & OCR (Heavy Dependencies)
|
|
140
|
+
```bash
|
|
141
|
+
# Audio transcription (~2GB download for Whisper models)
|
|
142
|
+
pip install pyxtxt[audio]
|
|
143
|
+
|
|
144
|
+
# OCR from images (~1GB download for EasyOCR models)
|
|
145
|
+
pip install pyxtxt[ocr]
|
|
146
|
+
|
|
147
|
+
# Both audio and OCR
|
|
148
|
+
pip install pyxtxt[audio,ocr]
|
|
149
|
+
```
|
|
150
|
+
|
|
94
151
|
Because needed libraries are common, installing the html module will also enable SVG and XML support.
|
|
95
152
|
The architecture is designed to grow with new modules for additional formats.
|
|
96
153
|
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
@@ -112,25 +169,24 @@ brew install libmagic
|
|
|
112
169
|
Use python-magic-bin instead of python-magic for easier installation.
|
|
113
170
|
|
|
114
171
|
## 🛠️ Dependencies
|
|
115
|
-
- PyMuPDF (fitz)
|
|
116
|
-
|
|
117
|
-
- beautifulsoup4
|
|
118
|
-
|
|
119
|
-
- python-docx
|
|
120
172
|
|
|
121
|
-
|
|
173
|
+
### Core Dependencies
|
|
174
|
+
- python-magic (automatic file type detection)
|
|
122
175
|
|
|
123
|
-
|
|
176
|
+
### Optional Dependencies by Format
|
|
177
|
+
- **PDF**: PyMuPDF
|
|
178
|
+
- **Office**: python-docx, python-pptx, openpyxl, xlrd
|
|
179
|
+
- **Web/HTML**: beautifulsoup4, lxml
|
|
180
|
+
- **OpenDocument**: odfpy
|
|
181
|
+
- **Markdown**: markdown
|
|
182
|
+
- **EPUB**: ebooklib
|
|
183
|
+
- **RTF**: striprtf
|
|
184
|
+
- **Email**: extract-msg (for MSG files)
|
|
185
|
+
- **LaTeX**: pylatexenc
|
|
186
|
+
- **Audio**: openai-whisper (heavy ~2GB models)
|
|
187
|
+
- **OCR**: easyocr, pillow (heavy ~1GB models)
|
|
124
188
|
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
- lxml
|
|
128
|
-
|
|
129
|
-
- xlrd (<2.0.0)
|
|
130
|
-
|
|
131
|
-
- python-magic
|
|
132
|
-
|
|
133
|
-
Dependencies are automatically installed from pyproject.toml.
|
|
189
|
+
Dependencies are automatically installed based on selected optional groups.
|
|
134
190
|
|
|
135
191
|
## 📚 Usage Examples
|
|
136
192
|
|
|
@@ -166,6 +222,36 @@ text = xtxt(response)
|
|
|
166
222
|
text = xtxt_from_url("https://example.com/document.pdf")
|
|
167
223
|
```
|
|
168
224
|
|
|
225
|
+
### Audio Transcription (NEW)
|
|
226
|
+
```python
|
|
227
|
+
from pyxtxt import xtxt
|
|
228
|
+
|
|
229
|
+
# Transcribe audio files
|
|
230
|
+
text = xtxt("meeting_recording.mp3")
|
|
231
|
+
text = xtxt("interview.wav")
|
|
232
|
+
text = xtxt("podcast.m4a")
|
|
233
|
+
|
|
234
|
+
# From web audio
|
|
235
|
+
import requests
|
|
236
|
+
audio_response = requests.get("https://example.com/audio.mp3")
|
|
237
|
+
text = xtxt(audio_response.content)
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
### OCR from Images (NEW)
|
|
241
|
+
```python
|
|
242
|
+
from pyxtxt import xtxt
|
|
243
|
+
|
|
244
|
+
# Extract text from images
|
|
245
|
+
text = xtxt("scanned_document.png")
|
|
246
|
+
text = xtxt("screenshot.jpg")
|
|
247
|
+
text = xtxt("invoice.tiff")
|
|
248
|
+
|
|
249
|
+
# From web images
|
|
250
|
+
import requests
|
|
251
|
+
image_response = requests.get("https://example.com/document.png")
|
|
252
|
+
text = xtxt(image_response.content)
|
|
253
|
+
```
|
|
254
|
+
|
|
169
255
|
### Show Available Formats
|
|
170
256
|
```python
|
|
171
257
|
from pyxtxt import extxt_available_formats
|
|
@@ -189,6 +275,14 @@ text = xtxt(api_response.content)
|
|
|
189
275
|
uploaded_bytes = request.files['document'].read()
|
|
190
276
|
text = xtxt(uploaded_bytes)
|
|
191
277
|
|
|
278
|
+
# Audio/video transcription services
|
|
279
|
+
audio_response = requests.get("https://api.example.com/recording.mp3")
|
|
280
|
+
transcript = xtxt(audio_response.content)
|
|
281
|
+
|
|
282
|
+
# OCR for uploaded images
|
|
283
|
+
image_bytes = request.files['receipt'].read()
|
|
284
|
+
text = xtxt(image_bytes)
|
|
285
|
+
|
|
192
286
|
# Email attachments
|
|
193
287
|
attachment_bytes = email_msg.get_payload(decode=True)
|
|
194
288
|
text = xtxt(attachment_bytes)
|
|
@@ -243,6 +337,14 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
243
337
|
|
|
244
338
|
## 📊 Changelog
|
|
245
339
|
|
|
340
|
+
### v0.2.3+
|
|
341
|
+
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
342
|
+
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
343
|
+
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
|
|
344
|
+
- ✅ Separate optional dependencies for heavy features (audio/OCR)
|
|
345
|
+
- ✅ Performance optimizations with model caching
|
|
346
|
+
- ✅ Improved multilingual OCR support (Italian/English)
|
|
347
|
+
|
|
246
348
|
### v0.1.24+
|
|
247
349
|
- ✅ Added support for `bytes` objects
|
|
248
350
|
- ✅ Added support for `requests.Response` objects
|
|
@@ -5,16 +5,18 @@
|
|
|
5
5
|
[](https://opensource.org/licenses/MIT)
|
|
6
6
|
|
|
7
7
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
8
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
8
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
|
|
9
9
|
|
|
10
|
-
**NEW in v0.2.
|
|
10
|
+
**NEW in v0.2.3+**: Added audio transcription (Whisper) and OCR from images (EasyOCR)!
|
|
11
11
|
|
|
12
12
|
---
|
|
13
13
|
|
|
14
14
|
## ✨ Features
|
|
15
15
|
|
|
16
16
|
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
17
|
-
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
|
|
17
|
+
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
|
|
18
|
+
- **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
|
|
19
|
+
- **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
|
|
18
20
|
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
19
21
|
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
20
22
|
- **Memory efficient**: Process files without saving to disk
|
|
@@ -31,8 +33,21 @@ pip install pyxtxt[all]
|
|
|
31
33
|
```
|
|
32
34
|
or just the modules you need:
|
|
33
35
|
```bash
|
|
34
|
-
pip install pyxtxt[pdf,
|
|
36
|
+
pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
|
|
35
37
|
```
|
|
38
|
+
|
|
39
|
+
### Audio & OCR (Heavy Dependencies)
|
|
40
|
+
```bash
|
|
41
|
+
# Audio transcription (~2GB download for Whisper models)
|
|
42
|
+
pip install pyxtxt[audio]
|
|
43
|
+
|
|
44
|
+
# OCR from images (~1GB download for EasyOCR models)
|
|
45
|
+
pip install pyxtxt[ocr]
|
|
46
|
+
|
|
47
|
+
# Both audio and OCR
|
|
48
|
+
pip install pyxtxt[audio,ocr]
|
|
49
|
+
```
|
|
50
|
+
|
|
36
51
|
Because needed libraries are common, installing the html module will also enable SVG and XML support.
|
|
37
52
|
The architecture is designed to grow with new modules for additional formats.
|
|
38
53
|
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
@@ -54,25 +69,24 @@ brew install libmagic
|
|
|
54
69
|
Use python-magic-bin instead of python-magic for easier installation.
|
|
55
70
|
|
|
56
71
|
## 🛠️ Dependencies
|
|
57
|
-
- PyMuPDF (fitz)
|
|
58
|
-
|
|
59
|
-
- beautifulsoup4
|
|
60
|
-
|
|
61
|
-
- python-docx
|
|
62
72
|
|
|
63
|
-
|
|
73
|
+
### Core Dependencies
|
|
74
|
+
- python-magic (automatic file type detection)
|
|
64
75
|
|
|
65
|
-
|
|
76
|
+
### Optional Dependencies by Format
|
|
77
|
+
- **PDF**: PyMuPDF
|
|
78
|
+
- **Office**: python-docx, python-pptx, openpyxl, xlrd
|
|
79
|
+
- **Web/HTML**: beautifulsoup4, lxml
|
|
80
|
+
- **OpenDocument**: odfpy
|
|
81
|
+
- **Markdown**: markdown
|
|
82
|
+
- **EPUB**: ebooklib
|
|
83
|
+
- **RTF**: striprtf
|
|
84
|
+
- **Email**: extract-msg (for MSG files)
|
|
85
|
+
- **LaTeX**: pylatexenc
|
|
86
|
+
- **Audio**: openai-whisper (heavy ~2GB models)
|
|
87
|
+
- **OCR**: easyocr, pillow (heavy ~1GB models)
|
|
66
88
|
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
- lxml
|
|
70
|
-
|
|
71
|
-
- xlrd (<2.0.0)
|
|
72
|
-
|
|
73
|
-
- python-magic
|
|
74
|
-
|
|
75
|
-
Dependencies are automatically installed from pyproject.toml.
|
|
89
|
+
Dependencies are automatically installed based on selected optional groups.
|
|
76
90
|
|
|
77
91
|
## 📚 Usage Examples
|
|
78
92
|
|
|
@@ -108,6 +122,36 @@ text = xtxt(response)
|
|
|
108
122
|
text = xtxt_from_url("https://example.com/document.pdf")
|
|
109
123
|
```
|
|
110
124
|
|
|
125
|
+
### Audio Transcription (NEW)
|
|
126
|
+
```python
|
|
127
|
+
from pyxtxt import xtxt
|
|
128
|
+
|
|
129
|
+
# Transcribe audio files
|
|
130
|
+
text = xtxt("meeting_recording.mp3")
|
|
131
|
+
text = xtxt("interview.wav")
|
|
132
|
+
text = xtxt("podcast.m4a")
|
|
133
|
+
|
|
134
|
+
# From web audio
|
|
135
|
+
import requests
|
|
136
|
+
audio_response = requests.get("https://example.com/audio.mp3")
|
|
137
|
+
text = xtxt(audio_response.content)
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
### OCR from Images (NEW)
|
|
141
|
+
```python
|
|
142
|
+
from pyxtxt import xtxt
|
|
143
|
+
|
|
144
|
+
# Extract text from images
|
|
145
|
+
text = xtxt("scanned_document.png")
|
|
146
|
+
text = xtxt("screenshot.jpg")
|
|
147
|
+
text = xtxt("invoice.tiff")
|
|
148
|
+
|
|
149
|
+
# From web images
|
|
150
|
+
import requests
|
|
151
|
+
image_response = requests.get("https://example.com/document.png")
|
|
152
|
+
text = xtxt(image_response.content)
|
|
153
|
+
```
|
|
154
|
+
|
|
111
155
|
### Show Available Formats
|
|
112
156
|
```python
|
|
113
157
|
from pyxtxt import extxt_available_formats
|
|
@@ -131,6 +175,14 @@ text = xtxt(api_response.content)
|
|
|
131
175
|
uploaded_bytes = request.files['document'].read()
|
|
132
176
|
text = xtxt(uploaded_bytes)
|
|
133
177
|
|
|
178
|
+
# Audio/video transcription services
|
|
179
|
+
audio_response = requests.get("https://api.example.com/recording.mp3")
|
|
180
|
+
transcript = xtxt(audio_response.content)
|
|
181
|
+
|
|
182
|
+
# OCR for uploaded images
|
|
183
|
+
image_bytes = request.files['receipt'].read()
|
|
184
|
+
text = xtxt(image_bytes)
|
|
185
|
+
|
|
134
186
|
# Email attachments
|
|
135
187
|
attachment_bytes = email_msg.get_payload(decode=True)
|
|
136
188
|
text = xtxt(attachment_bytes)
|
|
@@ -185,6 +237,14 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
185
237
|
|
|
186
238
|
## 📊 Changelog
|
|
187
239
|
|
|
240
|
+
### v0.2.3+
|
|
241
|
+
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
242
|
+
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
243
|
+
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
|
|
244
|
+
- ✅ Separate optional dependencies for heavy features (audio/OCR)
|
|
245
|
+
- ✅ Performance optimizations with model caching
|
|
246
|
+
- ✅ Improved multilingual OCR support (Italian/English)
|
|
247
|
+
|
|
188
248
|
### v0.1.24+
|
|
189
249
|
- ✅ Added support for `bytes` objects
|
|
190
250
|
- ✅ Added support for `requests.Response` objects
|
|
@@ -1,10 +1,26 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "pyxtxt"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.3"
|
|
4
4
|
description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
|
|
5
|
+
classifiers = [
|
|
6
|
+
"Development Status :: 4 - Beta",
|
|
7
|
+
"Intended Audience :: Developers",
|
|
8
|
+
"License :: OSI Approved :: MIT License",
|
|
9
|
+
"Operating System :: OS Independent",
|
|
10
|
+
"Programming Language :: Python :: 3",
|
|
11
|
+
"Programming Language :: Python :: 3.7",
|
|
12
|
+
"Programming Language :: Python :: 3.8",
|
|
13
|
+
"Programming Language :: Python :: 3.9",
|
|
14
|
+
"Programming Language :: Python :: 3.10",
|
|
15
|
+
"Programming Language :: Python :: 3.11",
|
|
16
|
+
"Topic :: Text Processing",
|
|
17
|
+
"Topic :: Utilities"
|
|
18
|
+
]
|
|
19
|
+
|
|
5
20
|
readme = "README.md"
|
|
6
21
|
requires-python = ">=3.7"
|
|
7
|
-
|
|
22
|
+
|
|
23
|
+
license = { file = "LICENSE" }
|
|
8
24
|
authors = [
|
|
9
25
|
{ name = "Giuseppe Levi", email = "giuseppe.levi@gmail.com" }
|
|
10
26
|
]
|
|
@@ -36,6 +52,34 @@ html = [
|
|
|
36
52
|
doc = [
|
|
37
53
|
"textract",
|
|
38
54
|
]
|
|
55
|
+
markdown = [
|
|
56
|
+
"markdown",
|
|
57
|
+
"beautifulsoup4",
|
|
58
|
+
]
|
|
59
|
+
epub = [
|
|
60
|
+
"ebooklib",
|
|
61
|
+
"beautifulsoup4",
|
|
62
|
+
]
|
|
63
|
+
rtf = [
|
|
64
|
+
"striprtf",
|
|
65
|
+
]
|
|
66
|
+
email = [
|
|
67
|
+
"beautifulsoup4",
|
|
68
|
+
]
|
|
69
|
+
outlook = [
|
|
70
|
+
"extract-msg",
|
|
71
|
+
"beautifulsoup4",
|
|
72
|
+
]
|
|
73
|
+
latex = [
|
|
74
|
+
"pylatexenc",
|
|
75
|
+
]
|
|
76
|
+
audio = [
|
|
77
|
+
"openai-whisper",
|
|
78
|
+
]
|
|
79
|
+
ocr = [
|
|
80
|
+
"easyocr",
|
|
81
|
+
"pillow",
|
|
82
|
+
]
|
|
39
83
|
all = [
|
|
40
84
|
"textract",
|
|
41
85
|
"PyMuPDF",
|
|
@@ -46,6 +90,14 @@ all = [
|
|
|
46
90
|
"odfpy",
|
|
47
91
|
"beautifulsoup4",
|
|
48
92
|
"lxml",
|
|
93
|
+
"markdown",
|
|
94
|
+
"ebooklib",
|
|
95
|
+
"striprtf",
|
|
96
|
+
"extract-msg",
|
|
97
|
+
"pylatexenc",
|
|
98
|
+
"openai-whisper",
|
|
99
|
+
"easyocr",
|
|
100
|
+
"pillow",
|
|
49
101
|
]
|
|
50
102
|
|
|
51
103
|
[build-system]
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# pyxtxt/extractors/audio_whisper.py
|
|
2
|
+
from . import register_extractor
|
|
3
|
+
import tempfile
|
|
4
|
+
import os
|
|
5
|
+
|
|
6
|
+
try:
|
|
7
|
+
import whisper
|
|
8
|
+
except ImportError:
|
|
9
|
+
whisper = None
|
|
10
|
+
|
|
11
|
+
if whisper:
|
|
12
|
+
_whisper_model = None
|
|
13
|
+
|
|
14
|
+
def _get_model():
|
|
15
|
+
global _whisper_model
|
|
16
|
+
if _whisper_model is None:
|
|
17
|
+
_whisper_model = whisper.load_model("base")
|
|
18
|
+
return _whisper_model
|
|
19
|
+
|
|
20
|
+
def xtxt_audio_whisper(file_buffer):
|
|
21
|
+
try:
|
|
22
|
+
# Usa un suffixe generico - Whisper + FFmpeg gestiscono il formato
|
|
23
|
+
with tempfile.NamedTemporaryFile(delete=False) as temp_file:
|
|
24
|
+
temp_file.write(file_buffer.read())
|
|
25
|
+
temp_file.flush()
|
|
26
|
+
|
|
27
|
+
model = _get_model()
|
|
28
|
+
result = model.transcribe(temp_file.name)
|
|
29
|
+
|
|
30
|
+
os.unlink(temp_file.name)
|
|
31
|
+
|
|
32
|
+
return result['text'].strip()
|
|
33
|
+
|
|
34
|
+
except Exception as e:
|
|
35
|
+
print(f"⚠️ Error while extracting audio with Whisper: {e}")
|
|
36
|
+
return ""
|
|
37
|
+
|
|
38
|
+
# Registra per tutti i formati audio comuni
|
|
39
|
+
audio_formats = [
|
|
40
|
+
"audio/wav", "audio/wave",
|
|
41
|
+
"audio/mp3", "audio/mpeg",
|
|
42
|
+
"audio/m4a", "audio/mp4",
|
|
43
|
+
"audio/flac",
|
|
44
|
+
"audio/ogg", "audio/ogg-vorbis",
|
|
45
|
+
"audio/opus",
|
|
46
|
+
"audio/aac",
|
|
47
|
+
"audio/wma",
|
|
48
|
+
"audio/webm"
|
|
49
|
+
]
|
|
50
|
+
|
|
51
|
+
for format_type in audio_formats:
|
|
52
|
+
register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio")
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# pyxtxt/extractors/image_ocr.py
|
|
2
|
+
from . import register_extractor
|
|
3
|
+
from io import BytesIO
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
import easyocr
|
|
7
|
+
from PIL import Image
|
|
8
|
+
except ImportError:
|
|
9
|
+
easyocr = None
|
|
10
|
+
Image = None
|
|
11
|
+
|
|
12
|
+
if easyocr and Image:
|
|
13
|
+
# Inizializza il reader OCR una volta sola
|
|
14
|
+
_ocr_reader = None
|
|
15
|
+
|
|
16
|
+
def _get_ocr_reader():
|
|
17
|
+
global _ocr_reader
|
|
18
|
+
if _ocr_reader is None:
|
|
19
|
+
_ocr_reader = easyocr.Reader(['it', 'en'], gpu=False)
|
|
20
|
+
return _ocr_reader
|
|
21
|
+
|
|
22
|
+
def xtxt_image_ocr(file_buffer):
|
|
23
|
+
try:
|
|
24
|
+
# Converti il buffer in immagine PIL
|
|
25
|
+
image = Image.open(BytesIO(file_buffer.read()))
|
|
26
|
+
|
|
27
|
+
# Converti in RGB se necessario
|
|
28
|
+
if image.mode != 'RGB':
|
|
29
|
+
image = image.convert('RGB')
|
|
30
|
+
|
|
31
|
+
reader = _get_ocr_reader()
|
|
32
|
+
results = reader.readtext(image)
|
|
33
|
+
|
|
34
|
+
# Estrai solo il testo, ordinato per posizione verticale
|
|
35
|
+
texts = []
|
|
36
|
+
for (bbox, text, confidence) in results:
|
|
37
|
+
if confidence > 0.3: # Filtra testo con bassa confidenza
|
|
38
|
+
texts.append(text)
|
|
39
|
+
|
|
40
|
+
return "\n".join(texts)
|
|
41
|
+
|
|
42
|
+
except Exception as e:
|
|
43
|
+
print(f"⚠️ Error while extracting text from image: {e}")
|
|
44
|
+
return ""
|
|
45
|
+
|
|
46
|
+
# Registra per i formati immagine più comuni
|
|
47
|
+
register_extractor("image/jpeg", xtxt_image_ocr, name="OCR")
|
|
48
|
+
register_extractor("image/jpg", xtxt_image_ocr, name="OCR")
|
|
49
|
+
register_extractor("image/png", xtxt_image_ocr, name="OCR")
|
|
50
|
+
register_extractor("image/bmp", xtxt_image_ocr, name="OCR")
|
|
51
|
+
register_extractor("image/tiff", xtxt_image_ocr, name="OCR")
|
|
52
|
+
register_extractor("image/webp", xtxt_image_ocr, name="OCR")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -25,8 +25,21 @@ License: MIT License
|
|
|
25
25
|
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
26
|
SOFTWARE.
|
|
27
27
|
|
|
28
|
+
Classifier: Development Status :: 4 - Beta
|
|
29
|
+
Classifier: Intended Audience :: Developers
|
|
30
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
31
|
+
Classifier: Operating System :: OS Independent
|
|
32
|
+
Classifier: Programming Language :: Python :: 3
|
|
33
|
+
Classifier: Programming Language :: Python :: 3.7
|
|
34
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
35
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
36
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
37
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
38
|
+
Classifier: Topic :: Text Processing
|
|
39
|
+
Classifier: Topic :: Utilities
|
|
28
40
|
Requires-Python: >=3.7
|
|
29
41
|
Description-Content-Type: text/markdown
|
|
42
|
+
License-File: LICENSE
|
|
30
43
|
Requires-Dist: python-magic; sys_platform != "win32"
|
|
31
44
|
Requires-Dist: python-magic-bin; sys_platform == "win32"
|
|
32
45
|
Provides-Extra: pdf
|
|
@@ -45,6 +58,26 @@ Requires-Dist: beautifulsoup4; extra == "html"
|
|
|
45
58
|
Requires-Dist: lxml; extra == "html"
|
|
46
59
|
Provides-Extra: doc
|
|
47
60
|
Requires-Dist: textract; extra == "doc"
|
|
61
|
+
Provides-Extra: markdown
|
|
62
|
+
Requires-Dist: markdown; extra == "markdown"
|
|
63
|
+
Requires-Dist: beautifulsoup4; extra == "markdown"
|
|
64
|
+
Provides-Extra: epub
|
|
65
|
+
Requires-Dist: ebooklib; extra == "epub"
|
|
66
|
+
Requires-Dist: beautifulsoup4; extra == "epub"
|
|
67
|
+
Provides-Extra: rtf
|
|
68
|
+
Requires-Dist: striprtf; extra == "rtf"
|
|
69
|
+
Provides-Extra: email
|
|
70
|
+
Requires-Dist: beautifulsoup4; extra == "email"
|
|
71
|
+
Provides-Extra: outlook
|
|
72
|
+
Requires-Dist: extract-msg; extra == "outlook"
|
|
73
|
+
Requires-Dist: beautifulsoup4; extra == "outlook"
|
|
74
|
+
Provides-Extra: latex
|
|
75
|
+
Requires-Dist: pylatexenc; extra == "latex"
|
|
76
|
+
Provides-Extra: audio
|
|
77
|
+
Requires-Dist: openai-whisper; extra == "audio"
|
|
78
|
+
Provides-Extra: ocr
|
|
79
|
+
Requires-Dist: easyocr; extra == "ocr"
|
|
80
|
+
Requires-Dist: pillow; extra == "ocr"
|
|
48
81
|
Provides-Extra: all
|
|
49
82
|
Requires-Dist: textract; extra == "all"
|
|
50
83
|
Requires-Dist: PyMuPDF; extra == "all"
|
|
@@ -55,6 +88,15 @@ Requires-Dist: xlrd; extra == "all"
|
|
|
55
88
|
Requires-Dist: odfpy; extra == "all"
|
|
56
89
|
Requires-Dist: beautifulsoup4; extra == "all"
|
|
57
90
|
Requires-Dist: lxml; extra == "all"
|
|
91
|
+
Requires-Dist: markdown; extra == "all"
|
|
92
|
+
Requires-Dist: ebooklib; extra == "all"
|
|
93
|
+
Requires-Dist: striprtf; extra == "all"
|
|
94
|
+
Requires-Dist: extract-msg; extra == "all"
|
|
95
|
+
Requires-Dist: pylatexenc; extra == "all"
|
|
96
|
+
Requires-Dist: openai-whisper; extra == "all"
|
|
97
|
+
Requires-Dist: easyocr; extra == "all"
|
|
98
|
+
Requires-Dist: pillow; extra == "all"
|
|
99
|
+
Dynamic: license-file
|
|
58
100
|
|
|
59
101
|
# PyxTxt
|
|
60
102
|
|
|
@@ -63,16 +105,18 @@ Requires-Dist: lxml; extra == "all"
|
|
|
63
105
|
[](https://opensource.org/licenses/MIT)
|
|
64
106
|
|
|
65
107
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
66
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
108
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
|
|
67
109
|
|
|
68
|
-
**NEW in v0.2.
|
|
110
|
+
**NEW in v0.2.3+**: Added audio transcription (Whisper) and OCR from images (EasyOCR)!
|
|
69
111
|
|
|
70
112
|
---
|
|
71
113
|
|
|
72
114
|
## ✨ Features
|
|
73
115
|
|
|
74
116
|
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
75
|
-
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
|
|
117
|
+
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
|
|
118
|
+
- **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
|
|
119
|
+
- **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
|
|
76
120
|
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
77
121
|
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
78
122
|
- **Memory efficient**: Process files without saving to disk
|
|
@@ -89,8 +133,21 @@ pip install pyxtxt[all]
|
|
|
89
133
|
```
|
|
90
134
|
or just the modules you need:
|
|
91
135
|
```bash
|
|
92
|
-
pip install pyxtxt[pdf,
|
|
136
|
+
pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
|
|
93
137
|
```
|
|
138
|
+
|
|
139
|
+
### Audio & OCR (Heavy Dependencies)
|
|
140
|
+
```bash
|
|
141
|
+
# Audio transcription (~2GB download for Whisper models)
|
|
142
|
+
pip install pyxtxt[audio]
|
|
143
|
+
|
|
144
|
+
# OCR from images (~1GB download for EasyOCR models)
|
|
145
|
+
pip install pyxtxt[ocr]
|
|
146
|
+
|
|
147
|
+
# Both audio and OCR
|
|
148
|
+
pip install pyxtxt[audio,ocr]
|
|
149
|
+
```
|
|
150
|
+
|
|
94
151
|
Because needed libraries are common, installing the html module will also enable SVG and XML support.
|
|
95
152
|
The architecture is designed to grow with new modules for additional formats.
|
|
96
153
|
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
@@ -112,25 +169,24 @@ brew install libmagic
|
|
|
112
169
|
Use python-magic-bin instead of python-magic for easier installation.
|
|
113
170
|
|
|
114
171
|
## 🛠️ Dependencies
|
|
115
|
-
- PyMuPDF (fitz)
|
|
116
|
-
|
|
117
|
-
- beautifulsoup4
|
|
118
|
-
|
|
119
|
-
- python-docx
|
|
120
172
|
|
|
121
|
-
|
|
173
|
+
### Core Dependencies
|
|
174
|
+
- python-magic (automatic file type detection)
|
|
122
175
|
|
|
123
|
-
|
|
176
|
+
### Optional Dependencies by Format
|
|
177
|
+
- **PDF**: PyMuPDF
|
|
178
|
+
- **Office**: python-docx, python-pptx, openpyxl, xlrd
|
|
179
|
+
- **Web/HTML**: beautifulsoup4, lxml
|
|
180
|
+
- **OpenDocument**: odfpy
|
|
181
|
+
- **Markdown**: markdown
|
|
182
|
+
- **EPUB**: ebooklib
|
|
183
|
+
- **RTF**: striprtf
|
|
184
|
+
- **Email**: extract-msg (for MSG files)
|
|
185
|
+
- **LaTeX**: pylatexenc
|
|
186
|
+
- **Audio**: openai-whisper (heavy ~2GB models)
|
|
187
|
+
- **OCR**: easyocr, pillow (heavy ~1GB models)
|
|
124
188
|
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
- lxml
|
|
128
|
-
|
|
129
|
-
- xlrd (<2.0.0)
|
|
130
|
-
|
|
131
|
-
- python-magic
|
|
132
|
-
|
|
133
|
-
Dependencies are automatically installed from pyproject.toml.
|
|
189
|
+
Dependencies are automatically installed based on selected optional groups.
|
|
134
190
|
|
|
135
191
|
## 📚 Usage Examples
|
|
136
192
|
|
|
@@ -166,6 +222,36 @@ text = xtxt(response)
|
|
|
166
222
|
text = xtxt_from_url("https://example.com/document.pdf")
|
|
167
223
|
```
|
|
168
224
|
|
|
225
|
+
### Audio Transcription (NEW)
|
|
226
|
+
```python
|
|
227
|
+
from pyxtxt import xtxt
|
|
228
|
+
|
|
229
|
+
# Transcribe audio files
|
|
230
|
+
text = xtxt("meeting_recording.mp3")
|
|
231
|
+
text = xtxt("interview.wav")
|
|
232
|
+
text = xtxt("podcast.m4a")
|
|
233
|
+
|
|
234
|
+
# From web audio
|
|
235
|
+
import requests
|
|
236
|
+
audio_response = requests.get("https://example.com/audio.mp3")
|
|
237
|
+
text = xtxt(audio_response.content)
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
### OCR from Images (NEW)
|
|
241
|
+
```python
|
|
242
|
+
from pyxtxt import xtxt
|
|
243
|
+
|
|
244
|
+
# Extract text from images
|
|
245
|
+
text = xtxt("scanned_document.png")
|
|
246
|
+
text = xtxt("screenshot.jpg")
|
|
247
|
+
text = xtxt("invoice.tiff")
|
|
248
|
+
|
|
249
|
+
# From web images
|
|
250
|
+
import requests
|
|
251
|
+
image_response = requests.get("https://example.com/document.png")
|
|
252
|
+
text = xtxt(image_response.content)
|
|
253
|
+
```
|
|
254
|
+
|
|
169
255
|
### Show Available Formats
|
|
170
256
|
```python
|
|
171
257
|
from pyxtxt import extxt_available_formats
|
|
@@ -189,6 +275,14 @@ text = xtxt(api_response.content)
|
|
|
189
275
|
uploaded_bytes = request.files['document'].read()
|
|
190
276
|
text = xtxt(uploaded_bytes)
|
|
191
277
|
|
|
278
|
+
# Audio/video transcription services
|
|
279
|
+
audio_response = requests.get("https://api.example.com/recording.mp3")
|
|
280
|
+
transcript = xtxt(audio_response.content)
|
|
281
|
+
|
|
282
|
+
# OCR for uploaded images
|
|
283
|
+
image_bytes = request.files['receipt'].read()
|
|
284
|
+
text = xtxt(image_bytes)
|
|
285
|
+
|
|
192
286
|
# Email attachments
|
|
193
287
|
attachment_bytes = email_msg.get_payload(decode=True)
|
|
194
288
|
text = xtxt(attachment_bytes)
|
|
@@ -243,6 +337,14 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
243
337
|
|
|
244
338
|
## 📊 Changelog
|
|
245
339
|
|
|
340
|
+
### v0.2.3+
|
|
341
|
+
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
342
|
+
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
343
|
+
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
|
|
344
|
+
- ✅ Separate optional dependencies for heavy features (audio/OCR)
|
|
345
|
+
- ✅ Performance optimizations with model caching
|
|
346
|
+
- ✅ Improved multilingual OCR support (Italian/English)
|
|
347
|
+
|
|
246
348
|
### v0.1.24+
|
|
247
349
|
- ✅ Added support for `bytes` objects
|
|
248
350
|
- ✅ Added support for `requests.Response` objects
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
|
|
1
|
+
LICENSE
|
|
2
2
|
MANIFEST.in
|
|
3
3
|
README.md
|
|
4
4
|
examples.py
|
|
@@ -12,6 +12,7 @@ src/pyxtxt.egg-info/dependency_links.txt
|
|
|
12
12
|
src/pyxtxt.egg-info/requires.txt
|
|
13
13
|
src/pyxtxt.egg-info/top_level.txt
|
|
14
14
|
src/pyxtxt/estrattori/__init__.py
|
|
15
|
+
src/pyxtxt/estrattori/audio.py
|
|
15
16
|
src/pyxtxt/estrattori/doc.py
|
|
16
17
|
src/pyxtxt/estrattori/docx.py
|
|
17
18
|
src/pyxtxt/estrattori/eml.py
|
|
@@ -19,6 +20,7 @@ src/pyxtxt/estrattori/epub.py
|
|
|
19
20
|
src/pyxtxt/estrattori/html.py
|
|
20
21
|
src/pyxtxt/estrattori/md.py
|
|
21
22
|
src/pyxtxt/estrattori/msg.py
|
|
23
|
+
src/pyxtxt/estrattori/ocr.py
|
|
22
24
|
src/pyxtxt/estrattori/odt.py
|
|
23
25
|
src/pyxtxt/estrattori/pdf.py
|
|
24
26
|
src/pyxtxt/estrattori/pptx.py
|
|
@@ -15,6 +15,17 @@ xlrd
|
|
|
15
15
|
odfpy
|
|
16
16
|
beautifulsoup4
|
|
17
17
|
lxml
|
|
18
|
+
markdown
|
|
19
|
+
ebooklib
|
|
20
|
+
striprtf
|
|
21
|
+
extract-msg
|
|
22
|
+
pylatexenc
|
|
23
|
+
openai-whisper
|
|
24
|
+
easyocr
|
|
25
|
+
pillow
|
|
26
|
+
|
|
27
|
+
[audio]
|
|
28
|
+
openai-whisper
|
|
18
29
|
|
|
19
30
|
[doc]
|
|
20
31
|
textract
|
|
@@ -22,19 +33,44 @@ textract
|
|
|
22
33
|
[docx]
|
|
23
34
|
python-docx
|
|
24
35
|
|
|
36
|
+
[email]
|
|
37
|
+
beautifulsoup4
|
|
38
|
+
|
|
39
|
+
[epub]
|
|
40
|
+
ebooklib
|
|
41
|
+
beautifulsoup4
|
|
42
|
+
|
|
25
43
|
[html]
|
|
26
44
|
beautifulsoup4
|
|
27
45
|
lxml
|
|
28
46
|
|
|
47
|
+
[latex]
|
|
48
|
+
pylatexenc
|
|
49
|
+
|
|
50
|
+
[markdown]
|
|
51
|
+
markdown
|
|
52
|
+
beautifulsoup4
|
|
53
|
+
|
|
54
|
+
[ocr]
|
|
55
|
+
easyocr
|
|
56
|
+
pillow
|
|
57
|
+
|
|
29
58
|
[odf]
|
|
30
59
|
odfpy
|
|
31
60
|
|
|
61
|
+
[outlook]
|
|
62
|
+
extract-msg
|
|
63
|
+
beautifulsoup4
|
|
64
|
+
|
|
32
65
|
[pdf]
|
|
33
66
|
PyMuPDF
|
|
34
67
|
|
|
35
68
|
[presentation]
|
|
36
69
|
python-pptx
|
|
37
70
|
|
|
71
|
+
[rtf]
|
|
72
|
+
striprtf
|
|
73
|
+
|
|
38
74
|
[spreadsheet]
|
|
39
75
|
openpyxl
|
|
40
76
|
xlrd
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|