pyxtxt 0.2.2.1__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyxtxt-0.2.2.1/src/pyxtxt.egg-info → pyxtxt-0.2.3}/PKG-INFO +109 -21
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/README.md +80 -20
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/pyproject.toml +37 -1
- pyxtxt-0.2.3/src/pyxtxt/estrattori/audio.py +52 -0
- pyxtxt-0.2.3/src/pyxtxt/estrattori/ocr.py +52 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3/src/pyxtxt.egg-info}/PKG-INFO +109 -21
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/SOURCES.txt +2 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/requires.txt +36 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/LICENSE +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/MANIFEST.in +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/examples.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/setup.cfg +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/__init__.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/core.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/doc.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/docx.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/eml.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/epub.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/md.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/msg.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/odt.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/pptx.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/rtf.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/svg.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/tex.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/xlsx.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt/pyxtxt.py +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.2.2.1 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -58,6 +58,26 @@ Requires-Dist: beautifulsoup4; extra == "html"
|
|
|
58
58
|
Requires-Dist: lxml; extra == "html"
|
|
59
59
|
Provides-Extra: doc
|
|
60
60
|
Requires-Dist: textract; extra == "doc"
|
|
61
|
+
Provides-Extra: markdown
|
|
62
|
+
Requires-Dist: markdown; extra == "markdown"
|
|
63
|
+
Requires-Dist: beautifulsoup4; extra == "markdown"
|
|
64
|
+
Provides-Extra: epub
|
|
65
|
+
Requires-Dist: ebooklib; extra == "epub"
|
|
66
|
+
Requires-Dist: beautifulsoup4; extra == "epub"
|
|
67
|
+
Provides-Extra: rtf
|
|
68
|
+
Requires-Dist: striprtf; extra == "rtf"
|
|
69
|
+
Provides-Extra: email
|
|
70
|
+
Requires-Dist: beautifulsoup4; extra == "email"
|
|
71
|
+
Provides-Extra: outlook
|
|
72
|
+
Requires-Dist: extract-msg; extra == "outlook"
|
|
73
|
+
Requires-Dist: beautifulsoup4; extra == "outlook"
|
|
74
|
+
Provides-Extra: latex
|
|
75
|
+
Requires-Dist: pylatexenc; extra == "latex"
|
|
76
|
+
Provides-Extra: audio
|
|
77
|
+
Requires-Dist: openai-whisper; extra == "audio"
|
|
78
|
+
Provides-Extra: ocr
|
|
79
|
+
Requires-Dist: easyocr; extra == "ocr"
|
|
80
|
+
Requires-Dist: pillow; extra == "ocr"
|
|
61
81
|
Provides-Extra: all
|
|
62
82
|
Requires-Dist: textract; extra == "all"
|
|
63
83
|
Requires-Dist: PyMuPDF; extra == "all"
|
|
@@ -68,6 +88,14 @@ Requires-Dist: xlrd; extra == "all"
|
|
|
68
88
|
Requires-Dist: odfpy; extra == "all"
|
|
69
89
|
Requires-Dist: beautifulsoup4; extra == "all"
|
|
70
90
|
Requires-Dist: lxml; extra == "all"
|
|
91
|
+
Requires-Dist: markdown; extra == "all"
|
|
92
|
+
Requires-Dist: ebooklib; extra == "all"
|
|
93
|
+
Requires-Dist: striprtf; extra == "all"
|
|
94
|
+
Requires-Dist: extract-msg; extra == "all"
|
|
95
|
+
Requires-Dist: pylatexenc; extra == "all"
|
|
96
|
+
Requires-Dist: openai-whisper; extra == "all"
|
|
97
|
+
Requires-Dist: easyocr; extra == "all"
|
|
98
|
+
Requires-Dist: pillow; extra == "all"
|
|
71
99
|
Dynamic: license-file
|
|
72
100
|
|
|
73
101
|
# PyxTxt
|
|
@@ -77,16 +105,18 @@ Dynamic: license-file
|
|
|
77
105
|
[](https://opensource.org/licenses/MIT)
|
|
78
106
|
|
|
79
107
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
80
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
108
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
|
|
81
109
|
|
|
82
|
-
**NEW in v0.2.
|
|
110
|
+
**NEW in v0.2.3+**: Added audio transcription (Whisper) and OCR from images (EasyOCR)!
|
|
83
111
|
|
|
84
112
|
---
|
|
85
113
|
|
|
86
114
|
## ✨ Features
|
|
87
115
|
|
|
88
116
|
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
89
|
-
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
|
|
117
|
+
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
|
|
118
|
+
- **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
|
|
119
|
+
- **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
|
|
90
120
|
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
91
121
|
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
92
122
|
- **Memory efficient**: Process files without saving to disk
|
|
@@ -103,8 +133,21 @@ pip install pyxtxt[all]
|
|
|
103
133
|
```
|
|
104
134
|
or just the modules you need:
|
|
105
135
|
```bash
|
|
106
|
-
pip install pyxtxt[pdf,
|
|
136
|
+
pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
|
|
107
137
|
```
|
|
138
|
+
|
|
139
|
+
### Audio & OCR (Heavy Dependencies)
|
|
140
|
+
```bash
|
|
141
|
+
# Audio transcription (~2GB download for Whisper models)
|
|
142
|
+
pip install pyxtxt[audio]
|
|
143
|
+
|
|
144
|
+
# OCR from images (~1GB download for EasyOCR models)
|
|
145
|
+
pip install pyxtxt[ocr]
|
|
146
|
+
|
|
147
|
+
# Both audio and OCR
|
|
148
|
+
pip install pyxtxt[audio,ocr]
|
|
149
|
+
```
|
|
150
|
+
|
|
108
151
|
Because needed libraries are common, installing the html module will also enable SVG and XML support.
|
|
109
152
|
The architecture is designed to grow with new modules for additional formats.
|
|
110
153
|
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
@@ -126,25 +169,24 @@ brew install libmagic
|
|
|
126
169
|
Use python-magic-bin instead of python-magic for easier installation.
|
|
127
170
|
|
|
128
171
|
## 🛠️ Dependencies
|
|
129
|
-
- PyMuPDF (fitz)
|
|
130
|
-
|
|
131
|
-
- beautifulsoup4
|
|
132
|
-
|
|
133
|
-
- python-docx
|
|
134
172
|
|
|
135
|
-
|
|
173
|
+
### Core Dependencies
|
|
174
|
+
- python-magic (automatic file type detection)
|
|
136
175
|
|
|
137
|
-
|
|
176
|
+
### Optional Dependencies by Format
|
|
177
|
+
- **PDF**: PyMuPDF
|
|
178
|
+
- **Office**: python-docx, python-pptx, openpyxl, xlrd
|
|
179
|
+
- **Web/HTML**: beautifulsoup4, lxml
|
|
180
|
+
- **OpenDocument**: odfpy
|
|
181
|
+
- **Markdown**: markdown
|
|
182
|
+
- **EPUB**: ebooklib
|
|
183
|
+
- **RTF**: striprtf
|
|
184
|
+
- **Email**: extract-msg (for MSG files)
|
|
185
|
+
- **LaTeX**: pylatexenc
|
|
186
|
+
- **Audio**: openai-whisper (heavy ~2GB models)
|
|
187
|
+
- **OCR**: easyocr, pillow (heavy ~1GB models)
|
|
138
188
|
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
- lxml
|
|
142
|
-
|
|
143
|
-
- xlrd (<2.0.0)
|
|
144
|
-
|
|
145
|
-
- python-magic
|
|
146
|
-
|
|
147
|
-
Dependencies are automatically installed from pyproject.toml.
|
|
189
|
+
Dependencies are automatically installed based on selected optional groups.
|
|
148
190
|
|
|
149
191
|
## 📚 Usage Examples
|
|
150
192
|
|
|
@@ -180,6 +222,36 @@ text = xtxt(response)
|
|
|
180
222
|
text = xtxt_from_url("https://example.com/document.pdf")
|
|
181
223
|
```
|
|
182
224
|
|
|
225
|
+
### Audio Transcription (NEW)
|
|
226
|
+
```python
|
|
227
|
+
from pyxtxt import xtxt
|
|
228
|
+
|
|
229
|
+
# Transcribe audio files
|
|
230
|
+
text = xtxt("meeting_recording.mp3")
|
|
231
|
+
text = xtxt("interview.wav")
|
|
232
|
+
text = xtxt("podcast.m4a")
|
|
233
|
+
|
|
234
|
+
# From web audio
|
|
235
|
+
import requests
|
|
236
|
+
audio_response = requests.get("https://example.com/audio.mp3")
|
|
237
|
+
text = xtxt(audio_response.content)
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
### OCR from Images (NEW)
|
|
241
|
+
```python
|
|
242
|
+
from pyxtxt import xtxt
|
|
243
|
+
|
|
244
|
+
# Extract text from images
|
|
245
|
+
text = xtxt("scanned_document.png")
|
|
246
|
+
text = xtxt("screenshot.jpg")
|
|
247
|
+
text = xtxt("invoice.tiff")
|
|
248
|
+
|
|
249
|
+
# From web images
|
|
250
|
+
import requests
|
|
251
|
+
image_response = requests.get("https://example.com/document.png")
|
|
252
|
+
text = xtxt(image_response.content)
|
|
253
|
+
```
|
|
254
|
+
|
|
183
255
|
### Show Available Formats
|
|
184
256
|
```python
|
|
185
257
|
from pyxtxt import extxt_available_formats
|
|
@@ -203,6 +275,14 @@ text = xtxt(api_response.content)
|
|
|
203
275
|
uploaded_bytes = request.files['document'].read()
|
|
204
276
|
text = xtxt(uploaded_bytes)
|
|
205
277
|
|
|
278
|
+
# Audio/video transcription services
|
|
279
|
+
audio_response = requests.get("https://api.example.com/recording.mp3")
|
|
280
|
+
transcript = xtxt(audio_response.content)
|
|
281
|
+
|
|
282
|
+
# OCR for uploaded images
|
|
283
|
+
image_bytes = request.files['receipt'].read()
|
|
284
|
+
text = xtxt(image_bytes)
|
|
285
|
+
|
|
206
286
|
# Email attachments
|
|
207
287
|
attachment_bytes = email_msg.get_payload(decode=True)
|
|
208
288
|
text = xtxt(attachment_bytes)
|
|
@@ -257,6 +337,14 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
257
337
|
|
|
258
338
|
## 📊 Changelog
|
|
259
339
|
|
|
340
|
+
### v0.2.3+
|
|
341
|
+
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
342
|
+
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
343
|
+
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
|
|
344
|
+
- ✅ Separate optional dependencies for heavy features (audio/OCR)
|
|
345
|
+
- ✅ Performance optimizations with model caching
|
|
346
|
+
- ✅ Improved multilingual OCR support (Italian/English)
|
|
347
|
+
|
|
260
348
|
### v0.1.24+
|
|
261
349
|
- ✅ Added support for `bytes` objects
|
|
262
350
|
- ✅ Added support for `requests.Response` objects
|
|
@@ -5,16 +5,18 @@
|
|
|
5
5
|
[](https://opensource.org/licenses/MIT)
|
|
6
6
|
|
|
7
7
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
8
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
8
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
|
|
9
9
|
|
|
10
|
-
**NEW in v0.2.
|
|
10
|
+
**NEW in v0.2.3+**: Added audio transcription (Whisper) and OCR from images (EasyOCR)!
|
|
11
11
|
|
|
12
12
|
---
|
|
13
13
|
|
|
14
14
|
## ✨ Features
|
|
15
15
|
|
|
16
16
|
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
17
|
-
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
|
|
17
|
+
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
|
|
18
|
+
- **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
|
|
19
|
+
- **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
|
|
18
20
|
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
19
21
|
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
20
22
|
- **Memory efficient**: Process files without saving to disk
|
|
@@ -31,8 +33,21 @@ pip install pyxtxt[all]
|
|
|
31
33
|
```
|
|
32
34
|
or just the modules you need:
|
|
33
35
|
```bash
|
|
34
|
-
pip install pyxtxt[pdf,
|
|
36
|
+
pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
|
|
35
37
|
```
|
|
38
|
+
|
|
39
|
+
### Audio & OCR (Heavy Dependencies)
|
|
40
|
+
```bash
|
|
41
|
+
# Audio transcription (~2GB download for Whisper models)
|
|
42
|
+
pip install pyxtxt[audio]
|
|
43
|
+
|
|
44
|
+
# OCR from images (~1GB download for EasyOCR models)
|
|
45
|
+
pip install pyxtxt[ocr]
|
|
46
|
+
|
|
47
|
+
# Both audio and OCR
|
|
48
|
+
pip install pyxtxt[audio,ocr]
|
|
49
|
+
```
|
|
50
|
+
|
|
36
51
|
Because needed libraries are common, installing the html module will also enable SVG and XML support.
|
|
37
52
|
The architecture is designed to grow with new modules for additional formats.
|
|
38
53
|
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
@@ -54,25 +69,24 @@ brew install libmagic
|
|
|
54
69
|
Use python-magic-bin instead of python-magic for easier installation.
|
|
55
70
|
|
|
56
71
|
## 🛠️ Dependencies
|
|
57
|
-
- PyMuPDF (fitz)
|
|
58
|
-
|
|
59
|
-
- beautifulsoup4
|
|
60
|
-
|
|
61
|
-
- python-docx
|
|
62
72
|
|
|
63
|
-
|
|
73
|
+
### Core Dependencies
|
|
74
|
+
- python-magic (automatic file type detection)
|
|
64
75
|
|
|
65
|
-
|
|
76
|
+
### Optional Dependencies by Format
|
|
77
|
+
- **PDF**: PyMuPDF
|
|
78
|
+
- **Office**: python-docx, python-pptx, openpyxl, xlrd
|
|
79
|
+
- **Web/HTML**: beautifulsoup4, lxml
|
|
80
|
+
- **OpenDocument**: odfpy
|
|
81
|
+
- **Markdown**: markdown
|
|
82
|
+
- **EPUB**: ebooklib
|
|
83
|
+
- **RTF**: striprtf
|
|
84
|
+
- **Email**: extract-msg (for MSG files)
|
|
85
|
+
- **LaTeX**: pylatexenc
|
|
86
|
+
- **Audio**: openai-whisper (heavy ~2GB models)
|
|
87
|
+
- **OCR**: easyocr, pillow (heavy ~1GB models)
|
|
66
88
|
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
- lxml
|
|
70
|
-
|
|
71
|
-
- xlrd (<2.0.0)
|
|
72
|
-
|
|
73
|
-
- python-magic
|
|
74
|
-
|
|
75
|
-
Dependencies are automatically installed from pyproject.toml.
|
|
89
|
+
Dependencies are automatically installed based on selected optional groups.
|
|
76
90
|
|
|
77
91
|
## 📚 Usage Examples
|
|
78
92
|
|
|
@@ -108,6 +122,36 @@ text = xtxt(response)
|
|
|
108
122
|
text = xtxt_from_url("https://example.com/document.pdf")
|
|
109
123
|
```
|
|
110
124
|
|
|
125
|
+
### Audio Transcription (NEW)
|
|
126
|
+
```python
|
|
127
|
+
from pyxtxt import xtxt
|
|
128
|
+
|
|
129
|
+
# Transcribe audio files
|
|
130
|
+
text = xtxt("meeting_recording.mp3")
|
|
131
|
+
text = xtxt("interview.wav")
|
|
132
|
+
text = xtxt("podcast.m4a")
|
|
133
|
+
|
|
134
|
+
# From web audio
|
|
135
|
+
import requests
|
|
136
|
+
audio_response = requests.get("https://example.com/audio.mp3")
|
|
137
|
+
text = xtxt(audio_response.content)
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
### OCR from Images (NEW)
|
|
141
|
+
```python
|
|
142
|
+
from pyxtxt import xtxt
|
|
143
|
+
|
|
144
|
+
# Extract text from images
|
|
145
|
+
text = xtxt("scanned_document.png")
|
|
146
|
+
text = xtxt("screenshot.jpg")
|
|
147
|
+
text = xtxt("invoice.tiff")
|
|
148
|
+
|
|
149
|
+
# From web images
|
|
150
|
+
import requests
|
|
151
|
+
image_response = requests.get("https://example.com/document.png")
|
|
152
|
+
text = xtxt(image_response.content)
|
|
153
|
+
```
|
|
154
|
+
|
|
111
155
|
### Show Available Formats
|
|
112
156
|
```python
|
|
113
157
|
from pyxtxt import extxt_available_formats
|
|
@@ -131,6 +175,14 @@ text = xtxt(api_response.content)
|
|
|
131
175
|
uploaded_bytes = request.files['document'].read()
|
|
132
176
|
text = xtxt(uploaded_bytes)
|
|
133
177
|
|
|
178
|
+
# Audio/video transcription services
|
|
179
|
+
audio_response = requests.get("https://api.example.com/recording.mp3")
|
|
180
|
+
transcript = xtxt(audio_response.content)
|
|
181
|
+
|
|
182
|
+
# OCR for uploaded images
|
|
183
|
+
image_bytes = request.files['receipt'].read()
|
|
184
|
+
text = xtxt(image_bytes)
|
|
185
|
+
|
|
134
186
|
# Email attachments
|
|
135
187
|
attachment_bytes = email_msg.get_payload(decode=True)
|
|
136
188
|
text = xtxt(attachment_bytes)
|
|
@@ -185,6 +237,14 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
185
237
|
|
|
186
238
|
## 📊 Changelog
|
|
187
239
|
|
|
240
|
+
### v0.2.3+
|
|
241
|
+
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
242
|
+
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
243
|
+
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
|
|
244
|
+
- ✅ Separate optional dependencies for heavy features (audio/OCR)
|
|
245
|
+
- ✅ Performance optimizations with model caching
|
|
246
|
+
- ✅ Improved multilingual OCR support (Italian/English)
|
|
247
|
+
|
|
188
248
|
### v0.1.24+
|
|
189
249
|
- ✅ Added support for `bytes` objects
|
|
190
250
|
- ✅ Added support for `requests.Response` objects
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "pyxtxt"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.3"
|
|
4
4
|
description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
|
|
5
5
|
classifiers = [
|
|
6
6
|
"Development Status :: 4 - Beta",
|
|
@@ -52,6 +52,34 @@ html = [
|
|
|
52
52
|
doc = [
|
|
53
53
|
"textract",
|
|
54
54
|
]
|
|
55
|
+
markdown = [
|
|
56
|
+
"markdown",
|
|
57
|
+
"beautifulsoup4",
|
|
58
|
+
]
|
|
59
|
+
epub = [
|
|
60
|
+
"ebooklib",
|
|
61
|
+
"beautifulsoup4",
|
|
62
|
+
]
|
|
63
|
+
rtf = [
|
|
64
|
+
"striprtf",
|
|
65
|
+
]
|
|
66
|
+
email = [
|
|
67
|
+
"beautifulsoup4",
|
|
68
|
+
]
|
|
69
|
+
outlook = [
|
|
70
|
+
"extract-msg",
|
|
71
|
+
"beautifulsoup4",
|
|
72
|
+
]
|
|
73
|
+
latex = [
|
|
74
|
+
"pylatexenc",
|
|
75
|
+
]
|
|
76
|
+
audio = [
|
|
77
|
+
"openai-whisper",
|
|
78
|
+
]
|
|
79
|
+
ocr = [
|
|
80
|
+
"easyocr",
|
|
81
|
+
"pillow",
|
|
82
|
+
]
|
|
55
83
|
all = [
|
|
56
84
|
"textract",
|
|
57
85
|
"PyMuPDF",
|
|
@@ -62,6 +90,14 @@ all = [
|
|
|
62
90
|
"odfpy",
|
|
63
91
|
"beautifulsoup4",
|
|
64
92
|
"lxml",
|
|
93
|
+
"markdown",
|
|
94
|
+
"ebooklib",
|
|
95
|
+
"striprtf",
|
|
96
|
+
"extract-msg",
|
|
97
|
+
"pylatexenc",
|
|
98
|
+
"openai-whisper",
|
|
99
|
+
"easyocr",
|
|
100
|
+
"pillow",
|
|
65
101
|
]
|
|
66
102
|
|
|
67
103
|
[build-system]
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# pyxtxt/extractors/audio_whisper.py
|
|
2
|
+
from . import register_extractor
|
|
3
|
+
import tempfile
|
|
4
|
+
import os
|
|
5
|
+
|
|
6
|
+
try:
|
|
7
|
+
import whisper
|
|
8
|
+
except ImportError:
|
|
9
|
+
whisper = None
|
|
10
|
+
|
|
11
|
+
if whisper:
|
|
12
|
+
_whisper_model = None
|
|
13
|
+
|
|
14
|
+
def _get_model():
|
|
15
|
+
global _whisper_model
|
|
16
|
+
if _whisper_model is None:
|
|
17
|
+
_whisper_model = whisper.load_model("base")
|
|
18
|
+
return _whisper_model
|
|
19
|
+
|
|
20
|
+
def xtxt_audio_whisper(file_buffer):
|
|
21
|
+
try:
|
|
22
|
+
# Usa un suffixe generico - Whisper + FFmpeg gestiscono il formato
|
|
23
|
+
with tempfile.NamedTemporaryFile(delete=False) as temp_file:
|
|
24
|
+
temp_file.write(file_buffer.read())
|
|
25
|
+
temp_file.flush()
|
|
26
|
+
|
|
27
|
+
model = _get_model()
|
|
28
|
+
result = model.transcribe(temp_file.name)
|
|
29
|
+
|
|
30
|
+
os.unlink(temp_file.name)
|
|
31
|
+
|
|
32
|
+
return result['text'].strip()
|
|
33
|
+
|
|
34
|
+
except Exception as e:
|
|
35
|
+
print(f"⚠️ Error while extracting audio with Whisper: {e}")
|
|
36
|
+
return ""
|
|
37
|
+
|
|
38
|
+
# Registra per tutti i formati audio comuni
|
|
39
|
+
audio_formats = [
|
|
40
|
+
"audio/wav", "audio/wave",
|
|
41
|
+
"audio/mp3", "audio/mpeg",
|
|
42
|
+
"audio/m4a", "audio/mp4",
|
|
43
|
+
"audio/flac",
|
|
44
|
+
"audio/ogg", "audio/ogg-vorbis",
|
|
45
|
+
"audio/opus",
|
|
46
|
+
"audio/aac",
|
|
47
|
+
"audio/wma",
|
|
48
|
+
"audio/webm"
|
|
49
|
+
]
|
|
50
|
+
|
|
51
|
+
for format_type in audio_formats:
|
|
52
|
+
register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio")
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# pyxtxt/extractors/image_ocr.py
|
|
2
|
+
from . import register_extractor
|
|
3
|
+
from io import BytesIO
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
import easyocr
|
|
7
|
+
from PIL import Image
|
|
8
|
+
except ImportError:
|
|
9
|
+
easyocr = None
|
|
10
|
+
Image = None
|
|
11
|
+
|
|
12
|
+
if easyocr and Image:
|
|
13
|
+
# Inizializza il reader OCR una volta sola
|
|
14
|
+
_ocr_reader = None
|
|
15
|
+
|
|
16
|
+
def _get_ocr_reader():
|
|
17
|
+
global _ocr_reader
|
|
18
|
+
if _ocr_reader is None:
|
|
19
|
+
_ocr_reader = easyocr.Reader(['it', 'en'], gpu=False)
|
|
20
|
+
return _ocr_reader
|
|
21
|
+
|
|
22
|
+
def xtxt_image_ocr(file_buffer):
|
|
23
|
+
try:
|
|
24
|
+
# Converti il buffer in immagine PIL
|
|
25
|
+
image = Image.open(BytesIO(file_buffer.read()))
|
|
26
|
+
|
|
27
|
+
# Converti in RGB se necessario
|
|
28
|
+
if image.mode != 'RGB':
|
|
29
|
+
image = image.convert('RGB')
|
|
30
|
+
|
|
31
|
+
reader = _get_ocr_reader()
|
|
32
|
+
results = reader.readtext(image)
|
|
33
|
+
|
|
34
|
+
# Estrai solo il testo, ordinato per posizione verticale
|
|
35
|
+
texts = []
|
|
36
|
+
for (bbox, text, confidence) in results:
|
|
37
|
+
if confidence > 0.3: # Filtra testo con bassa confidenza
|
|
38
|
+
texts.append(text)
|
|
39
|
+
|
|
40
|
+
return "\n".join(texts)
|
|
41
|
+
|
|
42
|
+
except Exception as e:
|
|
43
|
+
print(f"⚠️ Error while extracting text from image: {e}")
|
|
44
|
+
return ""
|
|
45
|
+
|
|
46
|
+
# Registra per i formati immagine più comuni
|
|
47
|
+
register_extractor("image/jpeg", xtxt_image_ocr, name="OCR")
|
|
48
|
+
register_extractor("image/jpg", xtxt_image_ocr, name="OCR")
|
|
49
|
+
register_extractor("image/png", xtxt_image_ocr, name="OCR")
|
|
50
|
+
register_extractor("image/bmp", xtxt_image_ocr, name="OCR")
|
|
51
|
+
register_extractor("image/tiff", xtxt_image_ocr, name="OCR")
|
|
52
|
+
register_extractor("image/webp", xtxt_image_ocr, name="OCR")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -58,6 +58,26 @@ Requires-Dist: beautifulsoup4; extra == "html"
|
|
|
58
58
|
Requires-Dist: lxml; extra == "html"
|
|
59
59
|
Provides-Extra: doc
|
|
60
60
|
Requires-Dist: textract; extra == "doc"
|
|
61
|
+
Provides-Extra: markdown
|
|
62
|
+
Requires-Dist: markdown; extra == "markdown"
|
|
63
|
+
Requires-Dist: beautifulsoup4; extra == "markdown"
|
|
64
|
+
Provides-Extra: epub
|
|
65
|
+
Requires-Dist: ebooklib; extra == "epub"
|
|
66
|
+
Requires-Dist: beautifulsoup4; extra == "epub"
|
|
67
|
+
Provides-Extra: rtf
|
|
68
|
+
Requires-Dist: striprtf; extra == "rtf"
|
|
69
|
+
Provides-Extra: email
|
|
70
|
+
Requires-Dist: beautifulsoup4; extra == "email"
|
|
71
|
+
Provides-Extra: outlook
|
|
72
|
+
Requires-Dist: extract-msg; extra == "outlook"
|
|
73
|
+
Requires-Dist: beautifulsoup4; extra == "outlook"
|
|
74
|
+
Provides-Extra: latex
|
|
75
|
+
Requires-Dist: pylatexenc; extra == "latex"
|
|
76
|
+
Provides-Extra: audio
|
|
77
|
+
Requires-Dist: openai-whisper; extra == "audio"
|
|
78
|
+
Provides-Extra: ocr
|
|
79
|
+
Requires-Dist: easyocr; extra == "ocr"
|
|
80
|
+
Requires-Dist: pillow; extra == "ocr"
|
|
61
81
|
Provides-Extra: all
|
|
62
82
|
Requires-Dist: textract; extra == "all"
|
|
63
83
|
Requires-Dist: PyMuPDF; extra == "all"
|
|
@@ -68,6 +88,14 @@ Requires-Dist: xlrd; extra == "all"
|
|
|
68
88
|
Requires-Dist: odfpy; extra == "all"
|
|
69
89
|
Requires-Dist: beautifulsoup4; extra == "all"
|
|
70
90
|
Requires-Dist: lxml; extra == "all"
|
|
91
|
+
Requires-Dist: markdown; extra == "all"
|
|
92
|
+
Requires-Dist: ebooklib; extra == "all"
|
|
93
|
+
Requires-Dist: striprtf; extra == "all"
|
|
94
|
+
Requires-Dist: extract-msg; extra == "all"
|
|
95
|
+
Requires-Dist: pylatexenc; extra == "all"
|
|
96
|
+
Requires-Dist: openai-whisper; extra == "all"
|
|
97
|
+
Requires-Dist: easyocr; extra == "all"
|
|
98
|
+
Requires-Dist: pillow; extra == "all"
|
|
71
99
|
Dynamic: license-file
|
|
72
100
|
|
|
73
101
|
# PyxTxt
|
|
@@ -77,16 +105,18 @@ Dynamic: license-file
|
|
|
77
105
|
[](https://opensource.org/licenses/MIT)
|
|
78
106
|
|
|
79
107
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
80
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
|
|
108
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
|
|
81
109
|
|
|
82
|
-
**NEW in v0.2.
|
|
110
|
+
**NEW in v0.2.3+**: Added audio transcription (Whisper) and OCR from images (EasyOCR)!
|
|
83
111
|
|
|
84
112
|
---
|
|
85
113
|
|
|
86
114
|
## ✨ Features
|
|
87
115
|
|
|
88
116
|
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
89
|
-
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
|
|
117
|
+
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
|
|
118
|
+
- **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
|
|
119
|
+
- **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
|
|
90
120
|
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
91
121
|
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
92
122
|
- **Memory efficient**: Process files without saving to disk
|
|
@@ -103,8 +133,21 @@ pip install pyxtxt[all]
|
|
|
103
133
|
```
|
|
104
134
|
or just the modules you need:
|
|
105
135
|
```bash
|
|
106
|
-
pip install pyxtxt[pdf,
|
|
136
|
+
pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
|
|
107
137
|
```
|
|
138
|
+
|
|
139
|
+
### Audio & OCR (Heavy Dependencies)
|
|
140
|
+
```bash
|
|
141
|
+
# Audio transcription (~2GB download for Whisper models)
|
|
142
|
+
pip install pyxtxt[audio]
|
|
143
|
+
|
|
144
|
+
# OCR from images (~1GB download for EasyOCR models)
|
|
145
|
+
pip install pyxtxt[ocr]
|
|
146
|
+
|
|
147
|
+
# Both audio and OCR
|
|
148
|
+
pip install pyxtxt[audio,ocr]
|
|
149
|
+
```
|
|
150
|
+
|
|
108
151
|
Because needed libraries are common, installing the html module will also enable SVG and XML support.
|
|
109
152
|
The architecture is designed to grow with new modules for additional formats.
|
|
110
153
|
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
@@ -126,25 +169,24 @@ brew install libmagic
|
|
|
126
169
|
Use python-magic-bin instead of python-magic for easier installation.
|
|
127
170
|
|
|
128
171
|
## 🛠️ Dependencies
|
|
129
|
-
- PyMuPDF (fitz)
|
|
130
|
-
|
|
131
|
-
- beautifulsoup4
|
|
132
|
-
|
|
133
|
-
- python-docx
|
|
134
172
|
|
|
135
|
-
|
|
173
|
+
### Core Dependencies
|
|
174
|
+
- python-magic (automatic file type detection)
|
|
136
175
|
|
|
137
|
-
|
|
176
|
+
### Optional Dependencies by Format
|
|
177
|
+
- **PDF**: PyMuPDF
|
|
178
|
+
- **Office**: python-docx, python-pptx, openpyxl, xlrd
|
|
179
|
+
- **Web/HTML**: beautifulsoup4, lxml
|
|
180
|
+
- **OpenDocument**: odfpy
|
|
181
|
+
- **Markdown**: markdown
|
|
182
|
+
- **EPUB**: ebooklib
|
|
183
|
+
- **RTF**: striprtf
|
|
184
|
+
- **Email**: extract-msg (for MSG files)
|
|
185
|
+
- **LaTeX**: pylatexenc
|
|
186
|
+
- **Audio**: openai-whisper (heavy ~2GB models)
|
|
187
|
+
- **OCR**: easyocr, pillow (heavy ~1GB models)
|
|
138
188
|
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
- lxml
|
|
142
|
-
|
|
143
|
-
- xlrd (<2.0.0)
|
|
144
|
-
|
|
145
|
-
- python-magic
|
|
146
|
-
|
|
147
|
-
Dependencies are automatically installed from pyproject.toml.
|
|
189
|
+
Dependencies are automatically installed based on selected optional groups.
|
|
148
190
|
|
|
149
191
|
## 📚 Usage Examples
|
|
150
192
|
|
|
@@ -180,6 +222,36 @@ text = xtxt(response)
|
|
|
180
222
|
text = xtxt_from_url("https://example.com/document.pdf")
|
|
181
223
|
```
|
|
182
224
|
|
|
225
|
+
### Audio Transcription (NEW)
|
|
226
|
+
```python
|
|
227
|
+
from pyxtxt import xtxt
|
|
228
|
+
|
|
229
|
+
# Transcribe audio files
|
|
230
|
+
text = xtxt("meeting_recording.mp3")
|
|
231
|
+
text = xtxt("interview.wav")
|
|
232
|
+
text = xtxt("podcast.m4a")
|
|
233
|
+
|
|
234
|
+
# From web audio
|
|
235
|
+
import requests
|
|
236
|
+
audio_response = requests.get("https://example.com/audio.mp3")
|
|
237
|
+
text = xtxt(audio_response.content)
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
### OCR from Images (NEW)
|
|
241
|
+
```python
|
|
242
|
+
from pyxtxt import xtxt
|
|
243
|
+
|
|
244
|
+
# Extract text from images
|
|
245
|
+
text = xtxt("scanned_document.png")
|
|
246
|
+
text = xtxt("screenshot.jpg")
|
|
247
|
+
text = xtxt("invoice.tiff")
|
|
248
|
+
|
|
249
|
+
# From web images
|
|
250
|
+
import requests
|
|
251
|
+
image_response = requests.get("https://example.com/document.png")
|
|
252
|
+
text = xtxt(image_response.content)
|
|
253
|
+
```
|
|
254
|
+
|
|
183
255
|
### Show Available Formats
|
|
184
256
|
```python
|
|
185
257
|
from pyxtxt import extxt_available_formats
|
|
@@ -203,6 +275,14 @@ text = xtxt(api_response.content)
|
|
|
203
275
|
uploaded_bytes = request.files['document'].read()
|
|
204
276
|
text = xtxt(uploaded_bytes)
|
|
205
277
|
|
|
278
|
+
# Audio/video transcription services
|
|
279
|
+
audio_response = requests.get("https://api.example.com/recording.mp3")
|
|
280
|
+
transcript = xtxt(audio_response.content)
|
|
281
|
+
|
|
282
|
+
# OCR for uploaded images
|
|
283
|
+
image_bytes = request.files['receipt'].read()
|
|
284
|
+
text = xtxt(image_bytes)
|
|
285
|
+
|
|
206
286
|
# Email attachments
|
|
207
287
|
attachment_bytes = email_msg.get_payload(decode=True)
|
|
208
288
|
text = xtxt(attachment_bytes)
|
|
@@ -257,6 +337,14 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
257
337
|
|
|
258
338
|
## 📊 Changelog
|
|
259
339
|
|
|
340
|
+
### v0.2.3+
|
|
341
|
+
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
342
|
+
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
343
|
+
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
|
|
344
|
+
- ✅ Separate optional dependencies for heavy features (audio/OCR)
|
|
345
|
+
- ✅ Performance optimizations with model caching
|
|
346
|
+
- ✅ Improved multilingual OCR support (Italian/English)
|
|
347
|
+
|
|
260
348
|
### v0.1.24+
|
|
261
349
|
- ✅ Added support for `bytes` objects
|
|
262
350
|
- ✅ Added support for `requests.Response` objects
|
|
@@ -12,6 +12,7 @@ src/pyxtxt.egg-info/dependency_links.txt
|
|
|
12
12
|
src/pyxtxt.egg-info/requires.txt
|
|
13
13
|
src/pyxtxt.egg-info/top_level.txt
|
|
14
14
|
src/pyxtxt/estrattori/__init__.py
|
|
15
|
+
src/pyxtxt/estrattori/audio.py
|
|
15
16
|
src/pyxtxt/estrattori/doc.py
|
|
16
17
|
src/pyxtxt/estrattori/docx.py
|
|
17
18
|
src/pyxtxt/estrattori/eml.py
|
|
@@ -19,6 +20,7 @@ src/pyxtxt/estrattori/epub.py
|
|
|
19
20
|
src/pyxtxt/estrattori/html.py
|
|
20
21
|
src/pyxtxt/estrattori/md.py
|
|
21
22
|
src/pyxtxt/estrattori/msg.py
|
|
23
|
+
src/pyxtxt/estrattori/ocr.py
|
|
22
24
|
src/pyxtxt/estrattori/odt.py
|
|
23
25
|
src/pyxtxt/estrattori/pdf.py
|
|
24
26
|
src/pyxtxt/estrattori/pptx.py
|
|
@@ -15,6 +15,17 @@ xlrd
|
|
|
15
15
|
odfpy
|
|
16
16
|
beautifulsoup4
|
|
17
17
|
lxml
|
|
18
|
+
markdown
|
|
19
|
+
ebooklib
|
|
20
|
+
striprtf
|
|
21
|
+
extract-msg
|
|
22
|
+
pylatexenc
|
|
23
|
+
openai-whisper
|
|
24
|
+
easyocr
|
|
25
|
+
pillow
|
|
26
|
+
|
|
27
|
+
[audio]
|
|
28
|
+
openai-whisper
|
|
18
29
|
|
|
19
30
|
[doc]
|
|
20
31
|
textract
|
|
@@ -22,19 +33,44 @@ textract
|
|
|
22
33
|
[docx]
|
|
23
34
|
python-docx
|
|
24
35
|
|
|
36
|
+
[email]
|
|
37
|
+
beautifulsoup4
|
|
38
|
+
|
|
39
|
+
[epub]
|
|
40
|
+
ebooklib
|
|
41
|
+
beautifulsoup4
|
|
42
|
+
|
|
25
43
|
[html]
|
|
26
44
|
beautifulsoup4
|
|
27
45
|
lxml
|
|
28
46
|
|
|
47
|
+
[latex]
|
|
48
|
+
pylatexenc
|
|
49
|
+
|
|
50
|
+
[markdown]
|
|
51
|
+
markdown
|
|
52
|
+
beautifulsoup4
|
|
53
|
+
|
|
54
|
+
[ocr]
|
|
55
|
+
easyocr
|
|
56
|
+
pillow
|
|
57
|
+
|
|
29
58
|
[odf]
|
|
30
59
|
odfpy
|
|
31
60
|
|
|
61
|
+
[outlook]
|
|
62
|
+
extract-msg
|
|
63
|
+
beautifulsoup4
|
|
64
|
+
|
|
32
65
|
[pdf]
|
|
33
66
|
PyMuPDF
|
|
34
67
|
|
|
35
68
|
[presentation]
|
|
36
69
|
python-pptx
|
|
37
70
|
|
|
71
|
+
[rtf]
|
|
72
|
+
striprtf
|
|
73
|
+
|
|
38
74
|
[spreadsheet]
|
|
39
75
|
openpyxl
|
|
40
76
|
xlrd
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|