pyxtxt 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {pyxtxt-0.2.2/src/pyxtxt.egg-info → pyxtxt-0.2.3}/PKG-INFO +123 -21
  2. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/README.md +80 -20
  3. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/pyproject.toml +54 -2
  4. pyxtxt-0.2.3/src/pyxtxt/estrattori/audio.py +52 -0
  5. pyxtxt-0.2.3/src/pyxtxt/estrattori/ocr.py +52 -0
  6. {pyxtxt-0.2.2 → pyxtxt-0.2.3/src/pyxtxt.egg-info}/PKG-INFO +123 -21
  7. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/SOURCES.txt +3 -1
  8. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/requires.txt +36 -0
  9. /pyxtxt-0.2.2/LICENCSE → /pyxtxt-0.2.3/LICENSE +0 -0
  10. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/MANIFEST.in +0 -0
  11. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/examples.py +0 -0
  12. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/setup.cfg +0 -0
  13. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/__init__.py +0 -0
  14. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/core.py +0 -0
  15. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/__init__.py +0 -0
  16. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/doc.py +0 -0
  17. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/docx.py +0 -0
  18. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/eml.py +0 -0
  19. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/epub.py +0 -0
  20. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/html.py +0 -0
  21. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/md.py +0 -0
  22. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/msg.py +0 -0
  23. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/odt.py +0 -0
  24. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/pdf.py +0 -0
  25. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/pptx.py +0 -0
  26. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/rtf.py +0 -0
  27. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/svg.py +0 -0
  28. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/tex.py +0 -0
  29. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/txt.py +0 -0
  30. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/xls.py +0 -0
  31. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/xlsx.py +0 -0
  32. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/estrattori/xml.py +0 -0
  33. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt/pyxtxt.py +0 -0
  34. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  35. {pyxtxt-0.2.2 → pyxtxt-0.2.3}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -25,8 +25,21 @@ License: MIT License
25
25
  OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
26
  SOFTWARE.
27
27
 
28
+ Classifier: Development Status :: 4 - Beta
29
+ Classifier: Intended Audience :: Developers
30
+ Classifier: License :: OSI Approved :: MIT License
31
+ Classifier: Operating System :: OS Independent
32
+ Classifier: Programming Language :: Python :: 3
33
+ Classifier: Programming Language :: Python :: 3.7
34
+ Classifier: Programming Language :: Python :: 3.8
35
+ Classifier: Programming Language :: Python :: 3.9
36
+ Classifier: Programming Language :: Python :: 3.10
37
+ Classifier: Programming Language :: Python :: 3.11
38
+ Classifier: Topic :: Text Processing
39
+ Classifier: Topic :: Utilities
28
40
  Requires-Python: >=3.7
29
41
  Description-Content-Type: text/markdown
42
+ License-File: LICENSE
30
43
  Requires-Dist: python-magic; sys_platform != "win32"
31
44
  Requires-Dist: python-magic-bin; sys_platform == "win32"
32
45
  Provides-Extra: pdf
@@ -45,6 +58,26 @@ Requires-Dist: beautifulsoup4; extra == "html"
45
58
  Requires-Dist: lxml; extra == "html"
46
59
  Provides-Extra: doc
47
60
  Requires-Dist: textract; extra == "doc"
61
+ Provides-Extra: markdown
62
+ Requires-Dist: markdown; extra == "markdown"
63
+ Requires-Dist: beautifulsoup4; extra == "markdown"
64
+ Provides-Extra: epub
65
+ Requires-Dist: ebooklib; extra == "epub"
66
+ Requires-Dist: beautifulsoup4; extra == "epub"
67
+ Provides-Extra: rtf
68
+ Requires-Dist: striprtf; extra == "rtf"
69
+ Provides-Extra: email
70
+ Requires-Dist: beautifulsoup4; extra == "email"
71
+ Provides-Extra: outlook
72
+ Requires-Dist: extract-msg; extra == "outlook"
73
+ Requires-Dist: beautifulsoup4; extra == "outlook"
74
+ Provides-Extra: latex
75
+ Requires-Dist: pylatexenc; extra == "latex"
76
+ Provides-Extra: audio
77
+ Requires-Dist: openai-whisper; extra == "audio"
78
+ Provides-Extra: ocr
79
+ Requires-Dist: easyocr; extra == "ocr"
80
+ Requires-Dist: pillow; extra == "ocr"
48
81
  Provides-Extra: all
49
82
  Requires-Dist: textract; extra == "all"
50
83
  Requires-Dist: PyMuPDF; extra == "all"
@@ -55,6 +88,15 @@ Requires-Dist: xlrd; extra == "all"
55
88
  Requires-Dist: odfpy; extra == "all"
56
89
  Requires-Dist: beautifulsoup4; extra == "all"
57
90
  Requires-Dist: lxml; extra == "all"
91
+ Requires-Dist: markdown; extra == "all"
92
+ Requires-Dist: ebooklib; extra == "all"
93
+ Requires-Dist: striprtf; extra == "all"
94
+ Requires-Dist: extract-msg; extra == "all"
95
+ Requires-Dist: pylatexenc; extra == "all"
96
+ Requires-Dist: openai-whisper; extra == "all"
97
+ Requires-Dist: easyocr; extra == "all"
98
+ Requires-Dist: pillow; extra == "all"
99
+ Dynamic: license-file
58
100
 
59
101
  # PyxTxt
60
102
 
@@ -63,16 +105,18 @@ Requires-Dist: lxml; extra == "all"
63
105
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
64
106
 
65
107
  **PyxTxt** is a simple and powerful Python library to extract text from various file formats.
66
- It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
108
+ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
67
109
 
68
- **NEW in v0.2.1+**: Enhanced support for web content, byte streams, and requests integration!
110
+ **NEW in v0.2.3+**: Added audio transcription (Whisper) and OCR from images (EasyOCR)!
69
111
 
70
112
  ---
71
113
 
72
114
  ## ✨ Features
73
115
 
74
116
  - **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
75
- - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
117
+ - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
118
+ - **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
119
+ - **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
76
120
  - **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
77
121
  - **Web-ready**: Direct support for downloading and extracting text from URLs
78
122
  - **Memory efficient**: Process files without saving to disk
@@ -89,8 +133,21 @@ pip install pyxtxt[all]
89
133
  ```
90
134
  or just the modules you need:
91
135
  ```bash
92
- pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
136
+ pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
93
137
  ```
138
+
139
+ ### Audio & OCR (Heavy Dependencies)
140
+ ```bash
141
+ # Audio transcription (~2GB download for Whisper models)
142
+ pip install pyxtxt[audio]
143
+
144
+ # OCR from images (~1GB download for EasyOCR models)
145
+ pip install pyxtxt[ocr]
146
+
147
+ # Both audio and OCR
148
+ pip install pyxtxt[audio,ocr]
149
+ ```
150
+
94
151
  Because needed libraries are common, installing the html module will also enable SVG and XML support.
95
152
  The architecture is designed to grow with new modules for additional formats.
96
153
  ## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
@@ -112,25 +169,24 @@ brew install libmagic
112
169
  Use python-magic-bin instead of python-magic for easier installation.
113
170
 
114
171
  ## 🛠️ Dependencies
115
- - PyMuPDF (fitz)
116
-
117
- - beautifulsoup4
118
-
119
- - python-docx
120
172
 
121
- - python-pptx
173
+ ### Core Dependencies
174
+ - python-magic (automatic file type detection)
122
175
 
123
- - odfpy
176
+ ### Optional Dependencies by Format
177
+ - **PDF**: PyMuPDF
178
+ - **Office**: python-docx, python-pptx, openpyxl, xlrd
179
+ - **Web/HTML**: beautifulsoup4, lxml
180
+ - **OpenDocument**: odfpy
181
+ - **Markdown**: markdown
182
+ - **EPUB**: ebooklib
183
+ - **RTF**: striprtf
184
+ - **Email**: extract-msg (for MSG files)
185
+ - **LaTeX**: pylatexenc
186
+ - **Audio**: openai-whisper (heavy ~2GB models)
187
+ - **OCR**: easyocr, pillow (heavy ~1GB models)
124
188
 
125
- - openpyxl
126
-
127
- - lxml
128
-
129
- - xlrd (<2.0.0)
130
-
131
- - python-magic
132
-
133
- Dependencies are automatically installed from pyproject.toml.
189
+ Dependencies are automatically installed based on selected optional groups.
134
190
 
135
191
  ## 📚 Usage Examples
136
192
 
@@ -166,6 +222,36 @@ text = xtxt(response)
166
222
  text = xtxt_from_url("https://example.com/document.pdf")
167
223
  ```
168
224
 
225
+ ### Audio Transcription (NEW)
226
+ ```python
227
+ from pyxtxt import xtxt
228
+
229
+ # Transcribe audio files
230
+ text = xtxt("meeting_recording.mp3")
231
+ text = xtxt("interview.wav")
232
+ text = xtxt("podcast.m4a")
233
+
234
+ # From web audio
235
+ import requests
236
+ audio_response = requests.get("https://example.com/audio.mp3")
237
+ text = xtxt(audio_response.content)
238
+ ```
239
+
240
+ ### OCR from Images (NEW)
241
+ ```python
242
+ from pyxtxt import xtxt
243
+
244
+ # Extract text from images
245
+ text = xtxt("scanned_document.png")
246
+ text = xtxt("screenshot.jpg")
247
+ text = xtxt("invoice.tiff")
248
+
249
+ # From web images
250
+ import requests
251
+ image_response = requests.get("https://example.com/document.png")
252
+ text = xtxt(image_response.content)
253
+ ```
254
+
169
255
  ### Show Available Formats
170
256
  ```python
171
257
  from pyxtxt import extxt_available_formats
@@ -189,6 +275,14 @@ text = xtxt(api_response.content)
189
275
  uploaded_bytes = request.files['document'].read()
190
276
  text = xtxt(uploaded_bytes)
191
277
 
278
+ # Audio/video transcription services
279
+ audio_response = requests.get("https://api.example.com/recording.mp3")
280
+ transcript = xtxt(audio_response.content)
281
+
282
+ # OCR for uploaded images
283
+ image_bytes = request.files['receipt'].read()
284
+ text = xtxt(image_bytes)
285
+
192
286
  # Email attachments
193
287
  attachment_bytes = email_msg.get_payload(decode=True)
194
288
  text = xtxt(attachment_bytes)
@@ -243,6 +337,14 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
243
337
 
244
338
  ## 📊 Changelog
245
339
 
340
+ ### v0.2.3+
341
+ - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
342
+ - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
343
+ - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
344
+ - ✅ Separate optional dependencies for heavy features (audio/OCR)
345
+ - ✅ Performance optimizations with model caching
346
+ - ✅ Improved multilingual OCR support (Italian/English)
347
+
246
348
  ### v0.1.24+
247
349
  - ✅ Added support for `bytes` objects
248
350
  - ✅ Added support for `requests.Response` objects
@@ -5,16 +5,18 @@
5
5
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
6
6
 
7
7
  **PyxTxt** is a simple and powerful Python library to extract text from various file formats.
8
- It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
8
+ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
9
9
 
10
- **NEW in v0.2.1+**: Enhanced support for web content, byte streams, and requests integration!
10
+ **NEW in v0.2.3+**: Added audio transcription (Whisper) and OCR from images (EasyOCR)!
11
11
 
12
12
  ---
13
13
 
14
14
  ## ✨ Features
15
15
 
16
16
  - **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
17
- - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
17
+ - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
18
+ - **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
19
+ - **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
18
20
  - **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
19
21
  - **Web-ready**: Direct support for downloading and extracting text from URLs
20
22
  - **Memory efficient**: Process files without saving to disk
@@ -31,8 +33,21 @@ pip install pyxtxt[all]
31
33
  ```
32
34
  or just the modules you need:
33
35
  ```bash
34
- pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
36
+ pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
35
37
  ```
38
+
39
+ ### Audio & OCR (Heavy Dependencies)
40
+ ```bash
41
+ # Audio transcription (~2GB download for Whisper models)
42
+ pip install pyxtxt[audio]
43
+
44
+ # OCR from images (~1GB download for EasyOCR models)
45
+ pip install pyxtxt[ocr]
46
+
47
+ # Both audio and OCR
48
+ pip install pyxtxt[audio,ocr]
49
+ ```
50
+
36
51
  Because needed libraries are common, installing the html module will also enable SVG and XML support.
37
52
  The architecture is designed to grow with new modules for additional formats.
38
53
  ## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
@@ -54,25 +69,24 @@ brew install libmagic
54
69
  Use python-magic-bin instead of python-magic for easier installation.
55
70
 
56
71
  ## 🛠️ Dependencies
57
- - PyMuPDF (fitz)
58
-
59
- - beautifulsoup4
60
-
61
- - python-docx
62
72
 
63
- - python-pptx
73
+ ### Core Dependencies
74
+ - python-magic (automatic file type detection)
64
75
 
65
- - odfpy
76
+ ### Optional Dependencies by Format
77
+ - **PDF**: PyMuPDF
78
+ - **Office**: python-docx, python-pptx, openpyxl, xlrd
79
+ - **Web/HTML**: beautifulsoup4, lxml
80
+ - **OpenDocument**: odfpy
81
+ - **Markdown**: markdown
82
+ - **EPUB**: ebooklib
83
+ - **RTF**: striprtf
84
+ - **Email**: extract-msg (for MSG files)
85
+ - **LaTeX**: pylatexenc
86
+ - **Audio**: openai-whisper (heavy ~2GB models)
87
+ - **OCR**: easyocr, pillow (heavy ~1GB models)
66
88
 
67
- - openpyxl
68
-
69
- - lxml
70
-
71
- - xlrd (<2.0.0)
72
-
73
- - python-magic
74
-
75
- Dependencies are automatically installed from pyproject.toml.
89
+ Dependencies are automatically installed based on selected optional groups.
76
90
 
77
91
  ## 📚 Usage Examples
78
92
 
@@ -108,6 +122,36 @@ text = xtxt(response)
108
122
  text = xtxt_from_url("https://example.com/document.pdf")
109
123
  ```
110
124
 
125
+ ### Audio Transcription (NEW)
126
+ ```python
127
+ from pyxtxt import xtxt
128
+
129
+ # Transcribe audio files
130
+ text = xtxt("meeting_recording.mp3")
131
+ text = xtxt("interview.wav")
132
+ text = xtxt("podcast.m4a")
133
+
134
+ # From web audio
135
+ import requests
136
+ audio_response = requests.get("https://example.com/audio.mp3")
137
+ text = xtxt(audio_response.content)
138
+ ```
139
+
140
+ ### OCR from Images (NEW)
141
+ ```python
142
+ from pyxtxt import xtxt
143
+
144
+ # Extract text from images
145
+ text = xtxt("scanned_document.png")
146
+ text = xtxt("screenshot.jpg")
147
+ text = xtxt("invoice.tiff")
148
+
149
+ # From web images
150
+ import requests
151
+ image_response = requests.get("https://example.com/document.png")
152
+ text = xtxt(image_response.content)
153
+ ```
154
+
111
155
  ### Show Available Formats
112
156
  ```python
113
157
  from pyxtxt import extxt_available_formats
@@ -131,6 +175,14 @@ text = xtxt(api_response.content)
131
175
  uploaded_bytes = request.files['document'].read()
132
176
  text = xtxt(uploaded_bytes)
133
177
 
178
+ # Audio/video transcription services
179
+ audio_response = requests.get("https://api.example.com/recording.mp3")
180
+ transcript = xtxt(audio_response.content)
181
+
182
+ # OCR for uploaded images
183
+ image_bytes = request.files['receipt'].read()
184
+ text = xtxt(image_bytes)
185
+
134
186
  # Email attachments
135
187
  attachment_bytes = email_msg.get_payload(decode=True)
136
188
  text = xtxt(attachment_bytes)
@@ -185,6 +237,14 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
185
237
 
186
238
  ## 📊 Changelog
187
239
 
240
+ ### v0.2.3+
241
+ - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
242
+ - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
243
+ - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
244
+ - ✅ Separate optional dependencies for heavy features (audio/OCR)
245
+ - ✅ Performance optimizations with model caching
246
+ - ✅ Improved multilingual OCR support (Italian/English)
247
+
188
248
  ### v0.1.24+
189
249
  - ✅ Added support for `bytes` objects
190
250
  - ✅ Added support for `requests.Response` objects
@@ -1,10 +1,26 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.2.2"
3
+ version = "0.2.3"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
+ classifiers = [
6
+ "Development Status :: 4 - Beta",
7
+ "Intended Audience :: Developers",
8
+ "License :: OSI Approved :: MIT License",
9
+ "Operating System :: OS Independent",
10
+ "Programming Language :: Python :: 3",
11
+ "Programming Language :: Python :: 3.7",
12
+ "Programming Language :: Python :: 3.8",
13
+ "Programming Language :: Python :: 3.9",
14
+ "Programming Language :: Python :: 3.10",
15
+ "Programming Language :: Python :: 3.11",
16
+ "Topic :: Text Processing",
17
+ "Topic :: Utilities"
18
+ ]
19
+
5
20
  readme = "README.md"
6
21
  requires-python = ">=3.7"
7
- license = { file = "LICENCSE" }
22
+
23
+ license = { file = "LICENSE" }
8
24
  authors = [
9
25
  { name = "Giuseppe Levi", email = "giuseppe.levi@gmail.com" }
10
26
  ]
@@ -36,6 +52,34 @@ html = [
36
52
  doc = [
37
53
  "textract",
38
54
  ]
55
+ markdown = [
56
+ "markdown",
57
+ "beautifulsoup4",
58
+ ]
59
+ epub = [
60
+ "ebooklib",
61
+ "beautifulsoup4",
62
+ ]
63
+ rtf = [
64
+ "striprtf",
65
+ ]
66
+ email = [
67
+ "beautifulsoup4",
68
+ ]
69
+ outlook = [
70
+ "extract-msg",
71
+ "beautifulsoup4",
72
+ ]
73
+ latex = [
74
+ "pylatexenc",
75
+ ]
76
+ audio = [
77
+ "openai-whisper",
78
+ ]
79
+ ocr = [
80
+ "easyocr",
81
+ "pillow",
82
+ ]
39
83
  all = [
40
84
  "textract",
41
85
  "PyMuPDF",
@@ -46,6 +90,14 @@ all = [
46
90
  "odfpy",
47
91
  "beautifulsoup4",
48
92
  "lxml",
93
+ "markdown",
94
+ "ebooklib",
95
+ "striprtf",
96
+ "extract-msg",
97
+ "pylatexenc",
98
+ "openai-whisper",
99
+ "easyocr",
100
+ "pillow",
49
101
  ]
50
102
 
51
103
  [build-system]
@@ -0,0 +1,52 @@
1
+ # pyxtxt/extractors/audio_whisper.py
2
+ from . import register_extractor
3
+ import tempfile
4
+ import os
5
+
6
+ try:
7
+ import whisper
8
+ except ImportError:
9
+ whisper = None
10
+
11
+ if whisper:
12
+ _whisper_model = None
13
+
14
+ def _get_model():
15
+ global _whisper_model
16
+ if _whisper_model is None:
17
+ _whisper_model = whisper.load_model("base")
18
+ return _whisper_model
19
+
20
+ def xtxt_audio_whisper(file_buffer):
21
+ try:
22
+ # Usa un suffixe generico - Whisper + FFmpeg gestiscono il formato
23
+ with tempfile.NamedTemporaryFile(delete=False) as temp_file:
24
+ temp_file.write(file_buffer.read())
25
+ temp_file.flush()
26
+
27
+ model = _get_model()
28
+ result = model.transcribe(temp_file.name)
29
+
30
+ os.unlink(temp_file.name)
31
+
32
+ return result['text'].strip()
33
+
34
+ except Exception as e:
35
+ print(f"⚠️ Error while extracting audio with Whisper: {e}")
36
+ return ""
37
+
38
+ # Registra per tutti i formati audio comuni
39
+ audio_formats = [
40
+ "audio/wav", "audio/wave",
41
+ "audio/mp3", "audio/mpeg",
42
+ "audio/m4a", "audio/mp4",
43
+ "audio/flac",
44
+ "audio/ogg", "audio/ogg-vorbis",
45
+ "audio/opus",
46
+ "audio/aac",
47
+ "audio/wma",
48
+ "audio/webm"
49
+ ]
50
+
51
+ for format_type in audio_formats:
52
+ register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio")
@@ -0,0 +1,52 @@
1
+ # pyxtxt/extractors/image_ocr.py
2
+ from . import register_extractor
3
+ from io import BytesIO
4
+
5
+ try:
6
+ import easyocr
7
+ from PIL import Image
8
+ except ImportError:
9
+ easyocr = None
10
+ Image = None
11
+
12
+ if easyocr and Image:
13
+ # Inizializza il reader OCR una volta sola
14
+ _ocr_reader = None
15
+
16
+ def _get_ocr_reader():
17
+ global _ocr_reader
18
+ if _ocr_reader is None:
19
+ _ocr_reader = easyocr.Reader(['it', 'en'], gpu=False)
20
+ return _ocr_reader
21
+
22
+ def xtxt_image_ocr(file_buffer):
23
+ try:
24
+ # Converti il buffer in immagine PIL
25
+ image = Image.open(BytesIO(file_buffer.read()))
26
+
27
+ # Converti in RGB se necessario
28
+ if image.mode != 'RGB':
29
+ image = image.convert('RGB')
30
+
31
+ reader = _get_ocr_reader()
32
+ results = reader.readtext(image)
33
+
34
+ # Estrai solo il testo, ordinato per posizione verticale
35
+ texts = []
36
+ for (bbox, text, confidence) in results:
37
+ if confidence > 0.3: # Filtra testo con bassa confidenza
38
+ texts.append(text)
39
+
40
+ return "\n".join(texts)
41
+
42
+ except Exception as e:
43
+ print(f"⚠️ Error while extracting text from image: {e}")
44
+ return ""
45
+
46
+ # Registra per i formati immagine più comuni
47
+ register_extractor("image/jpeg", xtxt_image_ocr, name="OCR")
48
+ register_extractor("image/jpg", xtxt_image_ocr, name="OCR")
49
+ register_extractor("image/png", xtxt_image_ocr, name="OCR")
50
+ register_extractor("image/bmp", xtxt_image_ocr, name="OCR")
51
+ register_extractor("image/tiff", xtxt_image_ocr, name="OCR")
52
+ register_extractor("image/webp", xtxt_image_ocr, name="OCR")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -25,8 +25,21 @@ License: MIT License
25
25
  OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
26
  SOFTWARE.
27
27
 
28
+ Classifier: Development Status :: 4 - Beta
29
+ Classifier: Intended Audience :: Developers
30
+ Classifier: License :: OSI Approved :: MIT License
31
+ Classifier: Operating System :: OS Independent
32
+ Classifier: Programming Language :: Python :: 3
33
+ Classifier: Programming Language :: Python :: 3.7
34
+ Classifier: Programming Language :: Python :: 3.8
35
+ Classifier: Programming Language :: Python :: 3.9
36
+ Classifier: Programming Language :: Python :: 3.10
37
+ Classifier: Programming Language :: Python :: 3.11
38
+ Classifier: Topic :: Text Processing
39
+ Classifier: Topic :: Utilities
28
40
  Requires-Python: >=3.7
29
41
  Description-Content-Type: text/markdown
42
+ License-File: LICENSE
30
43
  Requires-Dist: python-magic; sys_platform != "win32"
31
44
  Requires-Dist: python-magic-bin; sys_platform == "win32"
32
45
  Provides-Extra: pdf
@@ -45,6 +58,26 @@ Requires-Dist: beautifulsoup4; extra == "html"
45
58
  Requires-Dist: lxml; extra == "html"
46
59
  Provides-Extra: doc
47
60
  Requires-Dist: textract; extra == "doc"
61
+ Provides-Extra: markdown
62
+ Requires-Dist: markdown; extra == "markdown"
63
+ Requires-Dist: beautifulsoup4; extra == "markdown"
64
+ Provides-Extra: epub
65
+ Requires-Dist: ebooklib; extra == "epub"
66
+ Requires-Dist: beautifulsoup4; extra == "epub"
67
+ Provides-Extra: rtf
68
+ Requires-Dist: striprtf; extra == "rtf"
69
+ Provides-Extra: email
70
+ Requires-Dist: beautifulsoup4; extra == "email"
71
+ Provides-Extra: outlook
72
+ Requires-Dist: extract-msg; extra == "outlook"
73
+ Requires-Dist: beautifulsoup4; extra == "outlook"
74
+ Provides-Extra: latex
75
+ Requires-Dist: pylatexenc; extra == "latex"
76
+ Provides-Extra: audio
77
+ Requires-Dist: openai-whisper; extra == "audio"
78
+ Provides-Extra: ocr
79
+ Requires-Dist: easyocr; extra == "ocr"
80
+ Requires-Dist: pillow; extra == "ocr"
48
81
  Provides-Extra: all
49
82
  Requires-Dist: textract; extra == "all"
50
83
  Requires-Dist: PyMuPDF; extra == "all"
@@ -55,6 +88,15 @@ Requires-Dist: xlrd; extra == "all"
55
88
  Requires-Dist: odfpy; extra == "all"
56
89
  Requires-Dist: beautifulsoup4; extra == "all"
57
90
  Requires-Dist: lxml; extra == "all"
91
+ Requires-Dist: markdown; extra == "all"
92
+ Requires-Dist: ebooklib; extra == "all"
93
+ Requires-Dist: striprtf; extra == "all"
94
+ Requires-Dist: extract-msg; extra == "all"
95
+ Requires-Dist: pylatexenc; extra == "all"
96
+ Requires-Dist: openai-whisper; extra == "all"
97
+ Requires-Dist: easyocr; extra == "all"
98
+ Requires-Dist: pillow; extra == "all"
99
+ Dynamic: license-file
58
100
 
59
101
  # PyxTxt
60
102
 
@@ -63,16 +105,18 @@ Requires-Dist: lxml; extra == "all"
63
105
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
64
106
 
65
107
  **PyxTxt** is a simple and powerful Python library to extract text from various file formats.
66
- It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
108
+ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
67
109
 
68
- **NEW in v0.2.1+**: Enhanced support for web content, byte streams, and requests integration!
110
+ **NEW in v0.2.3+**: Added audio transcription (Whisper) and OCR from images (EasyOCR)!
69
111
 
70
112
  ---
71
113
 
72
114
  ## ✨ Features
73
115
 
74
116
  - **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
75
- - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
117
+ - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
118
+ - **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
119
+ - **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
76
120
  - **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
77
121
  - **Web-ready**: Direct support for downloading and extracting text from URLs
78
122
  - **Memory efficient**: Process files without saving to disk
@@ -89,8 +133,21 @@ pip install pyxtxt[all]
89
133
  ```
90
134
  or just the modules you need:
91
135
  ```bash
92
- pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
136
+ pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
93
137
  ```
138
+
139
+ ### Audio & OCR (Heavy Dependencies)
140
+ ```bash
141
+ # Audio transcription (~2GB download for Whisper models)
142
+ pip install pyxtxt[audio]
143
+
144
+ # OCR from images (~1GB download for EasyOCR models)
145
+ pip install pyxtxt[ocr]
146
+
147
+ # Both audio and OCR
148
+ pip install pyxtxt[audio,ocr]
149
+ ```
150
+
94
151
  Because needed libraries are common, installing the html module will also enable SVG and XML support.
95
152
  The architecture is designed to grow with new modules for additional formats.
96
153
  ## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
@@ -112,25 +169,24 @@ brew install libmagic
112
169
  Use python-magic-bin instead of python-magic for easier installation.
113
170
 
114
171
  ## 🛠️ Dependencies
115
- - PyMuPDF (fitz)
116
-
117
- - beautifulsoup4
118
-
119
- - python-docx
120
172
 
121
- - python-pptx
173
+ ### Core Dependencies
174
+ - python-magic (automatic file type detection)
122
175
 
123
- - odfpy
176
+ ### Optional Dependencies by Format
177
+ - **PDF**: PyMuPDF
178
+ - **Office**: python-docx, python-pptx, openpyxl, xlrd
179
+ - **Web/HTML**: beautifulsoup4, lxml
180
+ - **OpenDocument**: odfpy
181
+ - **Markdown**: markdown
182
+ - **EPUB**: ebooklib
183
+ - **RTF**: striprtf
184
+ - **Email**: extract-msg (for MSG files)
185
+ - **LaTeX**: pylatexenc
186
+ - **Audio**: openai-whisper (heavy ~2GB models)
187
+ - **OCR**: easyocr, pillow (heavy ~1GB models)
124
188
 
125
- - openpyxl
126
-
127
- - lxml
128
-
129
- - xlrd (<2.0.0)
130
-
131
- - python-magic
132
-
133
- Dependencies are automatically installed from pyproject.toml.
189
+ Dependencies are automatically installed based on selected optional groups.
134
190
 
135
191
  ## 📚 Usage Examples
136
192
 
@@ -166,6 +222,36 @@ text = xtxt(response)
166
222
  text = xtxt_from_url("https://example.com/document.pdf")
167
223
  ```
168
224
 
225
+ ### Audio Transcription (NEW)
226
+ ```python
227
+ from pyxtxt import xtxt
228
+
229
+ # Transcribe audio files
230
+ text = xtxt("meeting_recording.mp3")
231
+ text = xtxt("interview.wav")
232
+ text = xtxt("podcast.m4a")
233
+
234
+ # From web audio
235
+ import requests
236
+ audio_response = requests.get("https://example.com/audio.mp3")
237
+ text = xtxt(audio_response.content)
238
+ ```
239
+
240
+ ### OCR from Images (NEW)
241
+ ```python
242
+ from pyxtxt import xtxt
243
+
244
+ # Extract text from images
245
+ text = xtxt("scanned_document.png")
246
+ text = xtxt("screenshot.jpg")
247
+ text = xtxt("invoice.tiff")
248
+
249
+ # From web images
250
+ import requests
251
+ image_response = requests.get("https://example.com/document.png")
252
+ text = xtxt(image_response.content)
253
+ ```
254
+
169
255
  ### Show Available Formats
170
256
  ```python
171
257
  from pyxtxt import extxt_available_formats
@@ -189,6 +275,14 @@ text = xtxt(api_response.content)
189
275
  uploaded_bytes = request.files['document'].read()
190
276
  text = xtxt(uploaded_bytes)
191
277
 
278
+ # Audio/video transcription services
279
+ audio_response = requests.get("https://api.example.com/recording.mp3")
280
+ transcript = xtxt(audio_response.content)
281
+
282
+ # OCR for uploaded images
283
+ image_bytes = request.files['receipt'].read()
284
+ text = xtxt(image_bytes)
285
+
192
286
  # Email attachments
193
287
  attachment_bytes = email_msg.get_payload(decode=True)
194
288
  text = xtxt(attachment_bytes)
@@ -243,6 +337,14 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
243
337
 
244
338
  ## 📊 Changelog
245
339
 
340
+ ### v0.2.3+
341
+ - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
342
+ - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
343
+ - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
344
+ - ✅ Separate optional dependencies for heavy features (audio/OCR)
345
+ - ✅ Performance optimizations with model caching
346
+ - ✅ Improved multilingual OCR support (Italian/English)
347
+
246
348
  ### v0.1.24+
247
349
  - ✅ Added support for `bytes` objects
248
350
  - ✅ Added support for `requests.Response` objects
@@ -1,4 +1,4 @@
1
- LICENCSE
1
+ LICENSE
2
2
  MANIFEST.in
3
3
  README.md
4
4
  examples.py
@@ -12,6 +12,7 @@ src/pyxtxt.egg-info/dependency_links.txt
12
12
  src/pyxtxt.egg-info/requires.txt
13
13
  src/pyxtxt.egg-info/top_level.txt
14
14
  src/pyxtxt/estrattori/__init__.py
15
+ src/pyxtxt/estrattori/audio.py
15
16
  src/pyxtxt/estrattori/doc.py
16
17
  src/pyxtxt/estrattori/docx.py
17
18
  src/pyxtxt/estrattori/eml.py
@@ -19,6 +20,7 @@ src/pyxtxt/estrattori/epub.py
19
20
  src/pyxtxt/estrattori/html.py
20
21
  src/pyxtxt/estrattori/md.py
21
22
  src/pyxtxt/estrattori/msg.py
23
+ src/pyxtxt/estrattori/ocr.py
22
24
  src/pyxtxt/estrattori/odt.py
23
25
  src/pyxtxt/estrattori/pdf.py
24
26
  src/pyxtxt/estrattori/pptx.py
@@ -15,6 +15,17 @@ xlrd
15
15
  odfpy
16
16
  beautifulsoup4
17
17
  lxml
18
+ markdown
19
+ ebooklib
20
+ striprtf
21
+ extract-msg
22
+ pylatexenc
23
+ openai-whisper
24
+ easyocr
25
+ pillow
26
+
27
+ [audio]
28
+ openai-whisper
18
29
 
19
30
  [doc]
20
31
  textract
@@ -22,19 +33,44 @@ textract
22
33
  [docx]
23
34
  python-docx
24
35
 
36
+ [email]
37
+ beautifulsoup4
38
+
39
+ [epub]
40
+ ebooklib
41
+ beautifulsoup4
42
+
25
43
  [html]
26
44
  beautifulsoup4
27
45
  lxml
28
46
 
47
+ [latex]
48
+ pylatexenc
49
+
50
+ [markdown]
51
+ markdown
52
+ beautifulsoup4
53
+
54
+ [ocr]
55
+ easyocr
56
+ pillow
57
+
29
58
  [odf]
30
59
  odfpy
31
60
 
61
+ [outlook]
62
+ extract-msg
63
+ beautifulsoup4
64
+
32
65
  [pdf]
33
66
  PyMuPDF
34
67
 
35
68
  [presentation]
36
69
  python-pptx
37
70
 
71
+ [rtf]
72
+ striprtf
73
+
38
74
  [spreadsheet]
39
75
  openpyxl
40
76
  xlrd
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes