pyxtxt 0.2.2.1__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {pyxtxt-0.2.2.1/src/pyxtxt.egg-info → pyxtxt-0.2.4}/PKG-INFO +127 -21
  2. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/README.md +98 -20
  3. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/pyproject.toml +37 -1
  4. pyxtxt-0.2.4/src/pyxtxt/estrattori/audio.py +79 -0
  5. pyxtxt-0.2.4/src/pyxtxt/estrattori/ocr.py +52 -0
  6. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4/src/pyxtxt.egg-info}/PKG-INFO +127 -21
  7. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt.egg-info/SOURCES.txt +2 -0
  8. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt.egg-info/requires.txt +36 -0
  9. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/LICENSE +0 -0
  10. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/MANIFEST.in +0 -0
  11. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/examples.py +0 -0
  12. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/setup.cfg +0 -0
  13. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/__init__.py +0 -0
  14. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/core.py +0 -0
  15. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/__init__.py +0 -0
  16. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/doc.py +0 -0
  17. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/docx.py +0 -0
  18. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/eml.py +0 -0
  19. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/epub.py +0 -0
  20. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/html.py +0 -0
  21. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/md.py +0 -0
  22. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/msg.py +0 -0
  23. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/odt.py +0 -0
  24. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/pdf.py +0 -0
  25. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/pptx.py +0 -0
  26. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/rtf.py +0 -0
  27. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/svg.py +0 -0
  28. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/tex.py +0 -0
  29. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/txt.py +0 -0
  30. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/xls.py +0 -0
  31. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/xlsx.py +0 -0
  32. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/xml.py +0 -0
  33. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt/pyxtxt.py +0 -0
  34. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  35. {pyxtxt-0.2.2.1 → pyxtxt-0.2.4}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.2.2.1
3
+ Version: 0.2.4
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -58,6 +58,26 @@ Requires-Dist: beautifulsoup4; extra == "html"
58
58
  Requires-Dist: lxml; extra == "html"
59
59
  Provides-Extra: doc
60
60
  Requires-Dist: textract; extra == "doc"
61
+ Provides-Extra: markdown
62
+ Requires-Dist: markdown; extra == "markdown"
63
+ Requires-Dist: beautifulsoup4; extra == "markdown"
64
+ Provides-Extra: epub
65
+ Requires-Dist: ebooklib; extra == "epub"
66
+ Requires-Dist: beautifulsoup4; extra == "epub"
67
+ Provides-Extra: rtf
68
+ Requires-Dist: striprtf; extra == "rtf"
69
+ Provides-Extra: email
70
+ Requires-Dist: beautifulsoup4; extra == "email"
71
+ Provides-Extra: outlook
72
+ Requires-Dist: extract-msg; extra == "outlook"
73
+ Requires-Dist: beautifulsoup4; extra == "outlook"
74
+ Provides-Extra: latex
75
+ Requires-Dist: pylatexenc; extra == "latex"
76
+ Provides-Extra: audio
77
+ Requires-Dist: openai-whisper; extra == "audio"
78
+ Provides-Extra: ocr
79
+ Requires-Dist: easyocr; extra == "ocr"
80
+ Requires-Dist: pillow; extra == "ocr"
61
81
  Provides-Extra: all
62
82
  Requires-Dist: textract; extra == "all"
63
83
  Requires-Dist: PyMuPDF; extra == "all"
@@ -68,6 +88,14 @@ Requires-Dist: xlrd; extra == "all"
68
88
  Requires-Dist: odfpy; extra == "all"
69
89
  Requires-Dist: beautifulsoup4; extra == "all"
70
90
  Requires-Dist: lxml; extra == "all"
91
+ Requires-Dist: markdown; extra == "all"
92
+ Requires-Dist: ebooklib; extra == "all"
93
+ Requires-Dist: striprtf; extra == "all"
94
+ Requires-Dist: extract-msg; extra == "all"
95
+ Requires-Dist: pylatexenc; extra == "all"
96
+ Requires-Dist: openai-whisper; extra == "all"
97
+ Requires-Dist: easyocr; extra == "all"
98
+ Requires-Dist: pillow; extra == "all"
71
99
  Dynamic: license-file
72
100
 
73
101
  # PyxTxt
@@ -77,16 +105,18 @@ Dynamic: license-file
77
105
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
78
106
 
79
107
  **PyxTxt** is a simple and powerful Python library to extract text from various file formats.
80
- It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
108
+ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio/video transcription**, **OCR from images**, and more.
81
109
 
82
- **NEW in v0.2.1+**: Enhanced support for web content, byte streams, and requests integration!
110
+ **NEW in v0.2.4**: Added video transcription support! Now supports both audio and video files using Whisper.
83
111
 
84
112
  ---
85
113
 
86
114
  ## ✨ Features
87
115
 
88
116
  - **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
89
- - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
117
+ - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
118
+ - **Audio & Video transcription**: MP3, WAV, M4A, FLAC, MP4, MOV, AVI, WebM, MKV and more using OpenAI Whisper
119
+ - **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
90
120
  - **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
91
121
  - **Web-ready**: Direct support for downloading and extracting text from URLs
92
122
  - **Memory efficient**: Process files without saving to disk
@@ -103,8 +133,21 @@ pip install pyxtxt[all]
103
133
  ```
104
134
  or just the modules you need:
105
135
  ```bash
106
- pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
136
+ pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
107
137
  ```
138
+
139
+ ### Audio & OCR (Heavy Dependencies)
140
+ ```bash
141
+ # Audio transcription (~2GB download for Whisper models)
142
+ pip install pyxtxt[audio]
143
+
144
+ # OCR from images (~1GB download for EasyOCR models)
145
+ pip install pyxtxt[ocr]
146
+
147
+ # Both audio and OCR
148
+ pip install pyxtxt[audio,ocr]
149
+ ```
150
+
108
151
  Because needed libraries are common, installing the html module will also enable SVG and XML support.
109
152
  The architecture is designed to grow with new modules for additional formats.
110
153
  ## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
@@ -126,25 +169,24 @@ brew install libmagic
126
169
  Use python-magic-bin instead of python-magic for easier installation.
127
170
 
128
171
  ## 🛠️ Dependencies
129
- - PyMuPDF (fitz)
130
-
131
- - beautifulsoup4
132
-
133
- - python-docx
134
172
 
135
- - python-pptx
173
+ ### Core Dependencies
174
+ - python-magic (automatic file type detection)
136
175
 
137
- - odfpy
176
+ ### Optional Dependencies by Format
177
+ - **PDF**: PyMuPDF
178
+ - **Office**: python-docx, python-pptx, openpyxl, xlrd
179
+ - **Web/HTML**: beautifulsoup4, lxml
180
+ - **OpenDocument**: odfpy
181
+ - **Markdown**: markdown
182
+ - **EPUB**: ebooklib
183
+ - **RTF**: striprtf
184
+ - **Email**: extract-msg (for MSG files)
185
+ - **LaTeX**: pylatexenc
186
+ - **Audio**: openai-whisper (heavy ~2GB models)
187
+ - **OCR**: easyocr, pillow (heavy ~1GB models)
138
188
 
139
- - openpyxl
140
-
141
- - lxml
142
-
143
- - xlrd (<2.0.0)
144
-
145
- - python-magic
146
-
147
- Dependencies are automatically installed from pyproject.toml.
189
+ Dependencies are automatically installed based on selected optional groups.
148
190
 
149
191
  ## 📚 Usage Examples
150
192
 
@@ -180,6 +222,44 @@ text = xtxt(response)
180
222
  text = xtxt_from_url("https://example.com/document.pdf")
181
223
  ```
182
224
 
225
+ ### Audio & Video Transcription (NEW)
226
+ ```python
227
+ from pyxtxt import xtxt
228
+
229
+ # Transcribe audio files
230
+ text = xtxt("meeting_recording.mp3")
231
+ text = xtxt("interview.wav")
232
+ text = xtxt("podcast.m4a")
233
+
234
+ # Transcribe video files (extracts audio)
235
+ text = xtxt("presentation.mp4")
236
+ text = xtxt("conference_video.mov")
237
+ text = xtxt("webinar.avi")
238
+
239
+ # From web audio/video
240
+ import requests
241
+ audio_response = requests.get("https://example.com/audio.mp3")
242
+ text = xtxt(audio_response.content)
243
+
244
+ video_response = requests.get("https://example.com/video.mp4")
245
+ text = xtxt(video_response.content)
246
+ ```
247
+
248
+ ### OCR from Images (NEW)
249
+ ```python
250
+ from pyxtxt import xtxt
251
+
252
+ # Extract text from images
253
+ text = xtxt("scanned_document.png")
254
+ text = xtxt("screenshot.jpg")
255
+ text = xtxt("invoice.tiff")
256
+
257
+ # From web images
258
+ import requests
259
+ image_response = requests.get("https://example.com/document.png")
260
+ text = xtxt(image_response.content)
261
+ ```
262
+
183
263
  ### Show Available Formats
184
264
  ```python
185
265
  from pyxtxt import extxt_available_formats
@@ -203,6 +283,18 @@ text = xtxt(api_response.content)
203
283
  uploaded_bytes = request.files['document'].read()
204
284
  text = xtxt(uploaded_bytes)
205
285
 
286
+ # Audio/video transcription services
287
+ audio_response = requests.get("https://api.example.com/recording.mp3")
288
+ transcript = xtxt(audio_response.content)
289
+
290
+ # Video transcription from API
291
+ video_response = requests.get("https://api.example.com/meeting.mp4")
292
+ transcript = xtxt(video_response.content)
293
+
294
+ # OCR for uploaded images
295
+ image_bytes = request.files['receipt'].read()
296
+ text = xtxt(image_bytes)
297
+
206
298
  # Email attachments
207
299
  attachment_bytes = email_msg.get_payload(decode=True)
208
300
  text = xtxt(attachment_bytes)
@@ -257,6 +349,20 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
257
349
 
258
350
  ## 📊 Changelog
259
351
 
352
+ ### v0.2.4
353
+ - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
354
+ - ✅ **ENHANCED**: Audio transcription now supports video files
355
+ - ✅ Whisper automatically extracts audio track from videos
356
+ - ✅ Unified interface for both audio and video processing
357
+
358
+ ### v0.2.3
359
+ - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
360
+ - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
361
+ - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
362
+ - ✅ Separate optional dependencies for heavy features (audio/OCR)
363
+ - ✅ Performance optimizations with model caching
364
+ - ✅ Improved multilingual OCR support (Italian/English)
365
+
260
366
  ### v0.1.24+
261
367
  - ✅ Added support for `bytes` objects
262
368
  - ✅ Added support for `requests.Response` objects
@@ -5,16 +5,18 @@
5
5
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
6
6
 
7
7
  **PyxTxt** is a simple and powerful Python library to extract text from various file formats.
8
- It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
8
+ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio/video transcription**, **OCR from images**, and more.
9
9
 
10
- **NEW in v0.2.1+**: Enhanced support for web content, byte streams, and requests integration!
10
+ **NEW in v0.2.4**: Added video transcription support! Now supports both audio and video files using Whisper.
11
11
 
12
12
  ---
13
13
 
14
14
  ## ✨ Features
15
15
 
16
16
  - **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
17
- - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
17
+ - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
18
+ - **Audio & Video transcription**: MP3, WAV, M4A, FLAC, MP4, MOV, AVI, WebM, MKV and more using OpenAI Whisper
19
+ - **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
18
20
  - **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
19
21
  - **Web-ready**: Direct support for downloading and extracting text from URLs
20
22
  - **Memory efficient**: Process files without saving to disk
@@ -31,8 +33,21 @@ pip install pyxtxt[all]
31
33
  ```
32
34
  or just the modules you need:
33
35
  ```bash
34
- pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
36
+ pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
35
37
  ```
38
+
39
+ ### Audio & OCR (Heavy Dependencies)
40
+ ```bash
41
+ # Audio transcription (~2GB download for Whisper models)
42
+ pip install pyxtxt[audio]
43
+
44
+ # OCR from images (~1GB download for EasyOCR models)
45
+ pip install pyxtxt[ocr]
46
+
47
+ # Both audio and OCR
48
+ pip install pyxtxt[audio,ocr]
49
+ ```
50
+
36
51
  Because needed libraries are common, installing the html module will also enable SVG and XML support.
37
52
  The architecture is designed to grow with new modules for additional formats.
38
53
  ## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
@@ -54,25 +69,24 @@ brew install libmagic
54
69
  Use python-magic-bin instead of python-magic for easier installation.
55
70
 
56
71
  ## 🛠️ Dependencies
57
- - PyMuPDF (fitz)
58
-
59
- - beautifulsoup4
60
-
61
- - python-docx
62
72
 
63
- - python-pptx
73
+ ### Core Dependencies
74
+ - python-magic (automatic file type detection)
64
75
 
65
- - odfpy
76
+ ### Optional Dependencies by Format
77
+ - **PDF**: PyMuPDF
78
+ - **Office**: python-docx, python-pptx, openpyxl, xlrd
79
+ - **Web/HTML**: beautifulsoup4, lxml
80
+ - **OpenDocument**: odfpy
81
+ - **Markdown**: markdown
82
+ - **EPUB**: ebooklib
83
+ - **RTF**: striprtf
84
+ - **Email**: extract-msg (for MSG files)
85
+ - **LaTeX**: pylatexenc
86
+ - **Audio**: openai-whisper (heavy ~2GB models)
87
+ - **OCR**: easyocr, pillow (heavy ~1GB models)
66
88
 
67
- - openpyxl
68
-
69
- - lxml
70
-
71
- - xlrd (<2.0.0)
72
-
73
- - python-magic
74
-
75
- Dependencies are automatically installed from pyproject.toml.
89
+ Dependencies are automatically installed based on selected optional groups.
76
90
 
77
91
  ## 📚 Usage Examples
78
92
 
@@ -108,6 +122,44 @@ text = xtxt(response)
108
122
  text = xtxt_from_url("https://example.com/document.pdf")
109
123
  ```
110
124
 
125
+ ### Audio & Video Transcription (NEW)
126
+ ```python
127
+ from pyxtxt import xtxt
128
+
129
+ # Transcribe audio files
130
+ text = xtxt("meeting_recording.mp3")
131
+ text = xtxt("interview.wav")
132
+ text = xtxt("podcast.m4a")
133
+
134
+ # Transcribe video files (extracts audio)
135
+ text = xtxt("presentation.mp4")
136
+ text = xtxt("conference_video.mov")
137
+ text = xtxt("webinar.avi")
138
+
139
+ # From web audio/video
140
+ import requests
141
+ audio_response = requests.get("https://example.com/audio.mp3")
142
+ text = xtxt(audio_response.content)
143
+
144
+ video_response = requests.get("https://example.com/video.mp4")
145
+ text = xtxt(video_response.content)
146
+ ```
147
+
148
+ ### OCR from Images (NEW)
149
+ ```python
150
+ from pyxtxt import xtxt
151
+
152
+ # Extract text from images
153
+ text = xtxt("scanned_document.png")
154
+ text = xtxt("screenshot.jpg")
155
+ text = xtxt("invoice.tiff")
156
+
157
+ # From web images
158
+ import requests
159
+ image_response = requests.get("https://example.com/document.png")
160
+ text = xtxt(image_response.content)
161
+ ```
162
+
111
163
  ### Show Available Formats
112
164
  ```python
113
165
  from pyxtxt import extxt_available_formats
@@ -131,6 +183,18 @@ text = xtxt(api_response.content)
131
183
  uploaded_bytes = request.files['document'].read()
132
184
  text = xtxt(uploaded_bytes)
133
185
 
186
+ # Audio/video transcription services
187
+ audio_response = requests.get("https://api.example.com/recording.mp3")
188
+ transcript = xtxt(audio_response.content)
189
+
190
+ # Video transcription from API
191
+ video_response = requests.get("https://api.example.com/meeting.mp4")
192
+ transcript = xtxt(video_response.content)
193
+
194
+ # OCR for uploaded images
195
+ image_bytes = request.files['receipt'].read()
196
+ text = xtxt(image_bytes)
197
+
134
198
  # Email attachments
135
199
  attachment_bytes = email_msg.get_payload(decode=True)
136
200
  text = xtxt(attachment_bytes)
@@ -185,6 +249,20 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
185
249
 
186
250
  ## 📊 Changelog
187
251
 
252
+ ### v0.2.4
253
+ - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
254
+ - ✅ **ENHANCED**: Audio transcription now supports video files
255
+ - ✅ Whisper automatically extracts audio track from videos
256
+ - ✅ Unified interface for both audio and video processing
257
+
258
+ ### v0.2.3
259
+ - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
260
+ - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
261
+ - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
262
+ - ✅ Separate optional dependencies for heavy features (audio/OCR)
263
+ - ✅ Performance optimizations with model caching
264
+ - ✅ Improved multilingual OCR support (Italian/English)
265
+
188
266
  ### v0.1.24+
189
267
  - ✅ Added support for `bytes` objects
190
268
  - ✅ Added support for `requests.Response` objects
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.2.2.1"
3
+ version = "0.2.4"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -52,6 +52,34 @@ html = [
52
52
  doc = [
53
53
  "textract",
54
54
  ]
55
+ markdown = [
56
+ "markdown",
57
+ "beautifulsoup4",
58
+ ]
59
+ epub = [
60
+ "ebooklib",
61
+ "beautifulsoup4",
62
+ ]
63
+ rtf = [
64
+ "striprtf",
65
+ ]
66
+ email = [
67
+ "beautifulsoup4",
68
+ ]
69
+ outlook = [
70
+ "extract-msg",
71
+ "beautifulsoup4",
72
+ ]
73
+ latex = [
74
+ "pylatexenc",
75
+ ]
76
+ audio = [
77
+ "openai-whisper",
78
+ ]
79
+ ocr = [
80
+ "easyocr",
81
+ "pillow",
82
+ ]
55
83
  all = [
56
84
  "textract",
57
85
  "PyMuPDF",
@@ -62,6 +90,14 @@ all = [
62
90
  "odfpy",
63
91
  "beautifulsoup4",
64
92
  "lxml",
93
+ "markdown",
94
+ "ebooklib",
95
+ "striprtf",
96
+ "extract-msg",
97
+ "pylatexenc",
98
+ "openai-whisper",
99
+ "easyocr",
100
+ "pillow",
65
101
  ]
66
102
 
67
103
  [build-system]
@@ -0,0 +1,79 @@
1
+ # pyxtxt/extractors/audio_whisper.py
2
+ from . import register_extractor
3
+ import tempfile
4
+ import os
5
+
6
+ try:
7
+ import whisper
8
+ except ImportError:
9
+ whisper = None
10
+
11
+ if whisper:
12
+ _whisper_model = None
13
+
14
+ def _get_model():
15
+ global _whisper_model
16
+ if _whisper_model is None:
17
+ _whisper_model = whisper.load_model("base")
18
+ return _whisper_model
19
+
20
+ def xtxt_audio_whisper(file_buffer):
21
+ try:
22
+ # Usa un suffixe generico - Whisper + FFmpeg gestiscono il formato
23
+ with tempfile.NamedTemporaryFile(delete=False) as temp_file:
24
+ temp_file.write(file_buffer.read())
25
+ temp_file.flush()
26
+
27
+ model = _get_model()
28
+ result = model.transcribe(
29
+ temp_file.name,
30
+ language=None,
31
+ task="transcribe",
32
+ temperature=[0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
33
+ word_timestamps=True
34
+ )
35
+
36
+ os.unlink(temp_file.name)
37
+ text_with_languages = []
38
+ current_lang = None
39
+
40
+ for segment in result['segments']:
41
+ # Rileva cambio di lingua (logica semplificata)
42
+ if 'language' in segment and segment['language'] != current_lang:
43
+ current_lang = segment['language']
44
+ text_with_languages.append(f"\n[{current_lang.upper()}]")
45
+
46
+ text_with_languages.append(segment['text'])
47
+
48
+ return " ".join(text_with_languages).strip()
49
+
50
+ except Exception as e:
51
+ print(f"⚠️ Error while extracting audio with Whisper: {e}")
52
+ return ""
53
+
54
+ # Registra per tutti i formati audio comuni
55
+ audio_formats = [
56
+ "audio/wav", "audio/wave",
57
+ "audio/mp3", "audio/mpeg",
58
+ "audio/m4a", "audio/mp4",
59
+ "audio/flac",
60
+ "audio/ogg", "audio/ogg-vorbis",
61
+ "audio/opus",
62
+ "audio/aac",
63
+ "audio/wma",
64
+ "audio/webm"
65
+ ]
66
+
67
+ for format_type in audio_formats:
68
+ register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio")
69
+ # Aggiungi questi formati video al tuo registro
70
+ video_audio_formats = [
71
+ "video/mp4",
72
+ "video/quicktime", # .mov
73
+ "video/x-msvideo", # .avi
74
+ "video/webm",
75
+ "video/mkv"
76
+ ]
77
+
78
+ for format_type in video_audio_formats:
79
+ register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio from Video")
@@ -0,0 +1,52 @@
1
+ # pyxtxt/extractors/image_ocr.py
2
+ from . import register_extractor
3
+ from io import BytesIO
4
+
5
+ try:
6
+ import easyocr
7
+ from PIL import Image
8
+ except ImportError:
9
+ easyocr = None
10
+ Image = None
11
+
12
+ if easyocr and Image:
13
+ # Inizializza il reader OCR una volta sola
14
+ _ocr_reader = None
15
+
16
+ def _get_ocr_reader():
17
+ global _ocr_reader
18
+ if _ocr_reader is None:
19
+ _ocr_reader = easyocr.Reader(['it', 'en'], gpu=False)
20
+ return _ocr_reader
21
+
22
+ def xtxt_image_ocr(file_buffer):
23
+ try:
24
+ # Converti il buffer in immagine PIL
25
+ image = Image.open(BytesIO(file_buffer.read()))
26
+
27
+ # Converti in RGB se necessario
28
+ if image.mode != 'RGB':
29
+ image = image.convert('RGB')
30
+
31
+ reader = _get_ocr_reader()
32
+ results = reader.readtext(image)
33
+
34
+ # Estrai solo il testo, ordinato per posizione verticale
35
+ texts = []
36
+ for (bbox, text, confidence) in results:
37
+ if confidence > 0.3: # Filtra testo con bassa confidenza
38
+ texts.append(text)
39
+
40
+ return "\n".join(texts)
41
+
42
+ except Exception as e:
43
+ print(f"⚠️ Error while extracting text from image: {e}")
44
+ return ""
45
+
46
+ # Registra per i formati immagine più comuni
47
+ register_extractor("image/jpeg", xtxt_image_ocr, name="OCR")
48
+ register_extractor("image/jpg", xtxt_image_ocr, name="OCR")
49
+ register_extractor("image/png", xtxt_image_ocr, name="OCR")
50
+ register_extractor("image/bmp", xtxt_image_ocr, name="OCR")
51
+ register_extractor("image/tiff", xtxt_image_ocr, name="OCR")
52
+ register_extractor("image/webp", xtxt_image_ocr, name="OCR")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.2.2.1
3
+ Version: 0.2.4
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -58,6 +58,26 @@ Requires-Dist: beautifulsoup4; extra == "html"
58
58
  Requires-Dist: lxml; extra == "html"
59
59
  Provides-Extra: doc
60
60
  Requires-Dist: textract; extra == "doc"
61
+ Provides-Extra: markdown
62
+ Requires-Dist: markdown; extra == "markdown"
63
+ Requires-Dist: beautifulsoup4; extra == "markdown"
64
+ Provides-Extra: epub
65
+ Requires-Dist: ebooklib; extra == "epub"
66
+ Requires-Dist: beautifulsoup4; extra == "epub"
67
+ Provides-Extra: rtf
68
+ Requires-Dist: striprtf; extra == "rtf"
69
+ Provides-Extra: email
70
+ Requires-Dist: beautifulsoup4; extra == "email"
71
+ Provides-Extra: outlook
72
+ Requires-Dist: extract-msg; extra == "outlook"
73
+ Requires-Dist: beautifulsoup4; extra == "outlook"
74
+ Provides-Extra: latex
75
+ Requires-Dist: pylatexenc; extra == "latex"
76
+ Provides-Extra: audio
77
+ Requires-Dist: openai-whisper; extra == "audio"
78
+ Provides-Extra: ocr
79
+ Requires-Dist: easyocr; extra == "ocr"
80
+ Requires-Dist: pillow; extra == "ocr"
61
81
  Provides-Extra: all
62
82
  Requires-Dist: textract; extra == "all"
63
83
  Requires-Dist: PyMuPDF; extra == "all"
@@ -68,6 +88,14 @@ Requires-Dist: xlrd; extra == "all"
68
88
  Requires-Dist: odfpy; extra == "all"
69
89
  Requires-Dist: beautifulsoup4; extra == "all"
70
90
  Requires-Dist: lxml; extra == "all"
91
+ Requires-Dist: markdown; extra == "all"
92
+ Requires-Dist: ebooklib; extra == "all"
93
+ Requires-Dist: striprtf; extra == "all"
94
+ Requires-Dist: extract-msg; extra == "all"
95
+ Requires-Dist: pylatexenc; extra == "all"
96
+ Requires-Dist: openai-whisper; extra == "all"
97
+ Requires-Dist: easyocr; extra == "all"
98
+ Requires-Dist: pillow; extra == "all"
71
99
  Dynamic: license-file
72
100
 
73
101
  # PyxTxt
@@ -77,16 +105,18 @@ Dynamic: license-file
77
105
  [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
78
106
 
79
107
  **PyxTxt** is a simple and powerful Python library to extract text from various file formats.
80
- It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, and more.
108
+ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio/video transcription**, **OCR from images**, and more.
81
109
 
82
- **NEW in v0.2.1+**: Enhanced support for web content, byte streams, and requests integration!
110
+ **NEW in v0.2.4**: Added video transcription support! Now supports both audio and video files using Whisper.
83
111
 
84
112
  ---
85
113
 
86
114
  ## ✨ Features
87
115
 
88
116
  - **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
89
- - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .ppt, .doc)
117
+ - **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
118
+ - **Audio & Video transcription**: MP3, WAV, M4A, FLAC, MP4, MOV, AVI, WebM, MKV and more using OpenAI Whisper
119
+ - **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
90
120
  - **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
91
121
  - **Web-ready**: Direct support for downloading and extracting text from URLs
92
122
  - **Memory efficient**: Process files without saving to disk
@@ -103,8 +133,21 @@ pip install pyxtxt[all]
103
133
  ```
104
134
  or just the modules you need:
105
135
  ```bash
106
- pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
136
+ pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
107
137
  ```
138
+
139
+ ### Audio & OCR (Heavy Dependencies)
140
+ ```bash
141
+ # Audio transcription (~2GB download for Whisper models)
142
+ pip install pyxtxt[audio]
143
+
144
+ # OCR from images (~1GB download for EasyOCR models)
145
+ pip install pyxtxt[ocr]
146
+
147
+ # Both audio and OCR
148
+ pip install pyxtxt[audio,ocr]
149
+ ```
150
+
108
151
  Because needed libraries are common, installing the html module will also enable SVG and XML support.
109
152
  The architecture is designed to grow with new modules for additional formats.
110
153
  ## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
@@ -126,25 +169,24 @@ brew install libmagic
126
169
  Use python-magic-bin instead of python-magic for easier installation.
127
170
 
128
171
  ## 🛠️ Dependencies
129
- - PyMuPDF (fitz)
130
-
131
- - beautifulsoup4
132
-
133
- - python-docx
134
172
 
135
- - python-pptx
173
+ ### Core Dependencies
174
+ - python-magic (automatic file type detection)
136
175
 
137
- - odfpy
176
+ ### Optional Dependencies by Format
177
+ - **PDF**: PyMuPDF
178
+ - **Office**: python-docx, python-pptx, openpyxl, xlrd
179
+ - **Web/HTML**: beautifulsoup4, lxml
180
+ - **OpenDocument**: odfpy
181
+ - **Markdown**: markdown
182
+ - **EPUB**: ebooklib
183
+ - **RTF**: striprtf
184
+ - **Email**: extract-msg (for MSG files)
185
+ - **LaTeX**: pylatexenc
186
+ - **Audio**: openai-whisper (heavy ~2GB models)
187
+ - **OCR**: easyocr, pillow (heavy ~1GB models)
138
188
 
139
- - openpyxl
140
-
141
- - lxml
142
-
143
- - xlrd (<2.0.0)
144
-
145
- - python-magic
146
-
147
- Dependencies are automatically installed from pyproject.toml.
189
+ Dependencies are automatically installed based on selected optional groups.
148
190
 
149
191
  ## 📚 Usage Examples
150
192
 
@@ -180,6 +222,44 @@ text = xtxt(response)
180
222
  text = xtxt_from_url("https://example.com/document.pdf")
181
223
  ```
182
224
 
225
+ ### Audio & Video Transcription (NEW)
226
+ ```python
227
+ from pyxtxt import xtxt
228
+
229
+ # Transcribe audio files
230
+ text = xtxt("meeting_recording.mp3")
231
+ text = xtxt("interview.wav")
232
+ text = xtxt("podcast.m4a")
233
+
234
+ # Transcribe video files (extracts audio)
235
+ text = xtxt("presentation.mp4")
236
+ text = xtxt("conference_video.mov")
237
+ text = xtxt("webinar.avi")
238
+
239
+ # From web audio/video
240
+ import requests
241
+ audio_response = requests.get("https://example.com/audio.mp3")
242
+ text = xtxt(audio_response.content)
243
+
244
+ video_response = requests.get("https://example.com/video.mp4")
245
+ text = xtxt(video_response.content)
246
+ ```
247
+
248
+ ### OCR from Images (NEW)
249
+ ```python
250
+ from pyxtxt import xtxt
251
+
252
+ # Extract text from images
253
+ text = xtxt("scanned_document.png")
254
+ text = xtxt("screenshot.jpg")
255
+ text = xtxt("invoice.tiff")
256
+
257
+ # From web images
258
+ import requests
259
+ image_response = requests.get("https://example.com/document.png")
260
+ text = xtxt(image_response.content)
261
+ ```
262
+
183
263
  ### Show Available Formats
184
264
  ```python
185
265
  from pyxtxt import extxt_available_formats
@@ -203,6 +283,18 @@ text = xtxt(api_response.content)
203
283
  uploaded_bytes = request.files['document'].read()
204
284
  text = xtxt(uploaded_bytes)
205
285
 
286
+ # Audio/video transcription services
287
+ audio_response = requests.get("https://api.example.com/recording.mp3")
288
+ transcript = xtxt(audio_response.content)
289
+
290
+ # Video transcription from API
291
+ video_response = requests.get("https://api.example.com/meeting.mp4")
292
+ transcript = xtxt(video_response.content)
293
+
294
+ # OCR for uploaded images
295
+ image_bytes = request.files['receipt'].read()
296
+ text = xtxt(image_bytes)
297
+
206
298
  # Email attachments
207
299
  attachment_bytes = email_msg.get_payload(decode=True)
208
300
  text = xtxt(attachment_bytes)
@@ -257,6 +349,20 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
257
349
 
258
350
  ## 📊 Changelog
259
351
 
352
+ ### v0.2.4
353
+ - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
354
+ - ✅ **ENHANCED**: Audio transcription now supports video files
355
+ - ✅ Whisper automatically extracts audio track from videos
356
+ - ✅ Unified interface for both audio and video processing
357
+
358
+ ### v0.2.3
359
+ - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
360
+ - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
361
+ - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
362
+ - ✅ Separate optional dependencies for heavy features (audio/OCR)
363
+ - ✅ Performance optimizations with model caching
364
+ - ✅ Improved multilingual OCR support (Italian/English)
365
+
260
366
  ### v0.1.24+
261
367
  - ✅ Added support for `bytes` objects
262
368
  - ✅ Added support for `requests.Response` objects
@@ -12,6 +12,7 @@ src/pyxtxt.egg-info/dependency_links.txt
12
12
  src/pyxtxt.egg-info/requires.txt
13
13
  src/pyxtxt.egg-info/top_level.txt
14
14
  src/pyxtxt/estrattori/__init__.py
15
+ src/pyxtxt/estrattori/audio.py
15
16
  src/pyxtxt/estrattori/doc.py
16
17
  src/pyxtxt/estrattori/docx.py
17
18
  src/pyxtxt/estrattori/eml.py
@@ -19,6 +20,7 @@ src/pyxtxt/estrattori/epub.py
19
20
  src/pyxtxt/estrattori/html.py
20
21
  src/pyxtxt/estrattori/md.py
21
22
  src/pyxtxt/estrattori/msg.py
23
+ src/pyxtxt/estrattori/ocr.py
22
24
  src/pyxtxt/estrattori/odt.py
23
25
  src/pyxtxt/estrattori/pdf.py
24
26
  src/pyxtxt/estrattori/pptx.py
@@ -15,6 +15,17 @@ xlrd
15
15
  odfpy
16
16
  beautifulsoup4
17
17
  lxml
18
+ markdown
19
+ ebooklib
20
+ striprtf
21
+ extract-msg
22
+ pylatexenc
23
+ openai-whisper
24
+ easyocr
25
+ pillow
26
+
27
+ [audio]
28
+ openai-whisper
18
29
 
19
30
  [doc]
20
31
  textract
@@ -22,19 +33,44 @@ textract
22
33
  [docx]
23
34
  python-docx
24
35
 
36
+ [email]
37
+ beautifulsoup4
38
+
39
+ [epub]
40
+ ebooklib
41
+ beautifulsoup4
42
+
25
43
  [html]
26
44
  beautifulsoup4
27
45
  lxml
28
46
 
47
+ [latex]
48
+ pylatexenc
49
+
50
+ [markdown]
51
+ markdown
52
+ beautifulsoup4
53
+
54
+ [ocr]
55
+ easyocr
56
+ pillow
57
+
29
58
  [odf]
30
59
  odfpy
31
60
 
61
+ [outlook]
62
+ extract-msg
63
+ beautifulsoup4
64
+
32
65
  [pdf]
33
66
  PyMuPDF
34
67
 
35
68
  [presentation]
36
69
  python-pptx
37
70
 
71
+ [rtf]
72
+ striprtf
73
+
38
74
  [spreadsheet]
39
75
  openpyxl
40
76
  xlrd
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes