pyxtxt 0.2.3__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyxtxt-0.2.3/src/pyxtxt.egg-info → pyxtxt-0.2.4}/PKG-INFO +25 -7
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/README.md +24 -6
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/pyproject.toml +1 -1
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/audio.py +29 -2
- {pyxtxt-0.2.3 → pyxtxt-0.2.4/src/pyxtxt.egg-info}/PKG-INFO +25 -7
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/LICENSE +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/MANIFEST.in +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/examples.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/setup.cfg +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/__init__.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/core.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/doc.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/docx.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/eml.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/epub.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/md.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/msg.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/ocr.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/odt.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/pptx.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/rtf.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/svg.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/tex.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/xlsx.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt/pyxtxt.py +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt.egg-info/requires.txt +0 -0
- {pyxtxt-0.2.3 → pyxtxt-0.2.4}/src/pyxtxt.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -105,9 +105,9 @@ Dynamic: license-file
|
|
|
105
105
|
[](https://opensource.org/licenses/MIT)
|
|
106
106
|
|
|
107
107
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
108
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
|
|
108
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio/video transcription**, **OCR from images**, and more.
|
|
109
109
|
|
|
110
|
-
**NEW in v0.2.
|
|
110
|
+
**NEW in v0.2.4**: Added video transcription support! Now supports both audio and video files using Whisper.
|
|
111
111
|
|
|
112
112
|
---
|
|
113
113
|
|
|
@@ -115,7 +115,7 @@ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **a
|
|
|
115
115
|
|
|
116
116
|
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
117
117
|
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
|
|
118
|
-
- **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
|
|
118
|
+
- **Audio & Video transcription**: MP3, WAV, M4A, FLAC, MP4, MOV, AVI, WebM, MKV and more using OpenAI Whisper
|
|
119
119
|
- **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
|
|
120
120
|
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
121
121
|
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
@@ -222,7 +222,7 @@ text = xtxt(response)
|
|
|
222
222
|
text = xtxt_from_url("https://example.com/document.pdf")
|
|
223
223
|
```
|
|
224
224
|
|
|
225
|
-
### Audio Transcription (NEW)
|
|
225
|
+
### Audio & Video Transcription (NEW)
|
|
226
226
|
```python
|
|
227
227
|
from pyxtxt import xtxt
|
|
228
228
|
|
|
@@ -231,10 +231,18 @@ text = xtxt("meeting_recording.mp3")
|
|
|
231
231
|
text = xtxt("interview.wav")
|
|
232
232
|
text = xtxt("podcast.m4a")
|
|
233
233
|
|
|
234
|
-
#
|
|
234
|
+
# Transcribe video files (extracts audio)
|
|
235
|
+
text = xtxt("presentation.mp4")
|
|
236
|
+
text = xtxt("conference_video.mov")
|
|
237
|
+
text = xtxt("webinar.avi")
|
|
238
|
+
|
|
239
|
+
# From web audio/video
|
|
235
240
|
import requests
|
|
236
241
|
audio_response = requests.get("https://example.com/audio.mp3")
|
|
237
242
|
text = xtxt(audio_response.content)
|
|
243
|
+
|
|
244
|
+
video_response = requests.get("https://example.com/video.mp4")
|
|
245
|
+
text = xtxt(video_response.content)
|
|
238
246
|
```
|
|
239
247
|
|
|
240
248
|
### OCR from Images (NEW)
|
|
@@ -279,6 +287,10 @@ text = xtxt(uploaded_bytes)
|
|
|
279
287
|
audio_response = requests.get("https://api.example.com/recording.mp3")
|
|
280
288
|
transcript = xtxt(audio_response.content)
|
|
281
289
|
|
|
290
|
+
# Video transcription from API
|
|
291
|
+
video_response = requests.get("https://api.example.com/meeting.mp4")
|
|
292
|
+
transcript = xtxt(video_response.content)
|
|
293
|
+
|
|
282
294
|
# OCR for uploaded images
|
|
283
295
|
image_bytes = request.files['receipt'].read()
|
|
284
296
|
text = xtxt(image_bytes)
|
|
@@ -337,7 +349,13 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
337
349
|
|
|
338
350
|
## 📊 Changelog
|
|
339
351
|
|
|
340
|
-
### v0.2.
|
|
352
|
+
### v0.2.4
|
|
353
|
+
- ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
|
|
354
|
+
- ✅ **ENHANCED**: Audio transcription now supports video files
|
|
355
|
+
- ✅ Whisper automatically extracts audio track from videos
|
|
356
|
+
- ✅ Unified interface for both audio and video processing
|
|
357
|
+
|
|
358
|
+
### v0.2.3
|
|
341
359
|
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
342
360
|
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
343
361
|
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
|
|
@@ -5,9 +5,9 @@
|
|
|
5
5
|
[](https://opensource.org/licenses/MIT)
|
|
6
6
|
|
|
7
7
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
8
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
|
|
8
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio/video transcription**, **OCR from images**, and more.
|
|
9
9
|
|
|
10
|
-
**NEW in v0.2.
|
|
10
|
+
**NEW in v0.2.4**: Added video transcription support! Now supports both audio and video files using Whisper.
|
|
11
11
|
|
|
12
12
|
---
|
|
13
13
|
|
|
@@ -15,7 +15,7 @@ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **a
|
|
|
15
15
|
|
|
16
16
|
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
17
17
|
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
|
|
18
|
-
- **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
|
|
18
|
+
- **Audio & Video transcription**: MP3, WAV, M4A, FLAC, MP4, MOV, AVI, WebM, MKV and more using OpenAI Whisper
|
|
19
19
|
- **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
|
|
20
20
|
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
21
21
|
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
@@ -122,7 +122,7 @@ text = xtxt(response)
|
|
|
122
122
|
text = xtxt_from_url("https://example.com/document.pdf")
|
|
123
123
|
```
|
|
124
124
|
|
|
125
|
-
### Audio Transcription (NEW)
|
|
125
|
+
### Audio & Video Transcription (NEW)
|
|
126
126
|
```python
|
|
127
127
|
from pyxtxt import xtxt
|
|
128
128
|
|
|
@@ -131,10 +131,18 @@ text = xtxt("meeting_recording.mp3")
|
|
|
131
131
|
text = xtxt("interview.wav")
|
|
132
132
|
text = xtxt("podcast.m4a")
|
|
133
133
|
|
|
134
|
-
#
|
|
134
|
+
# Transcribe video files (extracts audio)
|
|
135
|
+
text = xtxt("presentation.mp4")
|
|
136
|
+
text = xtxt("conference_video.mov")
|
|
137
|
+
text = xtxt("webinar.avi")
|
|
138
|
+
|
|
139
|
+
# From web audio/video
|
|
135
140
|
import requests
|
|
136
141
|
audio_response = requests.get("https://example.com/audio.mp3")
|
|
137
142
|
text = xtxt(audio_response.content)
|
|
143
|
+
|
|
144
|
+
video_response = requests.get("https://example.com/video.mp4")
|
|
145
|
+
text = xtxt(video_response.content)
|
|
138
146
|
```
|
|
139
147
|
|
|
140
148
|
### OCR from Images (NEW)
|
|
@@ -179,6 +187,10 @@ text = xtxt(uploaded_bytes)
|
|
|
179
187
|
audio_response = requests.get("https://api.example.com/recording.mp3")
|
|
180
188
|
transcript = xtxt(audio_response.content)
|
|
181
189
|
|
|
190
|
+
# Video transcription from API
|
|
191
|
+
video_response = requests.get("https://api.example.com/meeting.mp4")
|
|
192
|
+
transcript = xtxt(video_response.content)
|
|
193
|
+
|
|
182
194
|
# OCR for uploaded images
|
|
183
195
|
image_bytes = request.files['receipt'].read()
|
|
184
196
|
text = xtxt(image_bytes)
|
|
@@ -237,7 +249,13 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
237
249
|
|
|
238
250
|
## 📊 Changelog
|
|
239
251
|
|
|
240
|
-
### v0.2.
|
|
252
|
+
### v0.2.4
|
|
253
|
+
- ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
|
|
254
|
+
- ✅ **ENHANCED**: Audio transcription now supports video files
|
|
255
|
+
- ✅ Whisper automatically extracts audio track from videos
|
|
256
|
+
- ✅ Unified interface for both audio and video processing
|
|
257
|
+
|
|
258
|
+
### v0.2.3
|
|
241
259
|
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
242
260
|
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
243
261
|
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
|
|
@@ -25,11 +25,27 @@ if whisper:
|
|
|
25
25
|
temp_file.flush()
|
|
26
26
|
|
|
27
27
|
model = _get_model()
|
|
28
|
-
result = model.transcribe(
|
|
28
|
+
result = model.transcribe(
|
|
29
|
+
temp_file.name,
|
|
30
|
+
language=None,
|
|
31
|
+
task="transcribe",
|
|
32
|
+
temperature=[0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
|
|
33
|
+
word_timestamps=True
|
|
34
|
+
)
|
|
29
35
|
|
|
30
36
|
os.unlink(temp_file.name)
|
|
37
|
+
text_with_languages = []
|
|
38
|
+
current_lang = None
|
|
39
|
+
|
|
40
|
+
for segment in result['segments']:
|
|
41
|
+
# Rileva cambio di lingua (logica semplificata)
|
|
42
|
+
if 'language' in segment and segment['language'] != current_lang:
|
|
43
|
+
current_lang = segment['language']
|
|
44
|
+
text_with_languages.append(f"\n[{current_lang.upper()}]")
|
|
31
45
|
|
|
32
|
-
|
|
46
|
+
text_with_languages.append(segment['text'])
|
|
47
|
+
|
|
48
|
+
return " ".join(text_with_languages).strip()
|
|
33
49
|
|
|
34
50
|
except Exception as e:
|
|
35
51
|
print(f"⚠️ Error while extracting audio with Whisper: {e}")
|
|
@@ -50,3 +66,14 @@ if whisper:
|
|
|
50
66
|
|
|
51
67
|
for format_type in audio_formats:
|
|
52
68
|
register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio")
|
|
69
|
+
# Aggiungi questi formati video al tuo registro
|
|
70
|
+
video_audio_formats = [
|
|
71
|
+
"video/mp4",
|
|
72
|
+
"video/quicktime", # .mov
|
|
73
|
+
"video/x-msvideo", # .avi
|
|
74
|
+
"video/webm",
|
|
75
|
+
"video/mkv"
|
|
76
|
+
]
|
|
77
|
+
|
|
78
|
+
for format_type in video_audio_formats:
|
|
79
|
+
register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio from Video")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -105,9 +105,9 @@ Dynamic: license-file
|
|
|
105
105
|
[](https://opensource.org/licenses/MIT)
|
|
106
106
|
|
|
107
107
|
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
108
|
-
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio transcription**, **OCR from images**, and more.
|
|
108
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **audio/video transcription**, **OCR from images**, and more.
|
|
109
109
|
|
|
110
|
-
**NEW in v0.2.
|
|
110
|
+
**NEW in v0.2.4**: Added video transcription support! Now supports both audio and video files using Whisper.
|
|
111
111
|
|
|
112
112
|
---
|
|
113
113
|
|
|
@@ -115,7 +115,7 @@ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy Office files, **a
|
|
|
115
115
|
|
|
116
116
|
- **Multiple input types**: File paths, `io.BytesIO` buffers, raw `bytes` objects, and `requests.Response` objects
|
|
117
117
|
- **Wide format support**: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, Markdown, EPUB, RTF, EML, MSG, LaTeX, legacy Office files (.xls, .ppt, .doc)
|
|
118
|
-
- **Audio transcription**: MP3, WAV, M4A, FLAC and more using OpenAI Whisper
|
|
118
|
+
- **Audio & Video transcription**: MP3, WAV, M4A, FLAC, MP4, MOV, AVI, WebM, MKV and more using OpenAI Whisper
|
|
119
119
|
- **OCR from images**: JPEG, PNG, TIFF, BMP using EasyOCR with multilingual support
|
|
120
120
|
- **Automatic MIME detection**: Uses `python-magic` for intelligent file type recognition
|
|
121
121
|
- **Web-ready**: Direct support for downloading and extracting text from URLs
|
|
@@ -222,7 +222,7 @@ text = xtxt(response)
|
|
|
222
222
|
text = xtxt_from_url("https://example.com/document.pdf")
|
|
223
223
|
```
|
|
224
224
|
|
|
225
|
-
### Audio Transcription (NEW)
|
|
225
|
+
### Audio & Video Transcription (NEW)
|
|
226
226
|
```python
|
|
227
227
|
from pyxtxt import xtxt
|
|
228
228
|
|
|
@@ -231,10 +231,18 @@ text = xtxt("meeting_recording.mp3")
|
|
|
231
231
|
text = xtxt("interview.wav")
|
|
232
232
|
text = xtxt("podcast.m4a")
|
|
233
233
|
|
|
234
|
-
#
|
|
234
|
+
# Transcribe video files (extracts audio)
|
|
235
|
+
text = xtxt("presentation.mp4")
|
|
236
|
+
text = xtxt("conference_video.mov")
|
|
237
|
+
text = xtxt("webinar.avi")
|
|
238
|
+
|
|
239
|
+
# From web audio/video
|
|
235
240
|
import requests
|
|
236
241
|
audio_response = requests.get("https://example.com/audio.mp3")
|
|
237
242
|
text = xtxt(audio_response.content)
|
|
243
|
+
|
|
244
|
+
video_response = requests.get("https://example.com/video.mp4")
|
|
245
|
+
text = xtxt(video_response.content)
|
|
238
246
|
```
|
|
239
247
|
|
|
240
248
|
### OCR from Images (NEW)
|
|
@@ -279,6 +287,10 @@ text = xtxt(uploaded_bytes)
|
|
|
279
287
|
audio_response = requests.get("https://api.example.com/recording.mp3")
|
|
280
288
|
transcript = xtxt(audio_response.content)
|
|
281
289
|
|
|
290
|
+
# Video transcription from API
|
|
291
|
+
video_response = requests.get("https://api.example.com/meeting.mp4")
|
|
292
|
+
transcript = xtxt(video_response.content)
|
|
293
|
+
|
|
282
294
|
# OCR for uploaded images
|
|
283
295
|
image_bytes = request.files['receipt'].read()
|
|
284
296
|
text = xtxt(image_bytes)
|
|
@@ -337,7 +349,13 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
337
349
|
|
|
338
350
|
## 📊 Changelog
|
|
339
351
|
|
|
340
|
-
### v0.2.
|
|
352
|
+
### v0.2.4
|
|
353
|
+
- ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
|
|
354
|
+
- ✅ **ENHANCED**: Audio transcription now supports video files
|
|
355
|
+
- ✅ Whisper automatically extracts audio track from videos
|
|
356
|
+
- ✅ Unified interface for both audio and video processing
|
|
357
|
+
|
|
358
|
+
### v0.2.3
|
|
341
359
|
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
342
360
|
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
343
361
|
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|