pyxtxt 0.3.7__tar.gz → 0.3.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyxtxt-0.3.7/src/pyxtxt.egg-info → pyxtxt-0.3.8}/PKG-INFO +31 -11
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/README.md +30 -10
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/pyproject.toml +1 -1
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/core.py +30 -4
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/audio.py +34 -37
- pyxtxt-0.3.8/src/pyxtxt/estrattori/docx.py +75 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/exif.py +51 -41
- pyxtxt-0.3.8/src/pyxtxt/estrattori/msg.py +31 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/ocr_ollama.py +143 -211
- pyxtxt-0.3.8/src/pyxtxt/estrattori/odt.py +30 -0
- pyxtxt-0.3.8/src/pyxtxt/estrattori/pptx.py +58 -0
- pyxtxt-0.3.8/src/pyxtxt/estrattori/svg.py +30 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8/src/pyxtxt.egg-info}/PKG-INFO +31 -11
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/SOURCES.txt +3 -1
- pyxtxt-0.3.8/tests/test_audio.py +66 -0
- pyxtxt-0.3.8/tests/test_extraction.py +289 -0
- pyxtxt-0.3.8/tests/test_ocr_ollama.py +69 -0
- pyxtxt-0.3.7/src/pyxtxt/estrattori/docx.py +0 -35
- pyxtxt-0.3.7/src/pyxtxt/estrattori/msg.py +0 -41
- pyxtxt-0.3.7/src/pyxtxt/estrattori/odt.py +0 -27
- pyxtxt-0.3.7/src/pyxtxt/estrattori/pptx.py +0 -40
- pyxtxt-0.3.7/src/pyxtxt/estrattori/svg.py +0 -23
- pyxtxt-0.3.7/tests/test_extraction.py +0 -122
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/LICENSE +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/setup.cfg +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/__init__.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/doc.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/eml.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/epub.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/md.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/ocr.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/rtf.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/tex.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/xlsx.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/examples.py +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/requires.txt +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/top_level.txt +0 -0
- {pyxtxt-0.3.7 → pyxtxt-0.3.8}/tests/test_import.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.8
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, etc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -102,7 +102,7 @@ text = xtxt("report.pdf")
|
|
|
102
102
|
|
|
103
103
|
## ✨ Features
|
|
104
104
|
|
|
105
|
-
- **One function for everything**: `xtxt()` accepts a file path, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
|
|
105
|
+
- **One function for everything**: `xtxt()` accepts a file path, a binary file object, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
|
|
106
106
|
- **Automatic type detection** with `python-magic`, refined by the file extension when libmagic is not specific enough (e.g. Markdown)
|
|
107
107
|
- **Modular dependencies**: each format is an optional extra, the core only needs `python-magic`
|
|
108
108
|
- **Office, web and document formats**: PDF, DOCX, PPTX, XLSX, XLS, ODT, HTML, XML, SVG, Markdown, EPUB, RTF, EML, MSG, LaTeX, DOC, TXT
|
|
@@ -117,8 +117,8 @@ text = xtxt("report.pdf")
|
|
|
117
117
|
| Format | Install extra | Notes |
|
|
118
118
|
|---|---|---|
|
|
119
119
|
| PDF | `pdf` | PyMuPDF |
|
|
120
|
-
| DOCX | `docx` |
|
|
121
|
-
| PPTX | `presentation` | Text
|
|
120
|
+
| DOCX | `docx` | Paragraphs and tables in document order, headers and footers |
|
|
121
|
+
| PPTX | `presentation` | Text boxes, grouped shapes, tables and speaker notes |
|
|
122
122
|
| XLSX, XLS | `spreadsheet` | Every row of every visible sheet, cells joined with ` \| ` |
|
|
123
123
|
| ODT | `odf` | |
|
|
124
124
|
| HTML | `html` | |
|
|
@@ -127,11 +127,11 @@ text = xtxt("report.pdf")
|
|
|
127
127
|
| EPUB | `epub` | |
|
|
128
128
|
| RTF | `rtf` | |
|
|
129
129
|
| EML | `email` | Plain-text and HTML parts |
|
|
130
|
-
| MSG (Outlook) | `outlook` | |
|
|
130
|
+
| MSG (Outlook) | `outlook` | Plain-text body, or the HTML body when there is no plain text |
|
|
131
131
|
| LaTeX | `latex` | |
|
|
132
132
|
| DOC (legacy Word) | — | Needs the `antiword` system tool |
|
|
133
133
|
| TXT and other `text/*` | — | Always available, decoded as UTF-8 |
|
|
134
|
-
| Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download |
|
|
134
|
+
| Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download. MP3, WAV, M4A, AAC, FLAC, OGG/Opus, AIFF, WMA, MP4, MOV, AVI, MKV, WebM |
|
|
135
135
|
| Images (OCR) | `ocr` or `ocr-ollama` | See [OCR from images](#-ocr-from-images) |
|
|
136
136
|
|
|
137
137
|
To list what is available in your installation:
|
|
@@ -208,9 +208,12 @@ from pyxtxt import xtxt
|
|
|
208
208
|
# From a file path
|
|
209
209
|
text = xtxt("document.pdf")
|
|
210
210
|
|
|
211
|
-
# From
|
|
211
|
+
# From a file object opened in binary mode
|
|
212
212
|
with open("document.docx", "rb") as f:
|
|
213
|
-
|
|
213
|
+
text = xtxt(f)
|
|
214
|
+
|
|
215
|
+
# From an in-memory buffer
|
|
216
|
+
buffer = io.BytesIO(docx_bytes)
|
|
214
217
|
text = xtxt(buffer)
|
|
215
218
|
|
|
216
219
|
# Give the buffer a name to help type detection (useful for Markdown, LaTeX, RTF)
|
|
@@ -344,12 +347,12 @@ print((files("pyxtxt") / "examples.py").read_text())
|
|
|
344
347
|
|
|
345
348
|
## ⚠️ Known limitations
|
|
346
349
|
|
|
347
|
-
- **Supported inputs**: file paths, `io.BytesIO`, `bytes` and `requests.Response`.
|
|
348
|
-
|
|
350
|
+
- **Supported inputs**: file paths, binary file objects, `io.BytesIO`, `bytes` and `requests.Response`.
|
|
351
|
+
File objects opened in text mode are rejected: open them with `"rb"`.
|
|
349
352
|
- **Type detection without a file name**: libmagic cannot tell apart some formats from raw bytes (legacy Office files
|
|
350
353
|
share the same signature; Markdown looks like plain text). Pass a file path, or set `buffer.name`, when possible.
|
|
351
354
|
- **Legacy PowerPoint (`.ppt`)** is not supported.
|
|
352
|
-
- **DOCX**: text
|
|
355
|
+
- **DOCX**: text boxes, footnotes and comments are not extracted. **PPTX**: text inside charts is not extracted.
|
|
353
356
|
- Errors are reported with messages printed to standard output and the functions return `None` or an empty string;
|
|
354
357
|
they do not raise exceptions.
|
|
355
358
|
|
|
@@ -407,6 +410,23 @@ Pull requests, issues and feedback are welcome.
|
|
|
407
410
|
|
|
408
411
|
## 📊 Changelog
|
|
409
412
|
|
|
413
|
+
### v0.3.8
|
|
414
|
+
- **NEW**: `xtxt()` accepts binary file objects, e.g. `xtxt(open("file.pdf", "rb"))`
|
|
415
|
+
- **FIXED**: XLSX passed as `bytes` or `BytesIO` was detected as a ZIP archive and rejected
|
|
416
|
+
- **FIXED**: WAV, M4A, AAC, AIFF, WMA and MKV files were never transcribed (libmagic reports them as `audio/x-wav`,
|
|
417
|
+
`audio/x-m4a`, `video/x-matroska`, ... which were not registered)
|
|
418
|
+
- **FIXED**: MSG extraction always failed (`extract_msg.Message` has no `extract()` method)
|
|
419
|
+
- **FIXED**: SVG extraction failed on `<text>` elements containing `<tspan>`
|
|
420
|
+
- **IMPROVED**: DOCX now includes tables (in document order), headers and footers
|
|
421
|
+
- **IMPROVED**: PPTX now includes grouped shapes, tables and speaker notes
|
|
422
|
+
- **IMPROVED**: ODT now includes headings and text inside spans, links and lists
|
|
423
|
+
- **IMPROVED**: EXIF output formats aperture, shutter speed and focal length as intended and reads PNG/WebP metadata
|
|
424
|
+
through Pillow's public API; the fake `image/*+exif` MIME types are gone from `extxt_available_formats()`
|
|
425
|
+
- **IMPROVED**: Whisper temporary files are always deleted, also when transcription fails
|
|
426
|
+
- **IMPROVED**: Ollama OCR stops trying fallback models when the server is not reachable;
|
|
427
|
+
`xtxt_image_with_confidence()` now uses the same context-aware prompts as `xtxt()`;
|
|
428
|
+
hallucination keywords match whole words (e.g. "roman" no longer fires on "romance")
|
|
429
|
+
|
|
410
430
|
### v0.3.7
|
|
411
431
|
- Python 3.10 or newer is now required; metadata no longer lists the end-of-life versions 3.7–3.9, which were never tested
|
|
412
432
|
- Python 3.14 is tested in CI and declared as supported
|
|
@@ -19,7 +19,7 @@ text = xtxt("report.pdf")
|
|
|
19
19
|
|
|
20
20
|
## ✨ Features
|
|
21
21
|
|
|
22
|
-
- **One function for everything**: `xtxt()` accepts a file path, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
|
|
22
|
+
- **One function for everything**: `xtxt()` accepts a file path, a binary file object, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
|
|
23
23
|
- **Automatic type detection** with `python-magic`, refined by the file extension when libmagic is not specific enough (e.g. Markdown)
|
|
24
24
|
- **Modular dependencies**: each format is an optional extra, the core only needs `python-magic`
|
|
25
25
|
- **Office, web and document formats**: PDF, DOCX, PPTX, XLSX, XLS, ODT, HTML, XML, SVG, Markdown, EPUB, RTF, EML, MSG, LaTeX, DOC, TXT
|
|
@@ -34,8 +34,8 @@ text = xtxt("report.pdf")
|
|
|
34
34
|
| Format | Install extra | Notes |
|
|
35
35
|
|---|---|---|
|
|
36
36
|
| PDF | `pdf` | PyMuPDF |
|
|
37
|
-
| DOCX | `docx` |
|
|
38
|
-
| PPTX | `presentation` | Text
|
|
37
|
+
| DOCX | `docx` | Paragraphs and tables in document order, headers and footers |
|
|
38
|
+
| PPTX | `presentation` | Text boxes, grouped shapes, tables and speaker notes |
|
|
39
39
|
| XLSX, XLS | `spreadsheet` | Every row of every visible sheet, cells joined with ` \| ` |
|
|
40
40
|
| ODT | `odf` | |
|
|
41
41
|
| HTML | `html` | |
|
|
@@ -44,11 +44,11 @@ text = xtxt("report.pdf")
|
|
|
44
44
|
| EPUB | `epub` | |
|
|
45
45
|
| RTF | `rtf` | |
|
|
46
46
|
| EML | `email` | Plain-text and HTML parts |
|
|
47
|
-
| MSG (Outlook) | `outlook` | |
|
|
47
|
+
| MSG (Outlook) | `outlook` | Plain-text body, or the HTML body when there is no plain text |
|
|
48
48
|
| LaTeX | `latex` | |
|
|
49
49
|
| DOC (legacy Word) | — | Needs the `antiword` system tool |
|
|
50
50
|
| TXT and other `text/*` | — | Always available, decoded as UTF-8 |
|
|
51
|
-
| Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download |
|
|
51
|
+
| Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download. MP3, WAV, M4A, AAC, FLAC, OGG/Opus, AIFF, WMA, MP4, MOV, AVI, MKV, WebM |
|
|
52
52
|
| Images (OCR) | `ocr` or `ocr-ollama` | See [OCR from images](#-ocr-from-images) |
|
|
53
53
|
|
|
54
54
|
To list what is available in your installation:
|
|
@@ -125,9 +125,12 @@ from pyxtxt import xtxt
|
|
|
125
125
|
# From a file path
|
|
126
126
|
text = xtxt("document.pdf")
|
|
127
127
|
|
|
128
|
-
# From
|
|
128
|
+
# From a file object opened in binary mode
|
|
129
129
|
with open("document.docx", "rb") as f:
|
|
130
|
-
|
|
130
|
+
text = xtxt(f)
|
|
131
|
+
|
|
132
|
+
# From an in-memory buffer
|
|
133
|
+
buffer = io.BytesIO(docx_bytes)
|
|
131
134
|
text = xtxt(buffer)
|
|
132
135
|
|
|
133
136
|
# Give the buffer a name to help type detection (useful for Markdown, LaTeX, RTF)
|
|
@@ -261,12 +264,12 @@ print((files("pyxtxt") / "examples.py").read_text())
|
|
|
261
264
|
|
|
262
265
|
## ⚠️ Known limitations
|
|
263
266
|
|
|
264
|
-
- **Supported inputs**: file paths, `io.BytesIO`, `bytes` and `requests.Response`.
|
|
265
|
-
|
|
267
|
+
- **Supported inputs**: file paths, binary file objects, `io.BytesIO`, `bytes` and `requests.Response`.
|
|
268
|
+
File objects opened in text mode are rejected: open them with `"rb"`.
|
|
266
269
|
- **Type detection without a file name**: libmagic cannot tell apart some formats from raw bytes (legacy Office files
|
|
267
270
|
share the same signature; Markdown looks like plain text). Pass a file path, or set `buffer.name`, when possible.
|
|
268
271
|
- **Legacy PowerPoint (`.ppt`)** is not supported.
|
|
269
|
-
- **DOCX**: text
|
|
272
|
+
- **DOCX**: text boxes, footnotes and comments are not extracted. **PPTX**: text inside charts is not extracted.
|
|
270
273
|
- Errors are reported with messages printed to standard output and the functions return `None` or an empty string;
|
|
271
274
|
they do not raise exceptions.
|
|
272
275
|
|
|
@@ -324,6 +327,23 @@ Pull requests, issues and feedback are welcome.
|
|
|
324
327
|
|
|
325
328
|
## 📊 Changelog
|
|
326
329
|
|
|
330
|
+
### v0.3.8
|
|
331
|
+
- **NEW**: `xtxt()` accepts binary file objects, e.g. `xtxt(open("file.pdf", "rb"))`
|
|
332
|
+
- **FIXED**: XLSX passed as `bytes` or `BytesIO` was detected as a ZIP archive and rejected
|
|
333
|
+
- **FIXED**: WAV, M4A, AAC, AIFF, WMA and MKV files were never transcribed (libmagic reports them as `audio/x-wav`,
|
|
334
|
+
`audio/x-m4a`, `video/x-matroska`, ... which were not registered)
|
|
335
|
+
- **FIXED**: MSG extraction always failed (`extract_msg.Message` has no `extract()` method)
|
|
336
|
+
- **FIXED**: SVG extraction failed on `<text>` elements containing `<tspan>`
|
|
337
|
+
- **IMPROVED**: DOCX now includes tables (in document order), headers and footers
|
|
338
|
+
- **IMPROVED**: PPTX now includes grouped shapes, tables and speaker notes
|
|
339
|
+
- **IMPROVED**: ODT now includes headings and text inside spans, links and lists
|
|
340
|
+
- **IMPROVED**: EXIF output formats aperture, shutter speed and focal length as intended and reads PNG/WebP metadata
|
|
341
|
+
through Pillow's public API; the fake `image/*+exif` MIME types are gone from `extxt_available_formats()`
|
|
342
|
+
- **IMPROVED**: Whisper temporary files are always deleted, also when transcription fails
|
|
343
|
+
- **IMPROVED**: Ollama OCR stops trying fallback models when the server is not reachable;
|
|
344
|
+
`xtxt_image_with_confidence()` now uses the same context-aware prompts as `xtxt()`;
|
|
345
|
+
hallucination keywords match whole words (e.g. "roman" no longer fires on "romance")
|
|
346
|
+
|
|
327
347
|
### v0.3.7
|
|
328
348
|
- Python 3.10 or newer is now required; metadata no longer lists the end-of-life versions 3.7–3.9, which were never tested
|
|
329
349
|
- Python 3.14 is tested in CI and declared as supported
|
|
@@ -17,6 +17,15 @@ _TEXT_EXTENSION_HINTS = {
|
|
|
17
17
|
}
|
|
18
18
|
|
|
19
19
|
|
|
20
|
+
def _detect_mime_type(data: bytes) -> str:
|
|
21
|
+
"""Detect the MIME type of in-memory data.
|
|
22
|
+
|
|
23
|
+
The whole buffer is passed to libmagic: the first few KB are not always
|
|
24
|
+
enough to tell apart ZIP-based formats (e.g. XLSX is seen as application/zip).
|
|
25
|
+
"""
|
|
26
|
+
return magic.from_buffer(data, mime=True)
|
|
27
|
+
|
|
28
|
+
|
|
20
29
|
def _resolve_mime_type(mime_type: str, name: Optional[str]) -> str:
|
|
21
30
|
"""Refine the detected MIME type and map unknown text types to text/plain."""
|
|
22
31
|
if mime_type == "text/plain" and name:
|
|
@@ -45,7 +54,7 @@ def _(file_input: str) -> Optional[str]:
|
|
|
45
54
|
data = f.read()
|
|
46
55
|
buffer = io.BytesIO(data)
|
|
47
56
|
buffer.name = file_input
|
|
48
|
-
buffer.mimeType = magic.
|
|
57
|
+
buffer.mimeType = magic.from_file(file_input, mime=True)
|
|
49
58
|
return xtxt(buffer)
|
|
50
59
|
except Exception as e:
|
|
51
60
|
print(f"⚠️ File opening error '{file_input}': {e}")
|
|
@@ -59,8 +68,7 @@ def _(file_input: io.BytesIO) -> Optional[str]:
|
|
|
59
68
|
if hasattr(file_input, "mimeType"):
|
|
60
69
|
mime_type = file_input.mimeType
|
|
61
70
|
else:
|
|
62
|
-
file_input.
|
|
63
|
-
mime_type = magic.Magic(mime=True).from_buffer(file_input.read(2048))
|
|
71
|
+
mime_type = _detect_mime_type(file_input.getvalue())
|
|
64
72
|
file_input.seek(0)
|
|
65
73
|
if not getattr(file_input, "name", None):
|
|
66
74
|
file_input.name = "IO_buffer"
|
|
@@ -83,13 +91,31 @@ def _(file_input: bytes) -> Optional[str]:
|
|
|
83
91
|
try:
|
|
84
92
|
buffer = io.BytesIO(file_input)
|
|
85
93
|
buffer.name = "bytes_input"
|
|
86
|
-
buffer.mimeType =
|
|
94
|
+
buffer.mimeType = _detect_mime_type(file_input)
|
|
87
95
|
return xtxt(buffer)
|
|
88
96
|
except Exception as e:
|
|
89
97
|
print(f"❌ Error processing bytes: {e}")
|
|
90
98
|
return None
|
|
91
99
|
|
|
92
100
|
|
|
101
|
+
@xtxt.register
|
|
102
|
+
def _(file_input: io.IOBase) -> Optional[str]:
|
|
103
|
+
"""Extract text from a binary file object, e.g. one returned by open(path, "rb")."""
|
|
104
|
+
try:
|
|
105
|
+
data = file_input.read()
|
|
106
|
+
if isinstance(data, str):
|
|
107
|
+
print("⚠️ File object opened in text mode: open it in binary mode ('rb')")
|
|
108
|
+
return None
|
|
109
|
+
buffer = io.BytesIO(data)
|
|
110
|
+
name = getattr(file_input, "name", None)
|
|
111
|
+
buffer.name = name if isinstance(name, str) else "file_object"
|
|
112
|
+
buffer.mimeType = _detect_mime_type(data)
|
|
113
|
+
return xtxt(buffer)
|
|
114
|
+
except Exception as e:
|
|
115
|
+
print(f"❌ Error processing file object: {e}")
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
|
|
93
119
|
# Optional support for requests.Response
|
|
94
120
|
try:
|
|
95
121
|
import requests
|
|
@@ -10,69 +10,66 @@ except ImportError:
|
|
|
10
10
|
|
|
11
11
|
if whisper:
|
|
12
12
|
_whisper_model = None
|
|
13
|
-
|
|
13
|
+
|
|
14
14
|
def _get_model():
|
|
15
15
|
global _whisper_model
|
|
16
16
|
if _whisper_model is None:
|
|
17
17
|
_whisper_model = whisper.load_model("base")
|
|
18
18
|
return _whisper_model
|
|
19
|
-
|
|
19
|
+
|
|
20
20
|
def xtxt_audio_whisper(file_buffer):
|
|
21
|
+
temp_path = None
|
|
21
22
|
try:
|
|
22
|
-
# Usa un
|
|
23
|
+
# Usa un suffisso generico - Whisper + FFmpeg gestiscono il formato
|
|
23
24
|
with tempfile.NamedTemporaryFile(delete=False) as temp_file:
|
|
24
25
|
temp_file.write(file_buffer.read())
|
|
25
|
-
temp_file.
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
26
|
+
temp_path = temp_file.name
|
|
27
|
+
|
|
28
|
+
model = _get_model()
|
|
29
|
+
result = model.transcribe(
|
|
30
|
+
temp_path,
|
|
30
31
|
language=None,
|
|
31
32
|
task="transcribe",
|
|
32
33
|
temperature=[0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
os.unlink(temp_file.name)
|
|
37
|
-
text_with_languages = []
|
|
38
|
-
current_lang = None
|
|
39
|
-
|
|
40
|
-
for segment in result['segments']:
|
|
41
|
-
# Rileva cambio di lingua (logica semplificata)
|
|
42
|
-
if 'language' in segment and segment['language'] != current_lang:
|
|
43
|
-
current_lang = segment['language']
|
|
44
|
-
text_with_languages.append(f"\n[{current_lang.upper()}]")
|
|
45
|
-
|
|
46
|
-
text_with_languages.append(segment['text'])
|
|
47
|
-
|
|
48
|
-
return " ".join(text_with_languages).strip()
|
|
49
|
-
|
|
34
|
+
)
|
|
35
|
+
return result["text"].strip()
|
|
36
|
+
|
|
50
37
|
except Exception as e:
|
|
51
38
|
print(f"⚠️ Error while extracting audio with Whisper: {e}")
|
|
52
39
|
return ""
|
|
53
|
-
|
|
54
|
-
|
|
40
|
+
finally:
|
|
41
|
+
if temp_path is not None:
|
|
42
|
+
os.unlink(temp_path)
|
|
43
|
+
|
|
44
|
+
# Registra per tutti i formati audio comuni.
|
|
45
|
+
# Comprende sia i nomi standard sia quelli restituiti da libmagic (es. audio/x-wav).
|
|
55
46
|
audio_formats = [
|
|
56
|
-
"audio/wav", "audio/wave",
|
|
47
|
+
"audio/wav", "audio/wave", "audio/x-wav",
|
|
57
48
|
"audio/mp3", "audio/mpeg",
|
|
58
|
-
"audio/m4a", "audio/mp4",
|
|
59
|
-
"audio/flac",
|
|
49
|
+
"audio/m4a", "audio/x-m4a", "audio/mp4",
|
|
50
|
+
"audio/flac", "audio/x-flac",
|
|
60
51
|
"audio/ogg", "audio/ogg-vorbis",
|
|
61
52
|
"audio/opus",
|
|
62
|
-
"audio/aac",
|
|
63
|
-
"audio/
|
|
64
|
-
"audio/
|
|
53
|
+
"audio/aac", "audio/x-hx-aac-adts",
|
|
54
|
+
"audio/aiff", "audio/x-aiff",
|
|
55
|
+
"audio/wma", "audio/x-ms-wma",
|
|
56
|
+
"audio/webm",
|
|
65
57
|
]
|
|
66
|
-
|
|
58
|
+
|
|
67
59
|
for format_type in audio_formats:
|
|
68
60
|
register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio")
|
|
69
|
-
|
|
61
|
+
|
|
62
|
+
# Formati video: Whisper estrae la traccia audio tramite FFmpeg
|
|
70
63
|
video_audio_formats = [
|
|
71
|
-
"video/mp4",
|
|
64
|
+
"video/mp4", "video/x-m4v",
|
|
72
65
|
"video/quicktime", # .mov
|
|
73
66
|
"video/x-msvideo", # .avi
|
|
74
67
|
"video/webm",
|
|
75
|
-
"video/mkv"
|
|
68
|
+
"video/mkv", "video/x-matroska",
|
|
69
|
+
"video/x-ms-asf", # .wmv / .wma
|
|
70
|
+
"video/mpeg",
|
|
71
|
+
"video/ogg",
|
|
72
|
+
"video/3gpp",
|
|
76
73
|
]
|
|
77
74
|
|
|
78
75
|
for format_type in video_audio_formats:
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
import io
|
|
3
|
+
import zipfile
|
|
4
|
+
try:
|
|
5
|
+
from docx import Document
|
|
6
|
+
from docx.oxml.ns import qn
|
|
7
|
+
from docx.table import Table
|
|
8
|
+
from docx.text.paragraph import Paragraph
|
|
9
|
+
except ImportError:
|
|
10
|
+
Document = None
|
|
11
|
+
|
|
12
|
+
if Document:
|
|
13
|
+
def _block_lines(parent, element):
|
|
14
|
+
"""Yield the text of paragraphs and tables in document order."""
|
|
15
|
+
for child in element.iterchildren():
|
|
16
|
+
if child.tag == qn("w:p"):
|
|
17
|
+
yield Paragraph(child, parent).text
|
|
18
|
+
elif child.tag == qn("w:tbl"):
|
|
19
|
+
yield from _table_lines(Table(child, parent))
|
|
20
|
+
|
|
21
|
+
def _table_lines(table):
|
|
22
|
+
"""Yield one line per table row, cells separated by ' | '."""
|
|
23
|
+
for row in table.rows:
|
|
24
|
+
seen = []
|
|
25
|
+
cells = []
|
|
26
|
+
for cell in row.cells:
|
|
27
|
+
# Merged cells are returned once per grid column: keep the first one
|
|
28
|
+
if any(cell._tc is tc for tc in seen):
|
|
29
|
+
continue
|
|
30
|
+
seen.append(cell._tc)
|
|
31
|
+
cell_text = " ".join(line for line in _block_lines(cell, cell._tc) if line)
|
|
32
|
+
cells.append(cell_text.strip())
|
|
33
|
+
if any(cells):
|
|
34
|
+
yield " | ".join(cells)
|
|
35
|
+
|
|
36
|
+
def _header_footer_lines(doc, attribute_names):
|
|
37
|
+
lines = []
|
|
38
|
+
for section in doc.sections:
|
|
39
|
+
for attribute in attribute_names:
|
|
40
|
+
part = getattr(section, attribute, None)
|
|
41
|
+
if part is None or part.is_linked_to_previous:
|
|
42
|
+
continue
|
|
43
|
+
for line in _block_lines(part, part._element):
|
|
44
|
+
if line and line not in lines:
|
|
45
|
+
lines.append(line)
|
|
46
|
+
return lines
|
|
47
|
+
|
|
48
|
+
def xtxt_docx(file_buffer) -> str:
|
|
49
|
+
try:
|
|
50
|
+
# Copia del buffer per poterlo riutilizzare
|
|
51
|
+
file_buffer.seek(0)
|
|
52
|
+
data = file_buffer.read()
|
|
53
|
+
buffer_copy = io.BytesIO(data)
|
|
54
|
+
|
|
55
|
+
if not zipfile.is_zipfile(buffer_copy):
|
|
56
|
+
print("⚠️ Invalid DOCX (not a ZIP file)")
|
|
57
|
+
return ""
|
|
58
|
+
|
|
59
|
+
buffer_copy.seek(0)
|
|
60
|
+
doc = Document(buffer_copy)
|
|
61
|
+
|
|
62
|
+
headers = _header_footer_lines(doc, ("first_page_header", "header", "even_page_header"))
|
|
63
|
+
body = list(_block_lines(doc, doc.element.body))
|
|
64
|
+
footers = _header_footer_lines(doc, ("first_page_footer", "footer", "even_page_footer"))
|
|
65
|
+
return "\n".join(headers + body + footers)
|
|
66
|
+
|
|
67
|
+
except Exception as e:
|
|
68
|
+
print(f"⚠️ Error during extraction DOCX: {e}")
|
|
69
|
+
return ""
|
|
70
|
+
|
|
71
|
+
register_extractor(
|
|
72
|
+
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
73
|
+
xtxt_docx,
|
|
74
|
+
name="DOCX"
|
|
75
|
+
)
|
|
@@ -1,7 +1,6 @@
|
|
|
1
1
|
# pyxtxt/extractors/image_exif.py
|
|
2
|
-
from . import register_extractor
|
|
3
2
|
from io import BytesIO
|
|
4
|
-
import
|
|
3
|
+
import numbers
|
|
5
4
|
|
|
6
5
|
try:
|
|
7
6
|
from PIL import Image, ExifTags
|
|
@@ -13,6 +12,27 @@ except ImportError:
|
|
|
13
12
|
GPSTAGS = None
|
|
14
13
|
|
|
15
14
|
if Image and ExifTags and TAGS:
|
|
15
|
+
_EXIF_IFD = 0x8769 # ExifOffset: pointer to camera settings (FNumber, ExposureTime, ...)
|
|
16
|
+
_GPS_IFD = 0x8825 # GPSInfo: pointer to GPS data
|
|
17
|
+
|
|
18
|
+
def _as_ratio(value):
|
|
19
|
+
"""Return (numerator, denominator) for EXIF rationals, or None.
|
|
20
|
+
|
|
21
|
+
Older Pillow versions return tuples, newer ones IFDRational objects.
|
|
22
|
+
"""
|
|
23
|
+
if isinstance(value, tuple) and len(value) == 2:
|
|
24
|
+
return value
|
|
25
|
+
if isinstance(value, numbers.Rational) or hasattr(value, "denominator"):
|
|
26
|
+
return value.numerator, value.denominator
|
|
27
|
+
return None
|
|
28
|
+
|
|
29
|
+
def _collect_exif(image):
|
|
30
|
+
"""Return (tags, gps) dictionaries keyed by numeric tag id."""
|
|
31
|
+
exif = image.getexif()
|
|
32
|
+
tags = {tag_id: value for tag_id, value in exif.items() if tag_id not in (_EXIF_IFD, _GPS_IFD)}
|
|
33
|
+
tags.update(exif.get_ifd(_EXIF_IFD))
|
|
34
|
+
return tags, dict(exif.get_ifd(_GPS_IFD))
|
|
35
|
+
|
|
16
36
|
def xtxt_image_exif(file_buffer):
|
|
17
37
|
"""
|
|
18
38
|
Extract EXIF metadata from images as human-readable text.
|
|
@@ -32,23 +52,19 @@ if Image and ExifTags and TAGS:
|
|
|
32
52
|
image = Image.open(BytesIO(image_data))
|
|
33
53
|
|
|
34
54
|
# Get EXIF data
|
|
35
|
-
exif_data = image
|
|
55
|
+
exif_data, gps_data = _collect_exif(image)
|
|
36
56
|
|
|
37
|
-
if not exif_data:
|
|
57
|
+
if not exif_data and not gps_data:
|
|
38
58
|
return "NO_EXIF_DATA_FOUND"
|
|
39
59
|
|
|
40
60
|
# Extract readable EXIF information
|
|
41
61
|
exif_text_lines = []
|
|
42
|
-
gps_data = {}
|
|
43
62
|
|
|
44
63
|
# Process main EXIF tags
|
|
45
64
|
for tag_id, value in exif_data.items():
|
|
46
65
|
tag_name = TAGS.get(tag_id, f"UnknownTag_{tag_id}")
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
if tag_name == "GPSInfo" and isinstance(value, dict):
|
|
50
|
-
gps_data = value
|
|
51
|
-
continue
|
|
66
|
+
if isinstance(value, str):
|
|
67
|
+
value = value.strip("\x00 ")
|
|
52
68
|
|
|
53
69
|
# Format common values
|
|
54
70
|
if tag_name in ["DateTime", "DateTimeOriginal", "DateTimeDigitized"]:
|
|
@@ -56,28 +72,27 @@ if Image and ExifTags and TAGS:
|
|
|
56
72
|
elif tag_name in ["Make", "Model", "Software", "Artist", "Copyright"]:
|
|
57
73
|
exif_text_lines.append(f"{tag_name}: {value}")
|
|
58
74
|
elif tag_name in ["XResolution", "YResolution"]:
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
exif_text_lines.append(f"{tag_name}: {
|
|
75
|
+
ratio = _as_ratio(value)
|
|
76
|
+
if ratio and ratio[1] != 0:
|
|
77
|
+
exif_text_lines.append(f"{tag_name}: {ratio[0] / ratio[1]:.1f} dpi")
|
|
62
78
|
else:
|
|
63
79
|
exif_text_lines.append(f"{tag_name}: {value}")
|
|
64
80
|
elif tag_name in ["FNumber", "FocalLength", "ExposureTime"]:
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
if value[0] == 1:
|
|
75
|
-
exif_text_lines.append(f"Shutter Speed: 1/{value[1]}s")
|
|
76
|
-
else:
|
|
77
|
-
exp_time = value[0] / value[1]
|
|
78
|
-
exif_text_lines.append(f"Shutter Speed: {exp_time:.3f}s")
|
|
81
|
+
ratio = _as_ratio(value)
|
|
82
|
+
if ratio and ratio[1] != 0:
|
|
83
|
+
numerator, denominator = ratio
|
|
84
|
+
if tag_name == "FNumber":
|
|
85
|
+
exif_text_lines.append(f"Aperture: f/{numerator / denominator:.1f}")
|
|
86
|
+
elif tag_name == "FocalLength":
|
|
87
|
+
exif_text_lines.append(f"Focal Length: {numerator / denominator:.0f}mm")
|
|
88
|
+
elif numerator == 1:
|
|
89
|
+
exif_text_lines.append(f"Shutter Speed: 1/{denominator}s")
|
|
79
90
|
else:
|
|
80
|
-
|
|
91
|
+
exp_time = numerator / denominator
|
|
92
|
+
if exp_time < 1:
|
|
93
|
+
exif_text_lines.append(f"Shutter Speed: 1/{round(1 / exp_time)}s")
|
|
94
|
+
else:
|
|
95
|
+
exif_text_lines.append(f"Shutter Speed: {exp_time:g}s")
|
|
81
96
|
else:
|
|
82
97
|
exif_text_lines.append(f"{tag_name}: {value}")
|
|
83
98
|
elif tag_name == "ISOSpeedRatings":
|
|
@@ -111,8 +126,10 @@ if Image and ExifTags and TAGS:
|
|
|
111
126
|
}
|
|
112
127
|
orient_desc = orientations.get(value, f"Orientation {value}")
|
|
113
128
|
exif_text_lines.append(f"Orientation: {orient_desc}")
|
|
114
|
-
elif isinstance(value, (str,
|
|
115
|
-
# Include other simple values
|
|
129
|
+
elif isinstance(value, (str, numbers.Number)):
|
|
130
|
+
# Include other simple values (rationals as decimals)
|
|
131
|
+
if not isinstance(value, (str, int)):
|
|
132
|
+
value = round(float(value), 4)
|
|
116
133
|
exif_text_lines.append(f"{tag_name}: {value}")
|
|
117
134
|
|
|
118
135
|
# Process GPS data if available
|
|
@@ -130,7 +147,7 @@ if Image and ExifTags and TAGS:
|
|
|
130
147
|
lat_dms = gps_info['GPSLatitude']
|
|
131
148
|
lat_ref = gps_info['GPSLatitudeRef']
|
|
132
149
|
if len(lat_dms) == 3:
|
|
133
|
-
lat_deg = lat_dms[0] + lat_dms[1]/60 + lat_dms[2]/3600
|
|
150
|
+
lat_deg = float(lat_dms[0]) + float(lat_dms[1])/60 + float(lat_dms[2])/3600
|
|
134
151
|
if lat_ref == 'S':
|
|
135
152
|
lat_deg = -lat_deg
|
|
136
153
|
gps_text_lines.append(f"GPS Latitude: {lat_deg:.6f}° {lat_ref}")
|
|
@@ -139,7 +156,7 @@ if Image and ExifTags and TAGS:
|
|
|
139
156
|
lon_dms = gps_info['GPSLongitude']
|
|
140
157
|
lon_ref = gps_info['GPSLongitudeRef']
|
|
141
158
|
if len(lon_dms) == 3:
|
|
142
|
-
lon_deg = lon_dms[0] + lon_dms[1]/60 + lon_dms[2]/3600
|
|
159
|
+
lon_deg = float(lon_dms[0]) + float(lon_dms[1])/60 + float(lon_dms[2])/3600
|
|
143
160
|
if lon_ref == 'W':
|
|
144
161
|
lon_deg = -lon_deg
|
|
145
162
|
gps_text_lines.append(f"GPS Longitude: {lon_deg:.6f}° {lon_ref}")
|
|
@@ -172,12 +189,5 @@ if Image and ExifTags and TAGS:
|
|
|
172
189
|
except Exception as e:
|
|
173
190
|
print(f"⚠️ Error extracting EXIF from image: {e}")
|
|
174
191
|
return ""
|
|
175
|
-
|
|
176
|
-
#
|
|
177
|
-
# Using EXIF-specific MIME types to avoid conflicts with OCR extractors
|
|
178
|
-
register_extractor("image/jpeg+exif", xtxt_image_exif, name="EXIF")
|
|
179
|
-
register_extractor("image/jpg+exif", xtxt_image_exif, name="EXIF")
|
|
180
|
-
register_extractor("image/png+exif", xtxt_image_exif, name="EXIF")
|
|
181
|
-
register_extractor("image/tiff+exif", xtxt_image_exif, name="EXIF")
|
|
182
|
-
register_extractor("image/bmp+exif", xtxt_image_exif, name="EXIF")
|
|
183
|
-
register_extractor("image/webp+exif", xtxt_image_exif, name="EXIF")
|
|
192
|
+
|
|
193
|
+
# EXIF metadata is exposed through pyxtxt.xtxt_exif(): xtxt() on an image runs OCR.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
|
|
3
|
+
try:
|
|
4
|
+
import extract_msg
|
|
5
|
+
from bs4 import BeautifulSoup
|
|
6
|
+
except ImportError:
|
|
7
|
+
extract_msg = None
|
|
8
|
+
|
|
9
|
+
if extract_msg:
|
|
10
|
+
def xtxt_msg(file_buffer):
|
|
11
|
+
# extract_msg accetta direttamente i bytes del file .msg
|
|
12
|
+
content = file_buffer.read()
|
|
13
|
+
|
|
14
|
+
with extract_msg.openMsg(content) as msg:
|
|
15
|
+
body = getattr(msg, "body", None)
|
|
16
|
+
if body and body.strip():
|
|
17
|
+
return body.strip()
|
|
18
|
+
|
|
19
|
+
# Nessun corpo in testo semplice: usa la versione HTML
|
|
20
|
+
html_body = getattr(msg, "htmlBody", None)
|
|
21
|
+
if html_body:
|
|
22
|
+
soup = BeautifulSoup(html_body, "html.parser")
|
|
23
|
+
return soup.get_text(separator="\n").strip()
|
|
24
|
+
|
|
25
|
+
return ""
|
|
26
|
+
|
|
27
|
+
register_extractor(
|
|
28
|
+
"application/vnd.ms-outlook",
|
|
29
|
+
xtxt_msg,
|
|
30
|
+
name="MSG"
|
|
31
|
+
)
|