pyxtxt 0.3.7__tar.gz → 0.3.8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {pyxtxt-0.3.7/src/pyxtxt.egg-info → pyxtxt-0.3.8}/PKG-INFO +31 -11
  2. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/README.md +30 -10
  3. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/pyproject.toml +1 -1
  4. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/core.py +30 -4
  5. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/audio.py +34 -37
  6. pyxtxt-0.3.8/src/pyxtxt/estrattori/docx.py +75 -0
  7. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/exif.py +51 -41
  8. pyxtxt-0.3.8/src/pyxtxt/estrattori/msg.py +31 -0
  9. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/ocr_ollama.py +143 -211
  10. pyxtxt-0.3.8/src/pyxtxt/estrattori/odt.py +30 -0
  11. pyxtxt-0.3.8/src/pyxtxt/estrattori/pptx.py +58 -0
  12. pyxtxt-0.3.8/src/pyxtxt/estrattori/svg.py +30 -0
  13. {pyxtxt-0.3.7 → pyxtxt-0.3.8/src/pyxtxt.egg-info}/PKG-INFO +31 -11
  14. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/SOURCES.txt +3 -1
  15. pyxtxt-0.3.8/tests/test_audio.py +66 -0
  16. pyxtxt-0.3.8/tests/test_extraction.py +289 -0
  17. pyxtxt-0.3.8/tests/test_ocr_ollama.py +69 -0
  18. pyxtxt-0.3.7/src/pyxtxt/estrattori/docx.py +0 -35
  19. pyxtxt-0.3.7/src/pyxtxt/estrattori/msg.py +0 -41
  20. pyxtxt-0.3.7/src/pyxtxt/estrattori/odt.py +0 -27
  21. pyxtxt-0.3.7/src/pyxtxt/estrattori/pptx.py +0 -40
  22. pyxtxt-0.3.7/src/pyxtxt/estrattori/svg.py +0 -23
  23. pyxtxt-0.3.7/tests/test_extraction.py +0 -122
  24. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/LICENSE +0 -0
  25. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/setup.cfg +0 -0
  26. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/__init__.py +0 -0
  27. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/__init__.py +0 -0
  28. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/doc.py +0 -0
  29. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/eml.py +0 -0
  30. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/epub.py +0 -0
  31. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/html.py +0 -0
  32. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/md.py +0 -0
  33. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/ocr.py +0 -0
  34. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/pdf.py +0 -0
  35. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/rtf.py +0 -0
  36. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/tex.py +0 -0
  37. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/txt.py +0 -0
  38. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/xls.py +0 -0
  39. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/xlsx.py +0 -0
  40. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/xml.py +0 -0
  41. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt/examples.py +0 -0
  42. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  43. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/requires.txt +0 -0
  44. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/top_level.txt +0 -0
  45. {pyxtxt-0.3.7 → pyxtxt-0.3.8}/tests/test_import.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.7
3
+ Version: 0.3.8
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, etc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License-Expression: MIT
@@ -102,7 +102,7 @@ text = xtxt("report.pdf")
102
102
 
103
103
  ## ✨ Features
104
104
 
105
- - **One function for everything**: `xtxt()` accepts a file path, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
105
+ - **One function for everything**: `xtxt()` accepts a file path, a binary file object, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
106
106
  - **Automatic type detection** with `python-magic`, refined by the file extension when libmagic is not specific enough (e.g. Markdown)
107
107
  - **Modular dependencies**: each format is an optional extra, the core only needs `python-magic`
108
108
  - **Office, web and document formats**: PDF, DOCX, PPTX, XLSX, XLS, ODT, HTML, XML, SVG, Markdown, EPUB, RTF, EML, MSG, LaTeX, DOC, TXT
@@ -117,8 +117,8 @@ text = xtxt("report.pdf")
117
117
  | Format | Install extra | Notes |
118
118
  |---|---|---|
119
119
  | PDF | `pdf` | PyMuPDF |
120
- | DOCX | `docx` | Paragraph text (tables are not extracted yet) |
121
- | PPTX | `presentation` | Text of all slide shapes |
120
+ | DOCX | `docx` | Paragraphs and tables in document order, headers and footers |
121
+ | PPTX | `presentation` | Text boxes, grouped shapes, tables and speaker notes |
122
122
  | XLSX, XLS | `spreadsheet` | Every row of every visible sheet, cells joined with ` \| ` |
123
123
  | ODT | `odf` | |
124
124
  | HTML | `html` | |
@@ -127,11 +127,11 @@ text = xtxt("report.pdf")
127
127
  | EPUB | `epub` | |
128
128
  | RTF | `rtf` | |
129
129
  | EML | `email` | Plain-text and HTML parts |
130
- | MSG (Outlook) | `outlook` | |
130
+ | MSG (Outlook) | `outlook` | Plain-text body, or the HTML body when there is no plain text |
131
131
  | LaTeX | `latex` | |
132
132
  | DOC (legacy Word) | — | Needs the `antiword` system tool |
133
133
  | TXT and other `text/*` | — | Always available, decoded as UTF-8 |
134
- | Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download |
134
+ | Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download. MP3, WAV, M4A, AAC, FLAC, OGG/Opus, AIFF, WMA, MP4, MOV, AVI, MKV, WebM |
135
135
  | Images (OCR) | `ocr` or `ocr-ollama` | See [OCR from images](#-ocr-from-images) |
136
136
 
137
137
  To list what is available in your installation:
@@ -208,9 +208,12 @@ from pyxtxt import xtxt
208
208
  # From a file path
209
209
  text = xtxt("document.pdf")
210
210
 
211
- # From an in-memory buffer
211
+ # From a file object opened in binary mode
212
212
  with open("document.docx", "rb") as f:
213
- buffer = io.BytesIO(f.read())
213
+ text = xtxt(f)
214
+
215
+ # From an in-memory buffer
216
+ buffer = io.BytesIO(docx_bytes)
214
217
  text = xtxt(buffer)
215
218
 
216
219
  # Give the buffer a name to help type detection (useful for Markdown, LaTeX, RTF)
@@ -344,12 +347,12 @@ print((files("pyxtxt") / "examples.py").read_text())
344
347
 
345
348
  ## ⚠️ Known limitations
346
349
 
347
- - **Supported inputs**: file paths, `io.BytesIO`, `bytes` and `requests.Response`. A file object returned by
348
- `open()` must be read first (`xtxt(f.read())`).
350
+ - **Supported inputs**: file paths, binary file objects, `io.BytesIO`, `bytes` and `requests.Response`.
351
+ File objects opened in text mode are rejected: open them with `"rb"`.
349
352
  - **Type detection without a file name**: libmagic cannot tell apart some formats from raw bytes (legacy Office files
350
353
  share the same signature; Markdown looks like plain text). Pass a file path, or set `buffer.name`, when possible.
351
354
  - **Legacy PowerPoint (`.ppt`)** is not supported.
352
- - **DOCX**: text inside tables, headers and footers is not extracted yet. **SVG**: text inside `<tspan>` elements is not extracted yet.
355
+ - **DOCX**: text boxes, footnotes and comments are not extracted. **PPTX**: text inside charts is not extracted.
353
356
  - Errors are reported with messages printed to standard output and the functions return `None` or an empty string;
354
357
  they do not raise exceptions.
355
358
 
@@ -407,6 +410,23 @@ Pull requests, issues and feedback are welcome.
407
410
 
408
411
  ## 📊 Changelog
409
412
 
413
+ ### v0.3.8
414
+ - **NEW**: `xtxt()` accepts binary file objects, e.g. `xtxt(open("file.pdf", "rb"))`
415
+ - **FIXED**: XLSX passed as `bytes` or `BytesIO` was detected as a ZIP archive and rejected
416
+ - **FIXED**: WAV, M4A, AAC, AIFF, WMA and MKV files were never transcribed (libmagic reports them as `audio/x-wav`,
417
+ `audio/x-m4a`, `video/x-matroska`, ... which were not registered)
418
+ - **FIXED**: MSG extraction always failed (`extract_msg.Message` has no `extract()` method)
419
+ - **FIXED**: SVG extraction failed on `<text>` elements containing `<tspan>`
420
+ - **IMPROVED**: DOCX now includes tables (in document order), headers and footers
421
+ - **IMPROVED**: PPTX now includes grouped shapes, tables and speaker notes
422
+ - **IMPROVED**: ODT now includes headings and text inside spans, links and lists
423
+ - **IMPROVED**: EXIF output formats aperture, shutter speed and focal length as intended and reads PNG/WebP metadata
424
+ through Pillow's public API; the fake `image/*+exif` MIME types are gone from `extxt_available_formats()`
425
+ - **IMPROVED**: Whisper temporary files are always deleted, also when transcription fails
426
+ - **IMPROVED**: Ollama OCR stops trying fallback models when the server is not reachable;
427
+ `xtxt_image_with_confidence()` now uses the same context-aware prompts as `xtxt()`;
428
+ hallucination keywords match whole words (e.g. "roman" no longer fires on "romance")
429
+
410
430
  ### v0.3.7
411
431
  - Python 3.10 or newer is now required; metadata no longer lists the end-of-life versions 3.7–3.9, which were never tested
412
432
  - Python 3.14 is tested in CI and declared as supported
@@ -19,7 +19,7 @@ text = xtxt("report.pdf")
19
19
 
20
20
  ## ✨ Features
21
21
 
22
- - **One function for everything**: `xtxt()` accepts a file path, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
22
+ - **One function for everything**: `xtxt()` accepts a file path, a binary file object, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
23
23
  - **Automatic type detection** with `python-magic`, refined by the file extension when libmagic is not specific enough (e.g. Markdown)
24
24
  - **Modular dependencies**: each format is an optional extra, the core only needs `python-magic`
25
25
  - **Office, web and document formats**: PDF, DOCX, PPTX, XLSX, XLS, ODT, HTML, XML, SVG, Markdown, EPUB, RTF, EML, MSG, LaTeX, DOC, TXT
@@ -34,8 +34,8 @@ text = xtxt("report.pdf")
34
34
  | Format | Install extra | Notes |
35
35
  |---|---|---|
36
36
  | PDF | `pdf` | PyMuPDF |
37
- | DOCX | `docx` | Paragraph text (tables are not extracted yet) |
38
- | PPTX | `presentation` | Text of all slide shapes |
37
+ | DOCX | `docx` | Paragraphs and tables in document order, headers and footers |
38
+ | PPTX | `presentation` | Text boxes, grouped shapes, tables and speaker notes |
39
39
  | XLSX, XLS | `spreadsheet` | Every row of every visible sheet, cells joined with ` \| ` |
40
40
  | ODT | `odf` | |
41
41
  | HTML | `html` | |
@@ -44,11 +44,11 @@ text = xtxt("report.pdf")
44
44
  | EPUB | `epub` | |
45
45
  | RTF | `rtf` | |
46
46
  | EML | `email` | Plain-text and HTML parts |
47
- | MSG (Outlook) | `outlook` | |
47
+ | MSG (Outlook) | `outlook` | Plain-text body, or the HTML body when there is no plain text |
48
48
  | LaTeX | `latex` | |
49
49
  | DOC (legacy Word) | — | Needs the `antiword` system tool |
50
50
  | TXT and other `text/*` | — | Always available, decoded as UTF-8 |
51
- | Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download |
51
+ | Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download. MP3, WAV, M4A, AAC, FLAC, OGG/Opus, AIFF, WMA, MP4, MOV, AVI, MKV, WebM |
52
52
  | Images (OCR) | `ocr` or `ocr-ollama` | See [OCR from images](#-ocr-from-images) |
53
53
 
54
54
  To list what is available in your installation:
@@ -125,9 +125,12 @@ from pyxtxt import xtxt
125
125
  # From a file path
126
126
  text = xtxt("document.pdf")
127
127
 
128
- # From an in-memory buffer
128
+ # From a file object opened in binary mode
129
129
  with open("document.docx", "rb") as f:
130
- buffer = io.BytesIO(f.read())
130
+ text = xtxt(f)
131
+
132
+ # From an in-memory buffer
133
+ buffer = io.BytesIO(docx_bytes)
131
134
  text = xtxt(buffer)
132
135
 
133
136
  # Give the buffer a name to help type detection (useful for Markdown, LaTeX, RTF)
@@ -261,12 +264,12 @@ print((files("pyxtxt") / "examples.py").read_text())
261
264
 
262
265
  ## ⚠️ Known limitations
263
266
 
264
- - **Supported inputs**: file paths, `io.BytesIO`, `bytes` and `requests.Response`. A file object returned by
265
- `open()` must be read first (`xtxt(f.read())`).
267
+ - **Supported inputs**: file paths, binary file objects, `io.BytesIO`, `bytes` and `requests.Response`.
268
+ File objects opened in text mode are rejected: open them with `"rb"`.
266
269
  - **Type detection without a file name**: libmagic cannot tell apart some formats from raw bytes (legacy Office files
267
270
  share the same signature; Markdown looks like plain text). Pass a file path, or set `buffer.name`, when possible.
268
271
  - **Legacy PowerPoint (`.ppt`)** is not supported.
269
- - **DOCX**: text inside tables, headers and footers is not extracted yet. **SVG**: text inside `<tspan>` elements is not extracted yet.
272
+ - **DOCX**: text boxes, footnotes and comments are not extracted. **PPTX**: text inside charts is not extracted.
270
273
  - Errors are reported with messages printed to standard output and the functions return `None` or an empty string;
271
274
  they do not raise exceptions.
272
275
 
@@ -324,6 +327,23 @@ Pull requests, issues and feedback are welcome.
324
327
 
325
328
  ## 📊 Changelog
326
329
 
330
+ ### v0.3.8
331
+ - **NEW**: `xtxt()` accepts binary file objects, e.g. `xtxt(open("file.pdf", "rb"))`
332
+ - **FIXED**: XLSX passed as `bytes` or `BytesIO` was detected as a ZIP archive and rejected
333
+ - **FIXED**: WAV, M4A, AAC, AIFF, WMA and MKV files were never transcribed (libmagic reports them as `audio/x-wav`,
334
+ `audio/x-m4a`, `video/x-matroska`, ... which were not registered)
335
+ - **FIXED**: MSG extraction always failed (`extract_msg.Message` has no `extract()` method)
336
+ - **FIXED**: SVG extraction failed on `<text>` elements containing `<tspan>`
337
+ - **IMPROVED**: DOCX now includes tables (in document order), headers and footers
338
+ - **IMPROVED**: PPTX now includes grouped shapes, tables and speaker notes
339
+ - **IMPROVED**: ODT now includes headings and text inside spans, links and lists
340
+ - **IMPROVED**: EXIF output formats aperture, shutter speed and focal length as intended and reads PNG/WebP metadata
341
+ through Pillow's public API; the fake `image/*+exif` MIME types are gone from `extxt_available_formats()`
342
+ - **IMPROVED**: Whisper temporary files are always deleted, also when transcription fails
343
+ - **IMPROVED**: Ollama OCR stops trying fallback models when the server is not reachable;
344
+ `xtxt_image_with_confidence()` now uses the same context-aware prompts as `xtxt()`;
345
+ hallucination keywords match whole words (e.g. "roman" no longer fires on "romance")
346
+
327
347
  ### v0.3.7
328
348
  - Python 3.10 or newer is now required; metadata no longer lists the end-of-life versions 3.7–3.9, which were never tested
329
349
  - Python 3.14 is tested in CI and declared as supported
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.3.7"
3
+ version = "0.3.8"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, etc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -17,6 +17,15 @@ _TEXT_EXTENSION_HINTS = {
17
17
  }
18
18
 
19
19
 
20
+ def _detect_mime_type(data: bytes) -> str:
21
+ """Detect the MIME type of in-memory data.
22
+
23
+ The whole buffer is passed to libmagic: the first few KB are not always
24
+ enough to tell apart ZIP-based formats (e.g. XLSX is seen as application/zip).
25
+ """
26
+ return magic.from_buffer(data, mime=True)
27
+
28
+
20
29
  def _resolve_mime_type(mime_type: str, name: Optional[str]) -> str:
21
30
  """Refine the detected MIME type and map unknown text types to text/plain."""
22
31
  if mime_type == "text/plain" and name:
@@ -45,7 +54,7 @@ def _(file_input: str) -> Optional[str]:
45
54
  data = f.read()
46
55
  buffer = io.BytesIO(data)
47
56
  buffer.name = file_input
48
- buffer.mimeType = magic.Magic(mime=True).from_file(file_input)
57
+ buffer.mimeType = magic.from_file(file_input, mime=True)
49
58
  return xtxt(buffer)
50
59
  except Exception as e:
51
60
  print(f"⚠️ File opening error '{file_input}': {e}")
@@ -59,8 +68,7 @@ def _(file_input: io.BytesIO) -> Optional[str]:
59
68
  if hasattr(file_input, "mimeType"):
60
69
  mime_type = file_input.mimeType
61
70
  else:
62
- file_input.seek(0)
63
- mime_type = magic.Magic(mime=True).from_buffer(file_input.read(2048))
71
+ mime_type = _detect_mime_type(file_input.getvalue())
64
72
  file_input.seek(0)
65
73
  if not getattr(file_input, "name", None):
66
74
  file_input.name = "IO_buffer"
@@ -83,13 +91,31 @@ def _(file_input: bytes) -> Optional[str]:
83
91
  try:
84
92
  buffer = io.BytesIO(file_input)
85
93
  buffer.name = "bytes_input"
86
- buffer.mimeType = magic.Magic(mime=True).from_buffer(file_input[:2048])
94
+ buffer.mimeType = _detect_mime_type(file_input)
87
95
  return xtxt(buffer)
88
96
  except Exception as e:
89
97
  print(f"❌ Error processing bytes: {e}")
90
98
  return None
91
99
 
92
100
 
101
+ @xtxt.register
102
+ def _(file_input: io.IOBase) -> Optional[str]:
103
+ """Extract text from a binary file object, e.g. one returned by open(path, "rb")."""
104
+ try:
105
+ data = file_input.read()
106
+ if isinstance(data, str):
107
+ print("⚠️ File object opened in text mode: open it in binary mode ('rb')")
108
+ return None
109
+ buffer = io.BytesIO(data)
110
+ name = getattr(file_input, "name", None)
111
+ buffer.name = name if isinstance(name, str) else "file_object"
112
+ buffer.mimeType = _detect_mime_type(data)
113
+ return xtxt(buffer)
114
+ except Exception as e:
115
+ print(f"❌ Error processing file object: {e}")
116
+ return None
117
+
118
+
93
119
  # Optional support for requests.Response
94
120
  try:
95
121
  import requests
@@ -10,69 +10,66 @@ except ImportError:
10
10
 
11
11
  if whisper:
12
12
  _whisper_model = None
13
-
13
+
14
14
  def _get_model():
15
15
  global _whisper_model
16
16
  if _whisper_model is None:
17
17
  _whisper_model = whisper.load_model("base")
18
18
  return _whisper_model
19
-
19
+
20
20
  def xtxt_audio_whisper(file_buffer):
21
+ temp_path = None
21
22
  try:
22
- # Usa un suffixe generico - Whisper + FFmpeg gestiscono il formato
23
+ # Usa un suffisso generico - Whisper + FFmpeg gestiscono il formato
23
24
  with tempfile.NamedTemporaryFile(delete=False) as temp_file:
24
25
  temp_file.write(file_buffer.read())
25
- temp_file.flush()
26
-
27
- model = _get_model()
28
- result = model.transcribe(
29
- temp_file.name,
26
+ temp_path = temp_file.name
27
+
28
+ model = _get_model()
29
+ result = model.transcribe(
30
+ temp_path,
30
31
  language=None,
31
32
  task="transcribe",
32
33
  temperature=[0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
33
- word_timestamps=True
34
- )
35
-
36
- os.unlink(temp_file.name)
37
- text_with_languages = []
38
- current_lang = None
39
-
40
- for segment in result['segments']:
41
- # Rileva cambio di lingua (logica semplificata)
42
- if 'language' in segment and segment['language'] != current_lang:
43
- current_lang = segment['language']
44
- text_with_languages.append(f"\n[{current_lang.upper()}]")
45
-
46
- text_with_languages.append(segment['text'])
47
-
48
- return " ".join(text_with_languages).strip()
49
-
34
+ )
35
+ return result["text"].strip()
36
+
50
37
  except Exception as e:
51
38
  print(f"⚠️ Error while extracting audio with Whisper: {e}")
52
39
  return ""
53
-
54
- # Registra per tutti i formati audio comuni
40
+ finally:
41
+ if temp_path is not None:
42
+ os.unlink(temp_path)
43
+
44
+ # Registra per tutti i formati audio comuni.
45
+ # Comprende sia i nomi standard sia quelli restituiti da libmagic (es. audio/x-wav).
55
46
  audio_formats = [
56
- "audio/wav", "audio/wave",
47
+ "audio/wav", "audio/wave", "audio/x-wav",
57
48
  "audio/mp3", "audio/mpeg",
58
- "audio/m4a", "audio/mp4",
59
- "audio/flac",
49
+ "audio/m4a", "audio/x-m4a", "audio/mp4",
50
+ "audio/flac", "audio/x-flac",
60
51
  "audio/ogg", "audio/ogg-vorbis",
61
52
  "audio/opus",
62
- "audio/aac",
63
- "audio/wma",
64
- "audio/webm"
53
+ "audio/aac", "audio/x-hx-aac-adts",
54
+ "audio/aiff", "audio/x-aiff",
55
+ "audio/wma", "audio/x-ms-wma",
56
+ "audio/webm",
65
57
  ]
66
-
58
+
67
59
  for format_type in audio_formats:
68
60
  register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio")
69
- # Aggiungi questi formati video al tuo registro
61
+
62
+ # Formati video: Whisper estrae la traccia audio tramite FFmpeg
70
63
  video_audio_formats = [
71
- "video/mp4",
64
+ "video/mp4", "video/x-m4v",
72
65
  "video/quicktime", # .mov
73
66
  "video/x-msvideo", # .avi
74
67
  "video/webm",
75
- "video/mkv"
68
+ "video/mkv", "video/x-matroska",
69
+ "video/x-ms-asf", # .wmv / .wma
70
+ "video/mpeg",
71
+ "video/ogg",
72
+ "video/3gpp",
76
73
  ]
77
74
 
78
75
  for format_type in video_audio_formats:
@@ -0,0 +1,75 @@
1
+ from . import register_extractor
2
+ import io
3
+ import zipfile
4
+ try:
5
+ from docx import Document
6
+ from docx.oxml.ns import qn
7
+ from docx.table import Table
8
+ from docx.text.paragraph import Paragraph
9
+ except ImportError:
10
+ Document = None
11
+
12
+ if Document:
13
+ def _block_lines(parent, element):
14
+ """Yield the text of paragraphs and tables in document order."""
15
+ for child in element.iterchildren():
16
+ if child.tag == qn("w:p"):
17
+ yield Paragraph(child, parent).text
18
+ elif child.tag == qn("w:tbl"):
19
+ yield from _table_lines(Table(child, parent))
20
+
21
+ def _table_lines(table):
22
+ """Yield one line per table row, cells separated by ' | '."""
23
+ for row in table.rows:
24
+ seen = []
25
+ cells = []
26
+ for cell in row.cells:
27
+ # Merged cells are returned once per grid column: keep the first one
28
+ if any(cell._tc is tc for tc in seen):
29
+ continue
30
+ seen.append(cell._tc)
31
+ cell_text = " ".join(line for line in _block_lines(cell, cell._tc) if line)
32
+ cells.append(cell_text.strip())
33
+ if any(cells):
34
+ yield " | ".join(cells)
35
+
36
+ def _header_footer_lines(doc, attribute_names):
37
+ lines = []
38
+ for section in doc.sections:
39
+ for attribute in attribute_names:
40
+ part = getattr(section, attribute, None)
41
+ if part is None or part.is_linked_to_previous:
42
+ continue
43
+ for line in _block_lines(part, part._element):
44
+ if line and line not in lines:
45
+ lines.append(line)
46
+ return lines
47
+
48
+ def xtxt_docx(file_buffer) -> str:
49
+ try:
50
+ # Copia del buffer per poterlo riutilizzare
51
+ file_buffer.seek(0)
52
+ data = file_buffer.read()
53
+ buffer_copy = io.BytesIO(data)
54
+
55
+ if not zipfile.is_zipfile(buffer_copy):
56
+ print("⚠️ Invalid DOCX (not a ZIP file)")
57
+ return ""
58
+
59
+ buffer_copy.seek(0)
60
+ doc = Document(buffer_copy)
61
+
62
+ headers = _header_footer_lines(doc, ("first_page_header", "header", "even_page_header"))
63
+ body = list(_block_lines(doc, doc.element.body))
64
+ footers = _header_footer_lines(doc, ("first_page_footer", "footer", "even_page_footer"))
65
+ return "\n".join(headers + body + footers)
66
+
67
+ except Exception as e:
68
+ print(f"⚠️ Error during extraction DOCX: {e}")
69
+ return ""
70
+
71
+ register_extractor(
72
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
73
+ xtxt_docx,
74
+ name="DOCX"
75
+ )
@@ -1,7 +1,6 @@
1
1
  # pyxtxt/extractors/image_exif.py
2
- from . import register_extractor
3
2
  from io import BytesIO
4
- import json
3
+ import numbers
5
4
 
6
5
  try:
7
6
  from PIL import Image, ExifTags
@@ -13,6 +12,27 @@ except ImportError:
13
12
  GPSTAGS = None
14
13
 
15
14
  if Image and ExifTags and TAGS:
15
+ _EXIF_IFD = 0x8769 # ExifOffset: pointer to camera settings (FNumber, ExposureTime, ...)
16
+ _GPS_IFD = 0x8825 # GPSInfo: pointer to GPS data
17
+
18
+ def _as_ratio(value):
19
+ """Return (numerator, denominator) for EXIF rationals, or None.
20
+
21
+ Older Pillow versions return tuples, newer ones IFDRational objects.
22
+ """
23
+ if isinstance(value, tuple) and len(value) == 2:
24
+ return value
25
+ if isinstance(value, numbers.Rational) or hasattr(value, "denominator"):
26
+ return value.numerator, value.denominator
27
+ return None
28
+
29
+ def _collect_exif(image):
30
+ """Return (tags, gps) dictionaries keyed by numeric tag id."""
31
+ exif = image.getexif()
32
+ tags = {tag_id: value for tag_id, value in exif.items() if tag_id not in (_EXIF_IFD, _GPS_IFD)}
33
+ tags.update(exif.get_ifd(_EXIF_IFD))
34
+ return tags, dict(exif.get_ifd(_GPS_IFD))
35
+
16
36
  def xtxt_image_exif(file_buffer):
17
37
  """
18
38
  Extract EXIF metadata from images as human-readable text.
@@ -32,23 +52,19 @@ if Image and ExifTags and TAGS:
32
52
  image = Image.open(BytesIO(image_data))
33
53
 
34
54
  # Get EXIF data
35
- exif_data = image._getexif()
55
+ exif_data, gps_data = _collect_exif(image)
36
56
 
37
- if not exif_data:
57
+ if not exif_data and not gps_data:
38
58
  return "NO_EXIF_DATA_FOUND"
39
59
 
40
60
  # Extract readable EXIF information
41
61
  exif_text_lines = []
42
- gps_data = {}
43
62
 
44
63
  # Process main EXIF tags
45
64
  for tag_id, value in exif_data.items():
46
65
  tag_name = TAGS.get(tag_id, f"UnknownTag_{tag_id}")
47
-
48
- # Handle special GPS data
49
- if tag_name == "GPSInfo" and isinstance(value, dict):
50
- gps_data = value
51
- continue
66
+ if isinstance(value, str):
67
+ value = value.strip("\x00 ")
52
68
 
53
69
  # Format common values
54
70
  if tag_name in ["DateTime", "DateTimeOriginal", "DateTimeDigitized"]:
@@ -56,28 +72,27 @@ if Image and ExifTags and TAGS:
56
72
  elif tag_name in ["Make", "Model", "Software", "Artist", "Copyright"]:
57
73
  exif_text_lines.append(f"{tag_name}: {value}")
58
74
  elif tag_name in ["XResolution", "YResolution"]:
59
- if isinstance(value, tuple) and len(value) == 2:
60
- resolution = value[0] / value[1] if value[1] != 0 else value[0]
61
- exif_text_lines.append(f"{tag_name}: {resolution:.1f} dpi")
75
+ ratio = _as_ratio(value)
76
+ if ratio and ratio[1] != 0:
77
+ exif_text_lines.append(f"{tag_name}: {ratio[0] / ratio[1]:.1f} dpi")
62
78
  else:
63
79
  exif_text_lines.append(f"{tag_name}: {value}")
64
80
  elif tag_name in ["FNumber", "FocalLength", "ExposureTime"]:
65
- if isinstance(value, tuple) and len(value) == 2:
66
- if value[1] != 0:
67
- if tag_name == "FNumber":
68
- f_value = value[0] / value[1]
69
- exif_text_lines.append(f"Aperture: f/{f_value:.1f}")
70
- elif tag_name == "FocalLength":
71
- focal_mm = value[0] / value[1]
72
- exif_text_lines.append(f"Focal Length: {focal_mm:.0f}mm")
73
- elif tag_name == "ExposureTime":
74
- if value[0] == 1:
75
- exif_text_lines.append(f"Shutter Speed: 1/{value[1]}s")
76
- else:
77
- exp_time = value[0] / value[1]
78
- exif_text_lines.append(f"Shutter Speed: {exp_time:.3f}s")
81
+ ratio = _as_ratio(value)
82
+ if ratio and ratio[1] != 0:
83
+ numerator, denominator = ratio
84
+ if tag_name == "FNumber":
85
+ exif_text_lines.append(f"Aperture: f/{numerator / denominator:.1f}")
86
+ elif tag_name == "FocalLength":
87
+ exif_text_lines.append(f"Focal Length: {numerator / denominator:.0f}mm")
88
+ elif numerator == 1:
89
+ exif_text_lines.append(f"Shutter Speed: 1/{denominator}s")
79
90
  else:
80
- exif_text_lines.append(f"{tag_name}: {value}")
91
+ exp_time = numerator / denominator
92
+ if exp_time < 1:
93
+ exif_text_lines.append(f"Shutter Speed: 1/{round(1 / exp_time)}s")
94
+ else:
95
+ exif_text_lines.append(f"Shutter Speed: {exp_time:g}s")
81
96
  else:
82
97
  exif_text_lines.append(f"{tag_name}: {value}")
83
98
  elif tag_name == "ISOSpeedRatings":
@@ -111,8 +126,10 @@ if Image and ExifTags and TAGS:
111
126
  }
112
127
  orient_desc = orientations.get(value, f"Orientation {value}")
113
128
  exif_text_lines.append(f"Orientation: {orient_desc}")
114
- elif isinstance(value, (str, int, float)):
115
- # Include other simple values
129
+ elif isinstance(value, (str, numbers.Number)):
130
+ # Include other simple values (rationals as decimals)
131
+ if not isinstance(value, (str, int)):
132
+ value = round(float(value), 4)
116
133
  exif_text_lines.append(f"{tag_name}: {value}")
117
134
 
118
135
  # Process GPS data if available
@@ -130,7 +147,7 @@ if Image and ExifTags and TAGS:
130
147
  lat_dms = gps_info['GPSLatitude']
131
148
  lat_ref = gps_info['GPSLatitudeRef']
132
149
  if len(lat_dms) == 3:
133
- lat_deg = lat_dms[0] + lat_dms[1]/60 + lat_dms[2]/3600
150
+ lat_deg = float(lat_dms[0]) + float(lat_dms[1])/60 + float(lat_dms[2])/3600
134
151
  if lat_ref == 'S':
135
152
  lat_deg = -lat_deg
136
153
  gps_text_lines.append(f"GPS Latitude: {lat_deg:.6f}° {lat_ref}")
@@ -139,7 +156,7 @@ if Image and ExifTags and TAGS:
139
156
  lon_dms = gps_info['GPSLongitude']
140
157
  lon_ref = gps_info['GPSLongitudeRef']
141
158
  if len(lon_dms) == 3:
142
- lon_deg = lon_dms[0] + lon_dms[1]/60 + lon_dms[2]/3600
159
+ lon_deg = float(lon_dms[0]) + float(lon_dms[1])/60 + float(lon_dms[2])/3600
143
160
  if lon_ref == 'W':
144
161
  lon_deg = -lon_deg
145
162
  gps_text_lines.append(f"GPS Longitude: {lon_deg:.6f}° {lon_ref}")
@@ -172,12 +189,5 @@ if Image and ExifTags and TAGS:
172
189
  except Exception as e:
173
190
  print(f"⚠️ Error extracting EXIF from image: {e}")
174
191
  return ""
175
-
176
- # Register EXIF extractor for image formats with dedicated MIME types
177
- # Using EXIF-specific MIME types to avoid conflicts with OCR extractors
178
- register_extractor("image/jpeg+exif", xtxt_image_exif, name="EXIF")
179
- register_extractor("image/jpg+exif", xtxt_image_exif, name="EXIF")
180
- register_extractor("image/png+exif", xtxt_image_exif, name="EXIF")
181
- register_extractor("image/tiff+exif", xtxt_image_exif, name="EXIF")
182
- register_extractor("image/bmp+exif", xtxt_image_exif, name="EXIF")
183
- register_extractor("image/webp+exif", xtxt_image_exif, name="EXIF")
192
+
193
+ # EXIF metadata is exposed through pyxtxt.xtxt_exif(): xtxt() on an image runs OCR.
@@ -0,0 +1,31 @@
1
+ from . import register_extractor
2
+
3
+ try:
4
+ import extract_msg
5
+ from bs4 import BeautifulSoup
6
+ except ImportError:
7
+ extract_msg = None
8
+
9
+ if extract_msg:
10
+ def xtxt_msg(file_buffer):
11
+ # extract_msg accetta direttamente i bytes del file .msg
12
+ content = file_buffer.read()
13
+
14
+ with extract_msg.openMsg(content) as msg:
15
+ body = getattr(msg, "body", None)
16
+ if body and body.strip():
17
+ return body.strip()
18
+
19
+ # Nessun corpo in testo semplice: usa la versione HTML
20
+ html_body = getattr(msg, "htmlBody", None)
21
+ if html_body:
22
+ soup = BeautifulSoup(html_body, "html.parser")
23
+ return soup.get_text(separator="\n").strip()
24
+
25
+ return ""
26
+
27
+ register_extractor(
28
+ "application/vnd.ms-outlook",
29
+ xtxt_msg,
30
+ name="MSG"
31
+ )