pyxtxt 0.3.6__tar.gz → 0.3.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyxtxt-0.3.6/src/pyxtxt.egg-info → pyxtxt-0.3.8}/PKG-INFO +40 -16
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/README.md +37 -11
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/pyproject.toml +3 -5
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/core.py +30 -4
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/audio.py +34 -37
- pyxtxt-0.3.8/src/pyxtxt/estrattori/docx.py +75 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/exif.py +51 -41
- pyxtxt-0.3.8/src/pyxtxt/estrattori/msg.py +31 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/ocr_ollama.py +143 -211
- pyxtxt-0.3.8/src/pyxtxt/estrattori/odt.py +30 -0
- pyxtxt-0.3.8/src/pyxtxt/estrattori/pptx.py +58 -0
- pyxtxt-0.3.8/src/pyxtxt/estrattori/svg.py +30 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8/src/pyxtxt.egg-info}/PKG-INFO +40 -16
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/SOURCES.txt +3 -1
- pyxtxt-0.3.8/tests/test_audio.py +66 -0
- pyxtxt-0.3.8/tests/test_extraction.py +289 -0
- pyxtxt-0.3.8/tests/test_ocr_ollama.py +69 -0
- pyxtxt-0.3.6/src/pyxtxt/estrattori/docx.py +0 -35
- pyxtxt-0.3.6/src/pyxtxt/estrattori/msg.py +0 -41
- pyxtxt-0.3.6/src/pyxtxt/estrattori/odt.py +0 -27
- pyxtxt-0.3.6/src/pyxtxt/estrattori/pptx.py +0 -40
- pyxtxt-0.3.6/src/pyxtxt/estrattori/svg.py +0 -23
- pyxtxt-0.3.6/tests/test_extraction.py +0 -122
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/LICENSE +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/setup.cfg +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/__init__.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/doc.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/eml.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/epub.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/md.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/ocr.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/rtf.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/tex.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/xlsx.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt/examples.py +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/requires.txt +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/src/pyxtxt.egg-info/top_level.txt +0 -0
- {pyxtxt-0.3.6 → pyxtxt-0.3.8}/tests/test_import.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.8
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, etc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -11,16 +11,14 @@ Classifier: Development Status :: 4 - Beta
|
|
|
11
11
|
Classifier: Intended Audience :: Developers
|
|
12
12
|
Classifier: Operating System :: OS Independent
|
|
13
13
|
Classifier: Programming Language :: Python :: 3
|
|
14
|
-
Classifier: Programming Language :: Python :: 3.7
|
|
15
|
-
Classifier: Programming Language :: Python :: 3.8
|
|
16
|
-
Classifier: Programming Language :: Python :: 3.9
|
|
17
14
|
Classifier: Programming Language :: Python :: 3.10
|
|
18
15
|
Classifier: Programming Language :: Python :: 3.11
|
|
19
16
|
Classifier: Programming Language :: Python :: 3.12
|
|
20
17
|
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
21
19
|
Classifier: Topic :: Text Processing
|
|
22
20
|
Classifier: Topic :: Utilities
|
|
23
|
-
Requires-Python: >=3.
|
|
21
|
+
Requires-Python: >=3.10
|
|
24
22
|
Description-Content-Type: text/markdown
|
|
25
23
|
License-File: LICENSE
|
|
26
24
|
Requires-Dist: python-magic; sys_platform != "win32"
|
|
@@ -104,7 +102,7 @@ text = xtxt("report.pdf")
|
|
|
104
102
|
|
|
105
103
|
## ✨ Features
|
|
106
104
|
|
|
107
|
-
- **One function for everything**: `xtxt()` accepts a file path, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
|
|
105
|
+
- **One function for everything**: `xtxt()` accepts a file path, a binary file object, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
|
|
108
106
|
- **Automatic type detection** with `python-magic`, refined by the file extension when libmagic is not specific enough (e.g. Markdown)
|
|
109
107
|
- **Modular dependencies**: each format is an optional extra, the core only needs `python-magic`
|
|
110
108
|
- **Office, web and document formats**: PDF, DOCX, PPTX, XLSX, XLS, ODT, HTML, XML, SVG, Markdown, EPUB, RTF, EML, MSG, LaTeX, DOC, TXT
|
|
@@ -119,8 +117,8 @@ text = xtxt("report.pdf")
|
|
|
119
117
|
| Format | Install extra | Notes |
|
|
120
118
|
|---|---|---|
|
|
121
119
|
| PDF | `pdf` | PyMuPDF |
|
|
122
|
-
| DOCX | `docx` |
|
|
123
|
-
| PPTX | `presentation` | Text
|
|
120
|
+
| DOCX | `docx` | Paragraphs and tables in document order, headers and footers |
|
|
121
|
+
| PPTX | `presentation` | Text boxes, grouped shapes, tables and speaker notes |
|
|
124
122
|
| XLSX, XLS | `spreadsheet` | Every row of every visible sheet, cells joined with ` \| ` |
|
|
125
123
|
| ODT | `odf` | |
|
|
126
124
|
| HTML | `html` | |
|
|
@@ -129,11 +127,11 @@ text = xtxt("report.pdf")
|
|
|
129
127
|
| EPUB | `epub` | |
|
|
130
128
|
| RTF | `rtf` | |
|
|
131
129
|
| EML | `email` | Plain-text and HTML parts |
|
|
132
|
-
| MSG (Outlook) | `outlook` | |
|
|
130
|
+
| MSG (Outlook) | `outlook` | Plain-text body, or the HTML body when there is no plain text |
|
|
133
131
|
| LaTeX | `latex` | |
|
|
134
132
|
| DOC (legacy Word) | — | Needs the `antiword` system tool |
|
|
135
133
|
| TXT and other `text/*` | — | Always available, decoded as UTF-8 |
|
|
136
|
-
| Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download |
|
|
134
|
+
| Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download. MP3, WAV, M4A, AAC, FLAC, OGG/Opus, AIFF, WMA, MP4, MOV, AVI, MKV, WebM |
|
|
137
135
|
| Images (OCR) | `ocr` or `ocr-ollama` | See [OCR from images](#-ocr-from-images) |
|
|
138
136
|
|
|
139
137
|
To list what is available in your installation:
|
|
@@ -149,6 +147,8 @@ print(extxt_available_formats(pretty=True)) # Short names
|
|
|
149
147
|
|
|
150
148
|
## 📦 Installation
|
|
151
149
|
|
|
150
|
+
PyxTxt requires Python 3.10 or newer.
|
|
151
|
+
|
|
152
152
|
Install every extractor (this includes the heavy audio and OCR dependencies):
|
|
153
153
|
|
|
154
154
|
```bash
|
|
@@ -208,9 +208,12 @@ from pyxtxt import xtxt
|
|
|
208
208
|
# From a file path
|
|
209
209
|
text = xtxt("document.pdf")
|
|
210
210
|
|
|
211
|
-
# From
|
|
211
|
+
# From a file object opened in binary mode
|
|
212
212
|
with open("document.docx", "rb") as f:
|
|
213
|
-
|
|
213
|
+
text = xtxt(f)
|
|
214
|
+
|
|
215
|
+
# From an in-memory buffer
|
|
216
|
+
buffer = io.BytesIO(docx_bytes)
|
|
214
217
|
text = xtxt(buffer)
|
|
215
218
|
|
|
216
219
|
# Give the buffer a name to help type detection (useful for Markdown, LaTeX, RTF)
|
|
@@ -344,12 +347,12 @@ print((files("pyxtxt") / "examples.py").read_text())
|
|
|
344
347
|
|
|
345
348
|
## ⚠️ Known limitations
|
|
346
349
|
|
|
347
|
-
- **Supported inputs**: file paths, `io.BytesIO`, `bytes` and `requests.Response`.
|
|
348
|
-
|
|
350
|
+
- **Supported inputs**: file paths, binary file objects, `io.BytesIO`, `bytes` and `requests.Response`.
|
|
351
|
+
File objects opened in text mode are rejected: open them with `"rb"`.
|
|
349
352
|
- **Type detection without a file name**: libmagic cannot tell apart some formats from raw bytes (legacy Office files
|
|
350
353
|
share the same signature; Markdown looks like plain text). Pass a file path, or set `buffer.name`, when possible.
|
|
351
354
|
- **Legacy PowerPoint (`.ppt`)** is not supported.
|
|
352
|
-
- **DOCX**: text
|
|
355
|
+
- **DOCX**: text boxes, footnotes and comments are not extracted. **PPTX**: text inside charts is not extracted.
|
|
353
356
|
- Errors are reported with messages printed to standard output and the functions return `None` or an empty string;
|
|
354
357
|
they do not raise exceptions.
|
|
355
358
|
|
|
@@ -374,7 +377,7 @@ pip install -e ".[pdf,docx,presentation,spreadsheet,odf,html,markdown,epub,rtf,e
|
|
|
374
377
|
pytest
|
|
375
378
|
```
|
|
376
379
|
|
|
377
|
-
Tests for formats whose libraries are not installed are skipped. CI runs the test suite on Python 3.10–3.
|
|
380
|
+
Tests for formats whose libraries are not installed are skipped. CI runs the test suite on Python 3.10–3.14.
|
|
378
381
|
|
|
379
382
|
### Releasing
|
|
380
383
|
|
|
@@ -407,6 +410,27 @@ Pull requests, issues and feedback are welcome.
|
|
|
407
410
|
|
|
408
411
|
## 📊 Changelog
|
|
409
412
|
|
|
413
|
+
### v0.3.8
|
|
414
|
+
- **NEW**: `xtxt()` accepts binary file objects, e.g. `xtxt(open("file.pdf", "rb"))`
|
|
415
|
+
- **FIXED**: XLSX passed as `bytes` or `BytesIO` was detected as a ZIP archive and rejected
|
|
416
|
+
- **FIXED**: WAV, M4A, AAC, AIFF, WMA and MKV files were never transcribed (libmagic reports them as `audio/x-wav`,
|
|
417
|
+
`audio/x-m4a`, `video/x-matroska`, ... which were not registered)
|
|
418
|
+
- **FIXED**: MSG extraction always failed (`extract_msg.Message` has no `extract()` method)
|
|
419
|
+
- **FIXED**: SVG extraction failed on `<text>` elements containing `<tspan>`
|
|
420
|
+
- **IMPROVED**: DOCX now includes tables (in document order), headers and footers
|
|
421
|
+
- **IMPROVED**: PPTX now includes grouped shapes, tables and speaker notes
|
|
422
|
+
- **IMPROVED**: ODT now includes headings and text inside spans, links and lists
|
|
423
|
+
- **IMPROVED**: EXIF output formats aperture, shutter speed and focal length as intended and reads PNG/WebP metadata
|
|
424
|
+
through Pillow's public API; the fake `image/*+exif` MIME types are gone from `extxt_available_formats()`
|
|
425
|
+
- **IMPROVED**: Whisper temporary files are always deleted, also when transcription fails
|
|
426
|
+
- **IMPROVED**: Ollama OCR stops trying fallback models when the server is not reachable;
|
|
427
|
+
`xtxt_image_with_confidence()` now uses the same context-aware prompts as `xtxt()`;
|
|
428
|
+
hallucination keywords match whole words (e.g. "roman" no longer fires on "romance")
|
|
429
|
+
|
|
430
|
+
### v0.3.7
|
|
431
|
+
- Python 3.10 or newer is now required; metadata no longer lists the end-of-life versions 3.7–3.9, which were never tested
|
|
432
|
+
- Python 3.14 is tested in CI and declared as supported
|
|
433
|
+
|
|
410
434
|
### v0.3.6
|
|
411
435
|
- **FIXED**: `import pyxtxt` crashed with `AttributeError` unless both `ollama` and Pillow were installed (regression in 0.3.4.2 and 0.3.5)
|
|
412
436
|
- **FIXED**: Markdown, RTF and LaTeX files were returned as raw source instead of being converted to text
|
|
@@ -19,7 +19,7 @@ text = xtxt("report.pdf")
|
|
|
19
19
|
|
|
20
20
|
## ✨ Features
|
|
21
21
|
|
|
22
|
-
- **One function for everything**: `xtxt()` accepts a file path, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
|
|
22
|
+
- **One function for everything**: `xtxt()` accepts a file path, a binary file object, an `io.BytesIO` buffer, raw `bytes` or a `requests.Response`
|
|
23
23
|
- **Automatic type detection** with `python-magic`, refined by the file extension when libmagic is not specific enough (e.g. Markdown)
|
|
24
24
|
- **Modular dependencies**: each format is an optional extra, the core only needs `python-magic`
|
|
25
25
|
- **Office, web and document formats**: PDF, DOCX, PPTX, XLSX, XLS, ODT, HTML, XML, SVG, Markdown, EPUB, RTF, EML, MSG, LaTeX, DOC, TXT
|
|
@@ -34,8 +34,8 @@ text = xtxt("report.pdf")
|
|
|
34
34
|
| Format | Install extra | Notes |
|
|
35
35
|
|---|---|---|
|
|
36
36
|
| PDF | `pdf` | PyMuPDF |
|
|
37
|
-
| DOCX | `docx` |
|
|
38
|
-
| PPTX | `presentation` | Text
|
|
37
|
+
| DOCX | `docx` | Paragraphs and tables in document order, headers and footers |
|
|
38
|
+
| PPTX | `presentation` | Text boxes, grouped shapes, tables and speaker notes |
|
|
39
39
|
| XLSX, XLS | `spreadsheet` | Every row of every visible sheet, cells joined with ` \| ` |
|
|
40
40
|
| ODT | `odf` | |
|
|
41
41
|
| HTML | `html` | |
|
|
@@ -44,11 +44,11 @@ text = xtxt("report.pdf")
|
|
|
44
44
|
| EPUB | `epub` | |
|
|
45
45
|
| RTF | `rtf` | |
|
|
46
46
|
| EML | `email` | Plain-text and HTML parts |
|
|
47
|
-
| MSG (Outlook) | `outlook` | |
|
|
47
|
+
| MSG (Outlook) | `outlook` | Plain-text body, or the HTML body when there is no plain text |
|
|
48
48
|
| LaTeX | `latex` | |
|
|
49
49
|
| DOC (legacy Word) | — | Needs the `antiword` system tool |
|
|
50
50
|
| TXT and other `text/*` | — | Always available, decoded as UTF-8 |
|
|
51
|
-
| Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download |
|
|
51
|
+
| Audio and video | `audio` | Whisper, needs `ffmpeg`; heavy download. MP3, WAV, M4A, AAC, FLAC, OGG/Opus, AIFF, WMA, MP4, MOV, AVI, MKV, WebM |
|
|
52
52
|
| Images (OCR) | `ocr` or `ocr-ollama` | See [OCR from images](#-ocr-from-images) |
|
|
53
53
|
|
|
54
54
|
To list what is available in your installation:
|
|
@@ -64,6 +64,8 @@ print(extxt_available_formats(pretty=True)) # Short names
|
|
|
64
64
|
|
|
65
65
|
## 📦 Installation
|
|
66
66
|
|
|
67
|
+
PyxTxt requires Python 3.10 or newer.
|
|
68
|
+
|
|
67
69
|
Install every extractor (this includes the heavy audio and OCR dependencies):
|
|
68
70
|
|
|
69
71
|
```bash
|
|
@@ -123,9 +125,12 @@ from pyxtxt import xtxt
|
|
|
123
125
|
# From a file path
|
|
124
126
|
text = xtxt("document.pdf")
|
|
125
127
|
|
|
126
|
-
# From
|
|
128
|
+
# From a file object opened in binary mode
|
|
127
129
|
with open("document.docx", "rb") as f:
|
|
128
|
-
|
|
130
|
+
text = xtxt(f)
|
|
131
|
+
|
|
132
|
+
# From an in-memory buffer
|
|
133
|
+
buffer = io.BytesIO(docx_bytes)
|
|
129
134
|
text = xtxt(buffer)
|
|
130
135
|
|
|
131
136
|
# Give the buffer a name to help type detection (useful for Markdown, LaTeX, RTF)
|
|
@@ -259,12 +264,12 @@ print((files("pyxtxt") / "examples.py").read_text())
|
|
|
259
264
|
|
|
260
265
|
## ⚠️ Known limitations
|
|
261
266
|
|
|
262
|
-
- **Supported inputs**: file paths, `io.BytesIO`, `bytes` and `requests.Response`.
|
|
263
|
-
|
|
267
|
+
- **Supported inputs**: file paths, binary file objects, `io.BytesIO`, `bytes` and `requests.Response`.
|
|
268
|
+
File objects opened in text mode are rejected: open them with `"rb"`.
|
|
264
269
|
- **Type detection without a file name**: libmagic cannot tell apart some formats from raw bytes (legacy Office files
|
|
265
270
|
share the same signature; Markdown looks like plain text). Pass a file path, or set `buffer.name`, when possible.
|
|
266
271
|
- **Legacy PowerPoint (`.ppt`)** is not supported.
|
|
267
|
-
- **DOCX**: text
|
|
272
|
+
- **DOCX**: text boxes, footnotes and comments are not extracted. **PPTX**: text inside charts is not extracted.
|
|
268
273
|
- Errors are reported with messages printed to standard output and the functions return `None` or an empty string;
|
|
269
274
|
they do not raise exceptions.
|
|
270
275
|
|
|
@@ -289,7 +294,7 @@ pip install -e ".[pdf,docx,presentation,spreadsheet,odf,html,markdown,epub,rtf,e
|
|
|
289
294
|
pytest
|
|
290
295
|
```
|
|
291
296
|
|
|
292
|
-
Tests for formats whose libraries are not installed are skipped. CI runs the test suite on Python 3.10–3.
|
|
297
|
+
Tests for formats whose libraries are not installed are skipped. CI runs the test suite on Python 3.10–3.14.
|
|
293
298
|
|
|
294
299
|
### Releasing
|
|
295
300
|
|
|
@@ -322,6 +327,27 @@ Pull requests, issues and feedback are welcome.
|
|
|
322
327
|
|
|
323
328
|
## 📊 Changelog
|
|
324
329
|
|
|
330
|
+
### v0.3.8
|
|
331
|
+
- **NEW**: `xtxt()` accepts binary file objects, e.g. `xtxt(open("file.pdf", "rb"))`
|
|
332
|
+
- **FIXED**: XLSX passed as `bytes` or `BytesIO` was detected as a ZIP archive and rejected
|
|
333
|
+
- **FIXED**: WAV, M4A, AAC, AIFF, WMA and MKV files were never transcribed (libmagic reports them as `audio/x-wav`,
|
|
334
|
+
`audio/x-m4a`, `video/x-matroska`, ... which were not registered)
|
|
335
|
+
- **FIXED**: MSG extraction always failed (`extract_msg.Message` has no `extract()` method)
|
|
336
|
+
- **FIXED**: SVG extraction failed on `<text>` elements containing `<tspan>`
|
|
337
|
+
- **IMPROVED**: DOCX now includes tables (in document order), headers and footers
|
|
338
|
+
- **IMPROVED**: PPTX now includes grouped shapes, tables and speaker notes
|
|
339
|
+
- **IMPROVED**: ODT now includes headings and text inside spans, links and lists
|
|
340
|
+
- **IMPROVED**: EXIF output formats aperture, shutter speed and focal length as intended and reads PNG/WebP metadata
|
|
341
|
+
through Pillow's public API; the fake `image/*+exif` MIME types are gone from `extxt_available_formats()`
|
|
342
|
+
- **IMPROVED**: Whisper temporary files are always deleted, also when transcription fails
|
|
343
|
+
- **IMPROVED**: Ollama OCR stops trying fallback models when the server is not reachable;
|
|
344
|
+
`xtxt_image_with_confidence()` now uses the same context-aware prompts as `xtxt()`;
|
|
345
|
+
hallucination keywords match whole words (e.g. "roman" no longer fires on "romance")
|
|
346
|
+
|
|
347
|
+
### v0.3.7
|
|
348
|
+
- Python 3.10 or newer is now required; metadata no longer lists the end-of-life versions 3.7–3.9, which were never tested
|
|
349
|
+
- Python 3.14 is tested in CI and declared as supported
|
|
350
|
+
|
|
325
351
|
### v0.3.6
|
|
326
352
|
- **FIXED**: `import pyxtxt` crashed with `AttributeError` unless both `ollama` and Pillow were installed (regression in 0.3.4.2 and 0.3.5)
|
|
327
353
|
- **FIXED**: Markdown, RTF and LaTeX files were returned as raw source instead of being converted to text
|
|
@@ -1,25 +1,23 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "pyxtxt"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.8"
|
|
4
4
|
description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, etc.)."
|
|
5
5
|
classifiers = [
|
|
6
6
|
"Development Status :: 4 - Beta",
|
|
7
7
|
"Intended Audience :: Developers",
|
|
8
8
|
"Operating System :: OS Independent",
|
|
9
9
|
"Programming Language :: Python :: 3",
|
|
10
|
-
"Programming Language :: Python :: 3.7",
|
|
11
|
-
"Programming Language :: Python :: 3.8",
|
|
12
|
-
"Programming Language :: Python :: 3.9",
|
|
13
10
|
"Programming Language :: Python :: 3.10",
|
|
14
11
|
"Programming Language :: Python :: 3.11",
|
|
15
12
|
"Programming Language :: Python :: 3.12",
|
|
16
13
|
"Programming Language :: Python :: 3.13",
|
|
14
|
+
"Programming Language :: Python :: 3.14",
|
|
17
15
|
"Topic :: Text Processing",
|
|
18
16
|
"Topic :: Utilities"
|
|
19
17
|
]
|
|
20
18
|
|
|
21
19
|
readme = "README.md"
|
|
22
|
-
requires-python = ">=3.
|
|
20
|
+
requires-python = ">=3.10"
|
|
23
21
|
|
|
24
22
|
license = "MIT"
|
|
25
23
|
license-files = ["LICENSE"]
|
|
@@ -17,6 +17,15 @@ _TEXT_EXTENSION_HINTS = {
|
|
|
17
17
|
}
|
|
18
18
|
|
|
19
19
|
|
|
20
|
+
def _detect_mime_type(data: bytes) -> str:
|
|
21
|
+
"""Detect the MIME type of in-memory data.
|
|
22
|
+
|
|
23
|
+
The whole buffer is passed to libmagic: the first few KB are not always
|
|
24
|
+
enough to tell apart ZIP-based formats (e.g. XLSX is seen as application/zip).
|
|
25
|
+
"""
|
|
26
|
+
return magic.from_buffer(data, mime=True)
|
|
27
|
+
|
|
28
|
+
|
|
20
29
|
def _resolve_mime_type(mime_type: str, name: Optional[str]) -> str:
|
|
21
30
|
"""Refine the detected MIME type and map unknown text types to text/plain."""
|
|
22
31
|
if mime_type == "text/plain" and name:
|
|
@@ -45,7 +54,7 @@ def _(file_input: str) -> Optional[str]:
|
|
|
45
54
|
data = f.read()
|
|
46
55
|
buffer = io.BytesIO(data)
|
|
47
56
|
buffer.name = file_input
|
|
48
|
-
buffer.mimeType = magic.
|
|
57
|
+
buffer.mimeType = magic.from_file(file_input, mime=True)
|
|
49
58
|
return xtxt(buffer)
|
|
50
59
|
except Exception as e:
|
|
51
60
|
print(f"⚠️ File opening error '{file_input}': {e}")
|
|
@@ -59,8 +68,7 @@ def _(file_input: io.BytesIO) -> Optional[str]:
|
|
|
59
68
|
if hasattr(file_input, "mimeType"):
|
|
60
69
|
mime_type = file_input.mimeType
|
|
61
70
|
else:
|
|
62
|
-
file_input.
|
|
63
|
-
mime_type = magic.Magic(mime=True).from_buffer(file_input.read(2048))
|
|
71
|
+
mime_type = _detect_mime_type(file_input.getvalue())
|
|
64
72
|
file_input.seek(0)
|
|
65
73
|
if not getattr(file_input, "name", None):
|
|
66
74
|
file_input.name = "IO_buffer"
|
|
@@ -83,13 +91,31 @@ def _(file_input: bytes) -> Optional[str]:
|
|
|
83
91
|
try:
|
|
84
92
|
buffer = io.BytesIO(file_input)
|
|
85
93
|
buffer.name = "bytes_input"
|
|
86
|
-
buffer.mimeType =
|
|
94
|
+
buffer.mimeType = _detect_mime_type(file_input)
|
|
87
95
|
return xtxt(buffer)
|
|
88
96
|
except Exception as e:
|
|
89
97
|
print(f"❌ Error processing bytes: {e}")
|
|
90
98
|
return None
|
|
91
99
|
|
|
92
100
|
|
|
101
|
+
@xtxt.register
|
|
102
|
+
def _(file_input: io.IOBase) -> Optional[str]:
|
|
103
|
+
"""Extract text from a binary file object, e.g. one returned by open(path, "rb")."""
|
|
104
|
+
try:
|
|
105
|
+
data = file_input.read()
|
|
106
|
+
if isinstance(data, str):
|
|
107
|
+
print("⚠️ File object opened in text mode: open it in binary mode ('rb')")
|
|
108
|
+
return None
|
|
109
|
+
buffer = io.BytesIO(data)
|
|
110
|
+
name = getattr(file_input, "name", None)
|
|
111
|
+
buffer.name = name if isinstance(name, str) else "file_object"
|
|
112
|
+
buffer.mimeType = _detect_mime_type(data)
|
|
113
|
+
return xtxt(buffer)
|
|
114
|
+
except Exception as e:
|
|
115
|
+
print(f"❌ Error processing file object: {e}")
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
|
|
93
119
|
# Optional support for requests.Response
|
|
94
120
|
try:
|
|
95
121
|
import requests
|
|
@@ -10,69 +10,66 @@ except ImportError:
|
|
|
10
10
|
|
|
11
11
|
if whisper:
|
|
12
12
|
_whisper_model = None
|
|
13
|
-
|
|
13
|
+
|
|
14
14
|
def _get_model():
|
|
15
15
|
global _whisper_model
|
|
16
16
|
if _whisper_model is None:
|
|
17
17
|
_whisper_model = whisper.load_model("base")
|
|
18
18
|
return _whisper_model
|
|
19
|
-
|
|
19
|
+
|
|
20
20
|
def xtxt_audio_whisper(file_buffer):
|
|
21
|
+
temp_path = None
|
|
21
22
|
try:
|
|
22
|
-
# Usa un
|
|
23
|
+
# Usa un suffisso generico - Whisper + FFmpeg gestiscono il formato
|
|
23
24
|
with tempfile.NamedTemporaryFile(delete=False) as temp_file:
|
|
24
25
|
temp_file.write(file_buffer.read())
|
|
25
|
-
temp_file.
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
26
|
+
temp_path = temp_file.name
|
|
27
|
+
|
|
28
|
+
model = _get_model()
|
|
29
|
+
result = model.transcribe(
|
|
30
|
+
temp_path,
|
|
30
31
|
language=None,
|
|
31
32
|
task="transcribe",
|
|
32
33
|
temperature=[0.0, 0.2, 0.4, 0.6, 0.8, 1.0],
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
os.unlink(temp_file.name)
|
|
37
|
-
text_with_languages = []
|
|
38
|
-
current_lang = None
|
|
39
|
-
|
|
40
|
-
for segment in result['segments']:
|
|
41
|
-
# Rileva cambio di lingua (logica semplificata)
|
|
42
|
-
if 'language' in segment and segment['language'] != current_lang:
|
|
43
|
-
current_lang = segment['language']
|
|
44
|
-
text_with_languages.append(f"\n[{current_lang.upper()}]")
|
|
45
|
-
|
|
46
|
-
text_with_languages.append(segment['text'])
|
|
47
|
-
|
|
48
|
-
return " ".join(text_with_languages).strip()
|
|
49
|
-
|
|
34
|
+
)
|
|
35
|
+
return result["text"].strip()
|
|
36
|
+
|
|
50
37
|
except Exception as e:
|
|
51
38
|
print(f"⚠️ Error while extracting audio with Whisper: {e}")
|
|
52
39
|
return ""
|
|
53
|
-
|
|
54
|
-
|
|
40
|
+
finally:
|
|
41
|
+
if temp_path is not None:
|
|
42
|
+
os.unlink(temp_path)
|
|
43
|
+
|
|
44
|
+
# Registra per tutti i formati audio comuni.
|
|
45
|
+
# Comprende sia i nomi standard sia quelli restituiti da libmagic (es. audio/x-wav).
|
|
55
46
|
audio_formats = [
|
|
56
|
-
"audio/wav", "audio/wave",
|
|
47
|
+
"audio/wav", "audio/wave", "audio/x-wav",
|
|
57
48
|
"audio/mp3", "audio/mpeg",
|
|
58
|
-
"audio/m4a", "audio/mp4",
|
|
59
|
-
"audio/flac",
|
|
49
|
+
"audio/m4a", "audio/x-m4a", "audio/mp4",
|
|
50
|
+
"audio/flac", "audio/x-flac",
|
|
60
51
|
"audio/ogg", "audio/ogg-vorbis",
|
|
61
52
|
"audio/opus",
|
|
62
|
-
"audio/aac",
|
|
63
|
-
"audio/
|
|
64
|
-
"audio/
|
|
53
|
+
"audio/aac", "audio/x-hx-aac-adts",
|
|
54
|
+
"audio/aiff", "audio/x-aiff",
|
|
55
|
+
"audio/wma", "audio/x-ms-wma",
|
|
56
|
+
"audio/webm",
|
|
65
57
|
]
|
|
66
|
-
|
|
58
|
+
|
|
67
59
|
for format_type in audio_formats:
|
|
68
60
|
register_extractor(format_type, xtxt_audio_whisper, name="Whisper Audio")
|
|
69
|
-
|
|
61
|
+
|
|
62
|
+
# Formati video: Whisper estrae la traccia audio tramite FFmpeg
|
|
70
63
|
video_audio_formats = [
|
|
71
|
-
"video/mp4",
|
|
64
|
+
"video/mp4", "video/x-m4v",
|
|
72
65
|
"video/quicktime", # .mov
|
|
73
66
|
"video/x-msvideo", # .avi
|
|
74
67
|
"video/webm",
|
|
75
|
-
"video/mkv"
|
|
68
|
+
"video/mkv", "video/x-matroska",
|
|
69
|
+
"video/x-ms-asf", # .wmv / .wma
|
|
70
|
+
"video/mpeg",
|
|
71
|
+
"video/ogg",
|
|
72
|
+
"video/3gpp",
|
|
76
73
|
]
|
|
77
74
|
|
|
78
75
|
for format_type in video_audio_formats:
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
import io
|
|
3
|
+
import zipfile
|
|
4
|
+
try:
|
|
5
|
+
from docx import Document
|
|
6
|
+
from docx.oxml.ns import qn
|
|
7
|
+
from docx.table import Table
|
|
8
|
+
from docx.text.paragraph import Paragraph
|
|
9
|
+
except ImportError:
|
|
10
|
+
Document = None
|
|
11
|
+
|
|
12
|
+
if Document:
|
|
13
|
+
def _block_lines(parent, element):
|
|
14
|
+
"""Yield the text of paragraphs and tables in document order."""
|
|
15
|
+
for child in element.iterchildren():
|
|
16
|
+
if child.tag == qn("w:p"):
|
|
17
|
+
yield Paragraph(child, parent).text
|
|
18
|
+
elif child.tag == qn("w:tbl"):
|
|
19
|
+
yield from _table_lines(Table(child, parent))
|
|
20
|
+
|
|
21
|
+
def _table_lines(table):
|
|
22
|
+
"""Yield one line per table row, cells separated by ' | '."""
|
|
23
|
+
for row in table.rows:
|
|
24
|
+
seen = []
|
|
25
|
+
cells = []
|
|
26
|
+
for cell in row.cells:
|
|
27
|
+
# Merged cells are returned once per grid column: keep the first one
|
|
28
|
+
if any(cell._tc is tc for tc in seen):
|
|
29
|
+
continue
|
|
30
|
+
seen.append(cell._tc)
|
|
31
|
+
cell_text = " ".join(line for line in _block_lines(cell, cell._tc) if line)
|
|
32
|
+
cells.append(cell_text.strip())
|
|
33
|
+
if any(cells):
|
|
34
|
+
yield " | ".join(cells)
|
|
35
|
+
|
|
36
|
+
def _header_footer_lines(doc, attribute_names):
|
|
37
|
+
lines = []
|
|
38
|
+
for section in doc.sections:
|
|
39
|
+
for attribute in attribute_names:
|
|
40
|
+
part = getattr(section, attribute, None)
|
|
41
|
+
if part is None or part.is_linked_to_previous:
|
|
42
|
+
continue
|
|
43
|
+
for line in _block_lines(part, part._element):
|
|
44
|
+
if line and line not in lines:
|
|
45
|
+
lines.append(line)
|
|
46
|
+
return lines
|
|
47
|
+
|
|
48
|
+
def xtxt_docx(file_buffer) -> str:
|
|
49
|
+
try:
|
|
50
|
+
# Copia del buffer per poterlo riutilizzare
|
|
51
|
+
file_buffer.seek(0)
|
|
52
|
+
data = file_buffer.read()
|
|
53
|
+
buffer_copy = io.BytesIO(data)
|
|
54
|
+
|
|
55
|
+
if not zipfile.is_zipfile(buffer_copy):
|
|
56
|
+
print("⚠️ Invalid DOCX (not a ZIP file)")
|
|
57
|
+
return ""
|
|
58
|
+
|
|
59
|
+
buffer_copy.seek(0)
|
|
60
|
+
doc = Document(buffer_copy)
|
|
61
|
+
|
|
62
|
+
headers = _header_footer_lines(doc, ("first_page_header", "header", "even_page_header"))
|
|
63
|
+
body = list(_block_lines(doc, doc.element.body))
|
|
64
|
+
footers = _header_footer_lines(doc, ("first_page_footer", "footer", "even_page_footer"))
|
|
65
|
+
return "\n".join(headers + body + footers)
|
|
66
|
+
|
|
67
|
+
except Exception as e:
|
|
68
|
+
print(f"⚠️ Error during extraction DOCX: {e}")
|
|
69
|
+
return ""
|
|
70
|
+
|
|
71
|
+
register_extractor(
|
|
72
|
+
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
73
|
+
xtxt_docx,
|
|
74
|
+
name="DOCX"
|
|
75
|
+
)
|