pyxtxt 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pyxtxt-0.0.1/LICENCSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2025 Giuseppe Levi
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
pyxtxt-0.0.1/PKG-INFO ADDED
@@ -0,0 +1,183 @@
1
+ Metadata-Version: 2.4
2
+ Name: pyxtxt
3
+ Version: 0.0.1
4
+ Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
+ Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
+ License: MIT License
7
+
8
+ Copyright (c) 2025 Giuseppe Levi
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ Requires-Python: >=3.7
29
+ Description-Content-Type: text/markdown
30
+ Requires-Dist: python-magic; sys_platform != "win32"
31
+ Requires-Dist: python-magic-bin; sys_platform == "win32"
32
+ Provides-Extra: pdf
33
+ Requires-Dist: PyMuPDF; extra == "pdf"
34
+ Provides-Extra: docx
35
+ Requires-Dist: python-docx; extra == "docx"
36
+ Provides-Extra: presentation
37
+ Requires-Dist: python-pptx; extra == "presentation"
38
+ Provides-Extra: spreadsheet
39
+ Requires-Dist: openpyxl; extra == "spreadsheet"
40
+ Requires-Dist: xlrd; extra == "spreadsheet"
41
+ Provides-Extra: odf
42
+ Requires-Dist: odfpy; extra == "odf"
43
+ Provides-Extra: html
44
+ Requires-Dist: beautifulsoup4; extra == "html"
45
+ Requires-Dist: lxml; extra == "html"
46
+ Provides-Extra: doc
47
+ Requires-Dist: textract; extra == "doc"
48
+ Provides-Extra: all
49
+ Requires-Dist: textract; extra == "all"
50
+ Requires-Dist: PyMuPDF; extra == "all"
51
+ Requires-Dist: python-docx; extra == "all"
52
+ Requires-Dist: python-pptx; extra == "all"
53
+ Requires-Dist: openpyxl; extra == "all"
54
+ Requires-Dist: xlrd; extra == "all"
55
+ Requires-Dist: odfpy; extra == "all"
56
+ Requires-Dist: beautifulsoup4; extra == "all"
57
+ Requires-Dist: lxml; extra == "all"
58
+
59
+ # PyxTxt
60
+
61
+ [![PyPI version](https://img.shields.io/pypi/v/pyxtxt.svg)](https://pypi.org/project/pyxtxt/)
62
+ [![Python versions](https://img.shields.io/pypi/pyversions/pyxtxt.svg)](https://pypi.org/project/pyxtxt/)
63
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
64
+
65
+ **PyxTxt** is a simple and powerful Python library to extract text from various file formats.
66
+ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy XLS files, and more.
67
+
68
+ ---
69
+
70
+ ## ✨ Features
71
+
72
+ - Extracts text from both file paths and in-memory buffers (`io.BytesIO`).
73
+ - Supports multiple formats: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls,.ppt).
74
+ - Automatically detects MIME type using `python-magic`.
75
+ - Compatible with modern and legacy formats.
76
+ - Can handle streamed content without saving to disk (with some limitations).
77
+
78
+ ---
79
+
80
+ ## 📦 Installation
81
+
82
+ The library i modular so you can install all modules:
83
+
84
+ ```bash
85
+ pip install pyxtxt[all]
86
+ ```
87
+ or just the modules you need:
88
+ ```bash
89
+ pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
90
+ ```
91
+ Beause needed libraries are common installing the html module will enable also SVG and XML.
92
+ The architecture is designed to be able to grow with new modules to work with other formats as well.
93
+ ## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
94
+ The pyproject.toml file should select the correct version for your system. But if you have any problem you can install it manually.
95
+
96
+ **On Ubuntu/Debian:**
97
+
98
+ ```bash
99
+ sudo apt install libmagic1
100
+ ```
101
+
102
+ **On Mac (Homebrew):**
103
+
104
+ ```bash
105
+ brew install libmagic
106
+ ```
107
+ **On Windows:**
108
+
109
+ Use python-magic-bin instead of python-magic for easier installation.
110
+
111
+ ## 🛠️ Dependencies
112
+ - PyMuPDF (fitz)
113
+
114
+ - beautifulsoup4
115
+
116
+ - python-docx
117
+
118
+ - python-pptx
119
+
120
+ - odfpy
121
+
122
+ - openpyxl
123
+
124
+ - lxml
125
+
126
+ - xlrd (<2.0.0)
127
+
128
+ - python-magic
129
+
130
+ Dependencies are automatically installed from pyproject.toml.
131
+
132
+ ## 📚 Usage Example
133
+ Extract text from a file path:
134
+
135
+ ```python
136
+ from pyxtxt import xtxt
137
+
138
+ text = xtxt("document.pdf")
139
+ print(text)
140
+ ```
141
+ Extract text from a file-like buffer:
142
+
143
+ ```python
144
+ import io
145
+
146
+ with open("document.docx", "rb") as f:
147
+ buffer = io.BytesIO(f.read())
148
+
149
+ from pyxtxt import xtxt
150
+ text = xtxt(buffer)
151
+ print(text)
152
+ ```
153
+ Show available formats:
154
+ from pyxtxt import extxt_available_formats
155
+ ```python
156
+ from pyxtxt import extxt_available_formats
157
+ text = extxt_available_formats()
158
+ print(text)
159
+ # For a pretty printing
160
+ text = extxt_available_formats(True)
161
+ print(text)
162
+ ```
163
+ ## ⚠️ Known Limitations
164
+ When passing a raw stream (io.BytesIO) without a filename, legacy files (.doc, .xls, .ppt) may not be correctly detected.
165
+
166
+ This is a limitation of libmagic beacuse the signature byte sequence at the start of doc/xls/ppt is exactly the same (b'\xD0\xCF\x11\xE0\xA1\xB1\x1A\xE1'),
167
+ not of pyxtxt.
168
+
169
+ If available, using the original filename is highly recommended.
170
+
171
+ To extract text from documents in MSWrite's old .doc format, it is necessary to install antiword.
172
+
173
+ ```bash
174
+ sudo apt-get update
175
+ sudo apt-get -y install antiword
176
+ ```
177
+
178
+ ## 🔒 License
179
+ Distributed under the MIT License.
180
+
181
+ The software is provided "as is" without any warranty of any kind.
182
+
183
+ Pull requests, issues, and feedback are warmly welcome! 🚀
pyxtxt-0.0.1/README.md ADDED
@@ -0,0 +1,125 @@
1
+ # PyxTxt
2
+
3
+ [![PyPI version](https://img.shields.io/pypi/v/pyxtxt.svg)](https://pypi.org/project/pyxtxt/)
4
+ [![Python versions](https://img.shields.io/pypi/pyversions/pyxtxt.svg)](https://pypi.org/project/pyxtxt/)
5
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
6
+
7
+ **PyxTxt** is a simple and powerful Python library to extract text from various file formats.
8
+ It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy XLS files, and more.
9
+
10
+ ---
11
+
12
+ ## ✨ Features
13
+
14
+ - Extracts text from both file paths and in-memory buffers (`io.BytesIO`).
15
+ - Supports multiple formats: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls,.ppt).
16
+ - Automatically detects MIME type using `python-magic`.
17
+ - Compatible with modern and legacy formats.
18
+ - Can handle streamed content without saving to disk (with some limitations).
19
+
20
+ ---
21
+
22
+ ## 📦 Installation
23
+
24
+ The library i modular so you can install all modules:
25
+
26
+ ```bash
27
+ pip install pyxtxt[all]
28
+ ```
29
+ or just the modules you need:
30
+ ```bash
31
+ pip install pyxtxt[pdf,odf,docx,presentation,spreadsheet,html]
32
+ ```
33
+ Beause needed libraries are common installing the html module will enable also SVG and XML.
34
+ The architecture is designed to be able to grow with new modules to work with other formats as well.
35
+ ## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
36
+ The pyproject.toml file should select the correct version for your system. But if you have any problem you can install it manually.
37
+
38
+ **On Ubuntu/Debian:**
39
+
40
+ ```bash
41
+ sudo apt install libmagic1
42
+ ```
43
+
44
+ **On Mac (Homebrew):**
45
+
46
+ ```bash
47
+ brew install libmagic
48
+ ```
49
+ **On Windows:**
50
+
51
+ Use python-magic-bin instead of python-magic for easier installation.
52
+
53
+ ## 🛠️ Dependencies
54
+ - PyMuPDF (fitz)
55
+
56
+ - beautifulsoup4
57
+
58
+ - python-docx
59
+
60
+ - python-pptx
61
+
62
+ - odfpy
63
+
64
+ - openpyxl
65
+
66
+ - lxml
67
+
68
+ - xlrd (<2.0.0)
69
+
70
+ - python-magic
71
+
72
+ Dependencies are automatically installed from pyproject.toml.
73
+
74
+ ## 📚 Usage Example
75
+ Extract text from a file path:
76
+
77
+ ```python
78
+ from pyxtxt import xtxt
79
+
80
+ text = xtxt("document.pdf")
81
+ print(text)
82
+ ```
83
+ Extract text from a file-like buffer:
84
+
85
+ ```python
86
+ import io
87
+
88
+ with open("document.docx", "rb") as f:
89
+ buffer = io.BytesIO(f.read())
90
+
91
+ from pyxtxt import xtxt
92
+ text = xtxt(buffer)
93
+ print(text)
94
+ ```
95
+ Show available formats:
96
+ from pyxtxt import extxt_available_formats
97
+ ```python
98
+ from pyxtxt import extxt_available_formats
99
+ text = extxt_available_formats()
100
+ print(text)
101
+ # For a pretty printing
102
+ text = extxt_available_formats(True)
103
+ print(text)
104
+ ```
105
+ ## ⚠️ Known Limitations
106
+ When passing a raw stream (io.BytesIO) without a filename, legacy files (.doc, .xls, .ppt) may not be correctly detected.
107
+
108
+ This is a limitation of libmagic beacuse the signature byte sequence at the start of doc/xls/ppt is exactly the same (b'\xD0\xCF\x11\xE0\xA1\xB1\x1A\xE1'),
109
+ not of pyxtxt.
110
+
111
+ If available, using the original filename is highly recommended.
112
+
113
+ To extract text from documents in MSWrite's old .doc format, it is necessary to install antiword.
114
+
115
+ ```bash
116
+ sudo apt-get update
117
+ sudo apt-get -y install antiword
118
+ ```
119
+
120
+ ## 🔒 License
121
+ Distributed under the MIT License.
122
+
123
+ The software is provided "as is" without any warranty of any kind.
124
+
125
+ Pull requests, issues, and feedback are warmly welcome! 🚀
@@ -0,0 +1,58 @@
1
+ [project]
2
+ name = "pyxtxt"
3
+ version = "0.0.1"
4
+ description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
+ readme = "README.md"
6
+ requires-python = ">=3.7"
7
+ license = { file = "LICENCSE" }
8
+ authors = [
9
+ { name = "Giuseppe Levi", email = "giuseppe.levi@gmail.com" }
10
+ ]
11
+ dependencies = [
12
+ "python-magic; sys_platform != 'win32'",
13
+ "python-magic-bin; sys_platform == 'win32'"
14
+ ]
15
+ [project.optional-dependencies]
16
+ pdf = [
17
+ "PyMuPDF",
18
+ ]
19
+ docx = [
20
+ "python-docx",
21
+ ]
22
+ presentation = [
23
+ "python-pptx",
24
+ ]
25
+ spreadsheet = [
26
+ "openpyxl",
27
+ "xlrd",
28
+ ]
29
+ odf = [
30
+ "odfpy",
31
+ ]
32
+ html = [
33
+ "beautifulsoup4",
34
+ "lxml",
35
+ ]
36
+ doc = [
37
+ "textract",
38
+ ]
39
+ all = [
40
+ "textract",
41
+ "PyMuPDF",
42
+ "python-docx",
43
+ "python-pptx",
44
+ "openpyxl",
45
+ "xlrd",
46
+ "odfpy",
47
+ "beautifulsoup4",
48
+ "lxml",
49
+ ]
50
+
51
+ [build-system]
52
+ requires = ["setuptools>=61.0"]
53
+ build-backend = "setuptools.build_meta"
54
+
55
+
56
+ [tool.setuptools.packages.find]
57
+ where = ["src"]
58
+
pyxtxt-0.0.1/setup.cfg ADDED
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1 @@
1
+ from .core import xtxt, extxt_available_formats
@@ -0,0 +1,71 @@
1
+ from .estrattori import estrattori
2
+ from functools import singledispatch
3
+ import io
4
+ import magic
5
+
6
+ @singledispatch
7
+ def xtxt(file_input):
8
+ raise NotImplementedError(f"Type not supported : {type(file_input)}")
9
+
10
+ # Caso 1: file path (str)
11
+ @xtxt.register
12
+ def _(file_input: str):
13
+ try:
14
+ with open(file_input, "rb") as f:
15
+ data = f.read()
16
+ buffer = io.BytesIO(data)
17
+ buffer.name=file_input
18
+ buffer.mimeType=magic.Magic(mime=True).from_file(file_input)
19
+ return xtxt(buffer)
20
+ except Exception as e:
21
+ print(f"⚠️ File opening error'{file_input}': {e}")
22
+ return None
23
+
24
+ # Caso 2: buffer (BytesIO)
25
+ @xtxt.register
26
+ def _(file_input: io.BytesIO):
27
+ try:
28
+
29
+
30
+ # Mappa dei MIME Type gestiti dagli estrattori
31
+ # estrattori = {
32
+ # "application/pdf": xtxt_pdf,
33
+ # "application/vnd.openxmlformats-officedocument.wordprocessingml.document": xtxt_docx,
34
+ # "application/vnd.openxmlformats-officedocument.presentationml.presentation": xtxt_pptx,
35
+ # "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": xtxt_xlsx,
36
+ # "application/vnd.ms-excel": xtxt_xls,
37
+ # "text/plain": xtxt_txt,
38
+ # "application/vnd.oasis.opendocument.text": xtxt_odt,
39
+ # "text/html": xtxt_html,
40
+ # # "text/rtf": xtxt_rtf,
41
+ # "application/xml": xtxt_xml,
42
+ # "text/xml": xtxt_xml,
43
+ # }
44
+ if hasattr(file_input,'mimeType'):
45
+ mime_type=file_input.mimeType
46
+ else:
47
+ mime_type = magic.Magic(mime=True).from_buffer(file_input.read(2048))
48
+ file_input.name='IO_buffer'
49
+ file_input.seek(0)
50
+ print(mime_type)
51
+ if mime_type.startswith("text/"):
52
+ if (mime_type != "text/html") and (mime_type != "text/xml") and (mime_type != "text/plain"):
53
+ print(f"📄 File recognized as text type: {mime_type}, treated as text/plain")
54
+ mime_type = "text/plain"
55
+ if mime_type not in estrattori:
56
+ print(f"⚠️ MIME type not supported {mime_type} ({file_input.name}) ignored.")
57
+ return None
58
+
59
+
60
+ # Estrai il testo
61
+ testo = estrattori[mime_type](file_input)
62
+ return f"{testo}"
63
+ except Exception as e:
64
+ print(f"❌ Error while reading: {e}")
65
+ return None
66
+ def extxt_available_formats(pretty=False):
67
+ if pretty:
68
+ from .estrattori import pretty_names
69
+ return sorted({pretty_names.get(mime, mime) for mime in estrattori.keys()})
70
+ else:
71
+ return sorted(estrattori.keys())
@@ -0,0 +1,18 @@
1
+ import os
2
+ import importlib
3
+
4
+ # Qui vengono registrati gli estrattori disponibili
5
+ estrattori = {}
6
+ pretty_names = {}
7
+
8
+ def register_extractor(mime_type, func, name=None):
9
+ estrattori[mime_type] = func
10
+ if name:
11
+ pretty_names[mime_type] = name
12
+
13
+ # Carica automaticamente tutti i moduli presenti
14
+ current_dir = os.path.dirname(__file__)
15
+ for filename in os.listdir(current_dir):
16
+ if filename.endswith(".py") and filename != "__init__.py":
17
+ module_name = f"{__name__}.{filename[:-3]}"
18
+ importlib.import_module(module_name)
@@ -0,0 +1,26 @@
1
+
2
+ from . import register_extractor
3
+ import shutil
4
+ try:
5
+ import textract
6
+ except ImportError:
7
+ Lib = None
8
+ if Lib:
9
+ def xtxt_doc(file_buffer):
10
+
11
+ if shutil.which("antiword") is None:
12
+ print("⚠️ 'antiword' is not installed or is not in the system PATH.")
13
+ return None
14
+ try:
15
+ file_buffer.seek(0)
16
+ data = file_buffer.read()
17
+ testo = textract.process("temp.doc", input_data=data)
18
+ return testo.decode("utf-8").strip()
19
+ except Exception as e:
20
+ print(f"⚠️ Error during extraction from DOC: {e}")
21
+ return None
22
+ register_extractor(
23
+ "application/msword",
24
+ xtxt__doc,
25
+ name="DOC"
26
+ )
@@ -0,0 +1,35 @@
1
+ from . import register_extractor
2
+ import io
3
+ import zipfile
4
+ try:
5
+ from docx import Document
6
+ except ImportError:
7
+ Lib = None
8
+
9
+ if Lib:
10
+ def xtxt_docx(file_buffer) -> str:
11
+ try:
12
+ # Copia del buffer per poterlo riutilizzare
13
+ file_buffer.seek(0)
14
+ data = file_buffer.read()
15
+ buffer_copy = io.BytesIO(data)
16
+
17
+ if not zipfile.is_zipfile(buffer_copy):
18
+ print("⚠️ Invalid DOCX (not a ZIP file)")
19
+ return ""
20
+
21
+ buffer_copy.seek(0)
22
+ doc = Document(buffer_copy)
23
+
24
+ text = "\n".join(p.text for p in doc.paragraphs)
25
+ return text
26
+
27
+ except Exception as e:
28
+ print(f"⚠️ Error during extraction DOCX: {e}")
29
+ return ""
30
+
31
+ register_extractor(
32
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
33
+ xtxt_docx,
34
+ name="DOCX"
35
+ )
@@ -0,0 +1,15 @@
1
+ from . import register_extractor
2
+ try:
3
+ from bs4 import BeautifulSoup
4
+ except ImportError:
5
+ BeautifulSoup = None
6
+
7
+ if BeautifulSoup:
8
+ def xtxt_html(file_buffer):
9
+ soup = BeautifulSoup(file_buffer.read(), "html.parser")
10
+ return soup.get_text(separator="\n")
11
+ register_extractor(
12
+ "text/html",
13
+ xtxt_html,
14
+ name="HTML"
15
+ )
@@ -0,0 +1,27 @@
1
+ from . import register_extractor
2
+ try:
3
+ from odf.opendocument import load
4
+ from odf.text import P
5
+ except ImportError:
6
+ Lib = None
7
+
8
+ if Lib:
9
+ def xtxt_odt(file_buffer):
10
+ odt_doc = load(file_buffer)
11
+ paragraphs = odt_doc.getElementsByType(P)
12
+
13
+ testo = []
14
+ for p in paragraphs:
15
+ contenuto = []
16
+ for n in p.childNodes:
17
+ if n.nodeType == 3: # TEXT_NODE
18
+ contenuto.append(n.data)
19
+ if contenuto:
20
+ testo.append("".join(contenuto))
21
+
22
+ return "\n".join(testo)
23
+ register_extractor(
24
+ "application/vnd.oasis.opendocument.text",
25
+ xtxt_odt,
26
+ name="ODT"
27
+ )
@@ -0,0 +1,32 @@
1
+ from . import register_extractor
2
+ try:
3
+ import fitz # PyMuPDF
4
+ except ImportError:
5
+ Lib = None
6
+
7
+ if Lib:
8
+ def xtxt_pdf(file_buffer):
9
+ try:
10
+ raw_data = file_buffer.read()
11
+ if not raw_data:
12
+ print("⚠️ PDF blank or not read correctly")
13
+ return None
14
+
15
+ doc = fitz.open(stream=raw_data, filetype="pdf")
16
+ return "\n".join(page.get_text() for page in doc)
17
+
18
+ except fitz.EmptyFileError:
19
+ print("⚠️ Error: PDF is blank or unreadable")
20
+ return None
21
+ except Exception as e:
22
+ print(f"⚠️ Error during PDF extraction: {e}")
23
+ return None
24
+
25
+
26
+
27
+
28
+ register_extractor(
29
+ "application/pdf",
30
+ xtxt_pdf,
31
+ name="PDF"
32
+ )
@@ -0,0 +1,40 @@
1
+ from . import register_extractor
2
+ import io
3
+ import zipfile
4
+ try:
5
+ from pptx import Presentation
6
+ except ImportError:
7
+ Lib = None
8
+
9
+ if Lib:
10
+ def xtxt_pptx(file_buffer) -> str:
11
+ try:
12
+ # Convertiamo il file_buffer (che è già un BytesIO o simile) in modo da poterlo riusare
13
+ file_buffer.seek(0)
14
+ data = file_buffer.read()
15
+ buffer_copy = io.BytesIO(data)
16
+
17
+ if not zipfile.is_zipfile(buffer_copy):
18
+ print("⚠️ Invalid PPTX (not a ZIP file)" )
19
+ return ""
20
+
21
+ # Se è un file zip valido, possiamo ripassare i dati a Presentation
22
+ buffer_copy.seek(0)
23
+ prs = Presentation(buffer_copy)
24
+
25
+ text = "\n".join(
26
+ shape.text
27
+ for slide in prs.slides
28
+ for shape in slide.shapes
29
+ if hasattr(shape, "text")
30
+ )
31
+ return text
32
+
33
+ except Exception as e:
34
+ print(f"⚠️ Error during PPTX extraction: {e}")
35
+ return ""
36
+ register_extractor(
37
+ "application/vnd.openxmlformats-officedocument.presentationml.presentation" ,
38
+ xtxt_pptx,
39
+ name="PPTX"
40
+ )
@@ -0,0 +1,23 @@
1
+ from . import register_extractor
2
+ try:
3
+ from lxml import etree
4
+ except ImportError:
5
+ Lib = None
6
+
7
+ if Lib:
8
+ def xtxt_svg(file_buffer):
9
+ try:
10
+ tree = etree.parse(file_buffer)
11
+ root = tree.getroot()
12
+
13
+ # Estrai tutto il testo dai tag <text>
14
+ texts = [element.text for element in root.findall('.//{http://www.w3.org/2000/svg}text')]
15
+ return "\n".join(texts)
16
+ except Exception as e:
17
+ print(f"⚠️ Error while extracting SVG: {e}")
18
+ return ""
19
+ register_extractor(
20
+ "image/svg+xml",
21
+ xtxt_svg,
22
+ name="SVG"
23
+ )