pyxtxt 0.1.8__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyxtxt/__init__.py +1 -0
- pyxtxt/core.py +71 -0
- pyxtxt/estrattori/__init__.py +18 -0
- pyxtxt/estrattori/docx.py +31 -0
- pyxtxt/estrattori/html.py +10 -0
- pyxtxt/estrattori/odt.py +22 -0
- pyxtxt/estrattori/pdf.py +27 -0
- pyxtxt/estrattori/pptx.py +35 -0
- pyxtxt/estrattori/svg.py +19 -0
- pyxtxt/estrattori/txt.py +8 -0
- pyxtxt/estrattori/xls.py +29 -0
- pyxtxt/estrattori/xlsx.py +45 -0
- pyxtxt/estrattori/xml.py +37 -0
- pyxtxt/pyxtxt.py +280 -0
- pyxtxt-0.1.8.dist-info/METADATA +154 -0
- pyxtxt-0.1.8.dist-info/RECORD +18 -0
- pyxtxt-0.1.8.dist-info/WHEEL +5 -0
- pyxtxt-0.1.8.dist-info/top_level.txt +1 -0
pyxtxt/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
from .core import xtxt, extxt_available_formats
|
pyxtxt/core.py
ADDED
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
from .estrattori import estrattori
|
|
2
|
+
from functools import singledispatch
|
|
3
|
+
import io
|
|
4
|
+
import magic
|
|
5
|
+
|
|
6
|
+
@singledispatch
|
|
7
|
+
def xtxt(file_input):
|
|
8
|
+
raise NotImplementedError(f"Type not supported : {type(file_input)}")
|
|
9
|
+
|
|
10
|
+
# Caso 1: file path (str)
|
|
11
|
+
@xtxt.register
|
|
12
|
+
def _(file_input: str):
|
|
13
|
+
try:
|
|
14
|
+
with open(file_input, "rb") as f:
|
|
15
|
+
data = f.read()
|
|
16
|
+
buffer = io.BytesIO(data)
|
|
17
|
+
buffer.name=file_input
|
|
18
|
+
buffer.mimeType=magic.Magic(mime=True).from_file(file_input)
|
|
19
|
+
return xtxt(buffer)
|
|
20
|
+
except Exception as e:
|
|
21
|
+
print(f"⚠️ File opening error'{file_input}': {e}")
|
|
22
|
+
return None
|
|
23
|
+
|
|
24
|
+
# Caso 2: buffer (BytesIO)
|
|
25
|
+
@xtxt.register
|
|
26
|
+
def _(file_input: io.BytesIO):
|
|
27
|
+
try:
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# Mappa dei MIME Type gestiti dagli estrattori
|
|
31
|
+
# estrattori = {
|
|
32
|
+
# "application/pdf": xtxt_pdf,
|
|
33
|
+
# "application/vnd.openxmlformats-officedocument.wordprocessingml.document": xtxt_docx,
|
|
34
|
+
# "application/vnd.openxmlformats-officedocument.presentationml.presentation": xtxt_pptx,
|
|
35
|
+
# "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": xtxt_xlsx,
|
|
36
|
+
# "application/vnd.ms-excel": xtxt_xls,
|
|
37
|
+
# "text/plain": xtxt_txt,
|
|
38
|
+
# "application/vnd.oasis.opendocument.text": xtxt_odt,
|
|
39
|
+
# "text/html": xtxt_html,
|
|
40
|
+
# # "text/rtf": xtxt_rtf,
|
|
41
|
+
# "application/xml": xtxt_xml,
|
|
42
|
+
# "text/xml": xtxt_xml,
|
|
43
|
+
# }
|
|
44
|
+
if hasattr(file_input,'mimeType'):
|
|
45
|
+
mime_type=file_input.mimeType
|
|
46
|
+
else:
|
|
47
|
+
mime_type = magic.Magic(mime=True).from_buffer(file_input.read(2048))
|
|
48
|
+
file_input.name='IO_buffer'
|
|
49
|
+
file_input.seek(0)
|
|
50
|
+
print(mime_type)
|
|
51
|
+
if mime_type.startswith("text/"):
|
|
52
|
+
if (mime_type != "text/html") and (mime_type != "text/xml") and (mime_type != "text/plain"):
|
|
53
|
+
print(f"📄 File recognized as text type: {mime_type}, treated as text/plain")
|
|
54
|
+
mime_type = "text/plain"
|
|
55
|
+
if mime_type not in estrattori:
|
|
56
|
+
print(f"⚠️ MIME type not supported {mime_type} ({file_input.name}) ignored.")
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
# Estrai il testo
|
|
61
|
+
testo = estrattori[mime_type](file_input)
|
|
62
|
+
return f"{testo}"
|
|
63
|
+
except Exception as e:
|
|
64
|
+
print(f"❌ Error while reading: {e}")
|
|
65
|
+
return None
|
|
66
|
+
def extxt_available_formats(pretty=False):
|
|
67
|
+
if pretty:
|
|
68
|
+
from .estrattori import pretty_names
|
|
69
|
+
return sorted({pretty_names.get(mime, mime) for mime in estrattori.keys()})
|
|
70
|
+
else:
|
|
71
|
+
return sorted(estrattori.keys())
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import os
|
|
2
|
+
import importlib
|
|
3
|
+
|
|
4
|
+
# Qui vengono registrati gli estrattori disponibili
|
|
5
|
+
estrattori = {}
|
|
6
|
+
pretty_names = {}
|
|
7
|
+
|
|
8
|
+
def register_extractor(mime_type, func, name=None):
|
|
9
|
+
estrattori[mime_type] = func
|
|
10
|
+
if name:
|
|
11
|
+
pretty_names[mime_type] = name
|
|
12
|
+
|
|
13
|
+
# Carica automaticamente tutti i moduli presenti
|
|
14
|
+
current_dir = os.path.dirname(__file__)
|
|
15
|
+
for filename in os.listdir(current_dir):
|
|
16
|
+
if filename.endswith(".py") and filename != "__init__.py":
|
|
17
|
+
module_name = f"{__name__}.{filename[:-3]}"
|
|
18
|
+
importlib.import_module(module_name)
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
import io
|
|
3
|
+
import zipfile
|
|
4
|
+
from docx import Document
|
|
5
|
+
|
|
6
|
+
def xtxt_docx(file_buffer) -> str:
|
|
7
|
+
try:
|
|
8
|
+
# Copia del buffer per poterlo riutilizzare
|
|
9
|
+
file_buffer.seek(0)
|
|
10
|
+
data = file_buffer.read()
|
|
11
|
+
buffer_copy = io.BytesIO(data)
|
|
12
|
+
|
|
13
|
+
if not zipfile.is_zipfile(buffer_copy):
|
|
14
|
+
print("⚠️ Invalid DOCX (not a ZIP file)")
|
|
15
|
+
return ""
|
|
16
|
+
|
|
17
|
+
buffer_copy.seek(0)
|
|
18
|
+
doc = Document(buffer_copy)
|
|
19
|
+
|
|
20
|
+
text = "\n".join(p.text for p in doc.paragraphs)
|
|
21
|
+
return text
|
|
22
|
+
|
|
23
|
+
except Exception as e:
|
|
24
|
+
print(f"⚠️ Error during extraction DOCX: {e}")
|
|
25
|
+
return ""
|
|
26
|
+
|
|
27
|
+
register_extractor(
|
|
28
|
+
"application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
29
|
+
xtxt_docx,
|
|
30
|
+
name="DOCX"
|
|
31
|
+
)
|
pyxtxt/estrattori/odt.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
from odf.opendocument import load
|
|
3
|
+
from odf.text import P
|
|
4
|
+
def xtxt_odt(file_buffer):
|
|
5
|
+
odt_doc = load(file_buffer)
|
|
6
|
+
paragraphs = odt_doc.getElementsByType(P)
|
|
7
|
+
|
|
8
|
+
testo = []
|
|
9
|
+
for p in paragraphs:
|
|
10
|
+
contenuto = []
|
|
11
|
+
for n in p.childNodes:
|
|
12
|
+
if n.nodeType == 3: # TEXT_NODE
|
|
13
|
+
contenuto.append(n.data)
|
|
14
|
+
if contenuto:
|
|
15
|
+
testo.append("".join(contenuto))
|
|
16
|
+
|
|
17
|
+
return "\n".join(testo)
|
|
18
|
+
register_extractor(
|
|
19
|
+
"application/vnd.oasis.opendocument.text",
|
|
20
|
+
xtxt_odt,
|
|
21
|
+
name="ODT"
|
|
22
|
+
)
|
pyxtxt/estrattori/pdf.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
import fitz # PyMuPDF
|
|
3
|
+
def xtxt_pdf(file_buffer):
|
|
4
|
+
try:
|
|
5
|
+
raw_data = file_buffer.read()
|
|
6
|
+
if not raw_data:
|
|
7
|
+
print("⚠️ PDF blank or not read correctly")
|
|
8
|
+
return None
|
|
9
|
+
|
|
10
|
+
doc = fitz.open(stream=raw_data, filetype="pdf")
|
|
11
|
+
return "\n".join(page.get_text() for page in doc)
|
|
12
|
+
|
|
13
|
+
except fitz.EmptyFileError:
|
|
14
|
+
print("⚠️ Error: PDF is blank or unreadable")
|
|
15
|
+
return None
|
|
16
|
+
except Exception as e:
|
|
17
|
+
print(f"⚠️ Error during PDF extraction: {e}")
|
|
18
|
+
return None
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
register_extractor(
|
|
24
|
+
"application/pdf",
|
|
25
|
+
xtxt_pdf,
|
|
26
|
+
name="PDF"
|
|
27
|
+
)
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
import io
|
|
3
|
+
import zipfile
|
|
4
|
+
from pptx import Presentation
|
|
5
|
+
def xtxt_pptx(file_buffer) -> str:
|
|
6
|
+
try:
|
|
7
|
+
# Convertiamo il file_buffer (che è già un BytesIO o simile) in modo da poterlo riusare
|
|
8
|
+
file_buffer.seek(0)
|
|
9
|
+
data = file_buffer.read()
|
|
10
|
+
buffer_copy = io.BytesIO(data)
|
|
11
|
+
|
|
12
|
+
if not zipfile.is_zipfile(buffer_copy):
|
|
13
|
+
print("⚠️ Invalid PPTX (not a ZIP file)" )
|
|
14
|
+
return ""
|
|
15
|
+
|
|
16
|
+
# Se è un file zip valido, possiamo ripassare i dati a Presentation
|
|
17
|
+
buffer_copy.seek(0)
|
|
18
|
+
prs = Presentation(buffer_copy)
|
|
19
|
+
|
|
20
|
+
text = "\n".join(
|
|
21
|
+
shape.text
|
|
22
|
+
for slide in prs.slides
|
|
23
|
+
for shape in slide.shapes
|
|
24
|
+
if hasattr(shape, "text")
|
|
25
|
+
)
|
|
26
|
+
return text
|
|
27
|
+
|
|
28
|
+
except Exception as e:
|
|
29
|
+
print(f"⚠️ Error during PPTX extraction: {e}")
|
|
30
|
+
return ""
|
|
31
|
+
register_extractor(
|
|
32
|
+
"application/vnd.openxmlformats-officedocument.presentationml.presentation" ,
|
|
33
|
+
xtxt_pptx,
|
|
34
|
+
name="PPTX"
|
|
35
|
+
)
|
pyxtxt/estrattori/svg.py
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
from lxml import etree
|
|
3
|
+
|
|
4
|
+
def xtxt_svg(file_buffer):
|
|
5
|
+
try:
|
|
6
|
+
tree = etree.parse(file_buffer)
|
|
7
|
+
root = tree.getroot()
|
|
8
|
+
|
|
9
|
+
# Estrai tutto il testo dai tag <text>
|
|
10
|
+
texts = [element.text for element in root.findall('.//{http://www.w3.org/2000/svg}text')]
|
|
11
|
+
return "\n".join(texts)
|
|
12
|
+
except Exception as e:
|
|
13
|
+
print(f"⚠️ Error while extracting SVG: {e}")
|
|
14
|
+
return ""
|
|
15
|
+
register_extractor(
|
|
16
|
+
"image/svg+xml",
|
|
17
|
+
xtxt_svg,
|
|
18
|
+
name="SVG"
|
|
19
|
+
)
|
pyxtxt/estrattori/txt.py
ADDED
pyxtxt/estrattori/xls.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
import io
|
|
2
|
+
from . import register_extractor
|
|
3
|
+
import xlrd
|
|
4
|
+
def xtxt_xls(file_buffer, max_rows_per_sheet: int = 100) -> str:
|
|
5
|
+
try:
|
|
6
|
+
file_buffer.seek(0)
|
|
7
|
+
workbook = xlrd.open_workbook(file_contents=file_buffer.read())
|
|
8
|
+
testo = []
|
|
9
|
+
|
|
10
|
+
for sheet in workbook.sheets():
|
|
11
|
+
testo.append(f"# {sheet.name}")
|
|
12
|
+
for row_idx in range(min(sheet.nrows, max_rows_per_sheet)):
|
|
13
|
+
row = sheet.row(row_idx)
|
|
14
|
+
valori = [str(cell.value).strip() for cell in row if str(cell.value).strip()]
|
|
15
|
+
if valori:
|
|
16
|
+
testo.append(" | ".join(valori))
|
|
17
|
+
|
|
18
|
+
return "\n".join(testo)
|
|
19
|
+
|
|
20
|
+
except Exception as e:
|
|
21
|
+
print(f"⚠️ Error while extracting XLS: {e}")
|
|
22
|
+
return ""
|
|
23
|
+
|
|
24
|
+
register_extractor(
|
|
25
|
+
"application/vnd.ms-excel",
|
|
26
|
+
xtxt_xls,
|
|
27
|
+
name="XLS"
|
|
28
|
+
)
|
|
29
|
+
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
import io
|
|
2
|
+
import zipfile
|
|
3
|
+
from openpyxl import load_workbook
|
|
4
|
+
from openpyxl.worksheet.worksheet import Worksheet
|
|
5
|
+
from . import register_extractor
|
|
6
|
+
import openpyxl
|
|
7
|
+
|
|
8
|
+
def xtxt_xlsx(file_buffer, max_rows_per_sheet: int = 200) -> str:
|
|
9
|
+
try:
|
|
10
|
+
file_buffer.seek(0)
|
|
11
|
+
data = file_buffer.read()
|
|
12
|
+
buffer_copy = io.BytesIO(data)
|
|
13
|
+
|
|
14
|
+
if not zipfile.is_zipfile(buffer_copy):
|
|
15
|
+
print("⚠️ Invalid XLSX (not a ZIP archive)")
|
|
16
|
+
return ""
|
|
17
|
+
|
|
18
|
+
buffer_copy.seek(0)
|
|
19
|
+
wb = openpyxl.load_workbook(buffer_copy, data_only=True, read_only=True)
|
|
20
|
+
except Exception as e:
|
|
21
|
+
print(f"⚠️ Error while reading XLSX : {e}")
|
|
22
|
+
return ""
|
|
23
|
+
|
|
24
|
+
testo = []
|
|
25
|
+
for sheet in wb.worksheets:
|
|
26
|
+
if sheet.sheet_state != 'visible':
|
|
27
|
+
continue
|
|
28
|
+
testo.append(f"# {sheet.title}")
|
|
29
|
+
count = 0
|
|
30
|
+
for row in sheet.iter_rows(values_only=True):
|
|
31
|
+
if max_rows_per_sheet != -1 and count >= max_rows_per_sheet:
|
|
32
|
+
break
|
|
33
|
+
valori = [str(cell).strip() if cell is not None else "" for cell in row]
|
|
34
|
+
if any(valori):
|
|
35
|
+
testo.append(" | ".join(valori))
|
|
36
|
+
count += 1
|
|
37
|
+
|
|
38
|
+
return "\n".join(testo)
|
|
39
|
+
register_extractor(
|
|
40
|
+
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
|
41
|
+
xtxt_xlsx,
|
|
42
|
+
name="XLSX"
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
pyxtxt/estrattori/xml.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
from . import register_extractor
|
|
2
|
+
from lxml import etree
|
|
3
|
+
|
|
4
|
+
def xtxt_xml(file_buffer) -> str:
|
|
5
|
+
try:
|
|
6
|
+
file_buffer.seek(0)
|
|
7
|
+
parser = etree.XMLParser(recover=True)
|
|
8
|
+
tree = etree.parse(file_buffer, parser)
|
|
9
|
+
root = tree.getroot()
|
|
10
|
+
|
|
11
|
+
# Estrai il testo ricorsivamente da tutti i nodi
|
|
12
|
+
def get_text_recursively(elem):
|
|
13
|
+
texts = []
|
|
14
|
+
if elem.text:
|
|
15
|
+
texts.append(elem.text.strip())
|
|
16
|
+
for child in elem:
|
|
17
|
+
texts.append(get_text_recursively(child))
|
|
18
|
+
if child.tail:
|
|
19
|
+
texts.append(child.tail.strip())
|
|
20
|
+
return " ".join(filter(None, texts))
|
|
21
|
+
|
|
22
|
+
testo = get_text_recursively(root)
|
|
23
|
+
return testo.strip()
|
|
24
|
+
|
|
25
|
+
except Exception as e:
|
|
26
|
+
print(f"⚠️ Error while extracting XML : {e}")
|
|
27
|
+
return ""
|
|
28
|
+
register_extractor(
|
|
29
|
+
"application/xml",
|
|
30
|
+
xtxt_xml,
|
|
31
|
+
name="XML"
|
|
32
|
+
)
|
|
33
|
+
register_extractor(
|
|
34
|
+
"text/xml",
|
|
35
|
+
xtxt_xml,
|
|
36
|
+
name="XML"
|
|
37
|
+
)
|
pyxtxt/pyxtxt.py
ADDED
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
import fitz # PyMuPDF
|
|
2
|
+
from bs4 import BeautifulSoup
|
|
3
|
+
from docx import Document
|
|
4
|
+
from pptx import Presentation
|
|
5
|
+
from odf.opendocument import load
|
|
6
|
+
from odf.text import P
|
|
7
|
+
import openpyxl
|
|
8
|
+
|
|
9
|
+
def xtxt_pdf(file_buffer):
|
|
10
|
+
try:
|
|
11
|
+
raw_data = file_buffer.read()
|
|
12
|
+
if not raw_data:
|
|
13
|
+
print("⚠️ PDF blank or not read correctly")
|
|
14
|
+
return None
|
|
15
|
+
|
|
16
|
+
doc = fitz.open(stream=raw_data, filetype="pdf")
|
|
17
|
+
return "\n".join(page.get_text() for page in doc)
|
|
18
|
+
|
|
19
|
+
except fitz.EmptyFileError:
|
|
20
|
+
print("⚠️ Error: PDF is blank or unreadable")
|
|
21
|
+
return None
|
|
22
|
+
except Exception as e:
|
|
23
|
+
print(f"⚠️ Error during PDF extraction: {e}")
|
|
24
|
+
return None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def xtxt_docx(file_buffer) -> str:
|
|
29
|
+
try:
|
|
30
|
+
# Copia del buffer per poterlo riutilizzare
|
|
31
|
+
file_buffer.seek(0)
|
|
32
|
+
data = file_buffer.read()
|
|
33
|
+
buffer_copy = io.BytesIO(data)
|
|
34
|
+
|
|
35
|
+
if not zipfile.is_zipfile(buffer_copy):
|
|
36
|
+
print("⚠️ Invalid DOCX (not a ZIP file)")
|
|
37
|
+
return ""
|
|
38
|
+
|
|
39
|
+
buffer_copy.seek(0)
|
|
40
|
+
doc = Document(buffer_copy)
|
|
41
|
+
|
|
42
|
+
text = "\n".join(p.text for p in doc.paragraphs)
|
|
43
|
+
return text
|
|
44
|
+
|
|
45
|
+
except Exception as e:
|
|
46
|
+
print(f"⚠️ Error during extraction DOCX: {e}")
|
|
47
|
+
return ""
|
|
48
|
+
|
|
49
|
+
import zipfile
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def xtxt_pptx(file_buffer) -> str:
|
|
53
|
+
try:
|
|
54
|
+
# Convertiamo il file_buffer (che è già un BytesIO o simile) in modo da poterlo riusare
|
|
55
|
+
file_buffer.seek(0)
|
|
56
|
+
data = file_buffer.read()
|
|
57
|
+
buffer_copy = io.BytesIO(data)
|
|
58
|
+
|
|
59
|
+
if not zipfile.is_zipfile(buffer_copy):
|
|
60
|
+
print("⚠️ Invalid PPTX (not a ZIP file)" )
|
|
61
|
+
return ""
|
|
62
|
+
|
|
63
|
+
# Se è un file zip valido, possiamo ripassare i dati a Presentation
|
|
64
|
+
buffer_copy.seek(0)
|
|
65
|
+
prs = Presentation(buffer_copy)
|
|
66
|
+
|
|
67
|
+
text = "\n".join(
|
|
68
|
+
shape.text
|
|
69
|
+
for slide in prs.slides
|
|
70
|
+
for shape in slide.shapes
|
|
71
|
+
if hasattr(shape, "text")
|
|
72
|
+
)
|
|
73
|
+
return text
|
|
74
|
+
|
|
75
|
+
except Exception as e:
|
|
76
|
+
print(f"⚠️ Error during PPTX extraction: {e}")
|
|
77
|
+
return ""
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
from openpyxl import load_workbook
|
|
81
|
+
from openpyxl.worksheet.worksheet import Worksheet
|
|
82
|
+
import openpyxl
|
|
83
|
+
|
|
84
|
+
def xtxt_xlsx(file_buffer, max_rows_per_sheet: int = 200) -> str:
|
|
85
|
+
try:
|
|
86
|
+
file_buffer.seek(0)
|
|
87
|
+
data = file_buffer.read()
|
|
88
|
+
buffer_copy = io.BytesIO(data)
|
|
89
|
+
|
|
90
|
+
if not zipfile.is_zipfile(buffer_copy):
|
|
91
|
+
print("⚠️ Invalid XLSX (not a ZIP archive)")
|
|
92
|
+
return ""
|
|
93
|
+
|
|
94
|
+
buffer_copy.seek(0)
|
|
95
|
+
wb = openpyxl.load_workbook(buffer_copy, data_only=True, read_only=True)
|
|
96
|
+
except Exception as e:
|
|
97
|
+
print(f"⚠️ Error while reading XLSX : {e}")
|
|
98
|
+
return ""
|
|
99
|
+
|
|
100
|
+
testo = []
|
|
101
|
+
for sheet in wb.worksheets:
|
|
102
|
+
if sheet.sheet_state != 'visible':
|
|
103
|
+
continue
|
|
104
|
+
testo.append(f"# {sheet.title}")
|
|
105
|
+
count = 0
|
|
106
|
+
for row in sheet.iter_rows(values_only=True):
|
|
107
|
+
if max_rows_per_sheet != -1 and count >= max_rows_per_sheet:
|
|
108
|
+
break
|
|
109
|
+
valori = [str(cell).strip() if cell is not None else "" for cell in row]
|
|
110
|
+
if any(valori):
|
|
111
|
+
testo.append(" | ".join(valori))
|
|
112
|
+
count += 1
|
|
113
|
+
|
|
114
|
+
return "\n".join(testo)
|
|
115
|
+
|
|
116
|
+
def xtxt_txt(file_buffer):
|
|
117
|
+
return file_buffer.read().decode("utf-8", errors="ignore")
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def xtxt_odt(file_buffer):
|
|
121
|
+
odt_doc = load(file_buffer)
|
|
122
|
+
paragraphs = odt_doc.getElementsByType(P)
|
|
123
|
+
|
|
124
|
+
testo = []
|
|
125
|
+
for p in paragraphs:
|
|
126
|
+
contenuto = []
|
|
127
|
+
for n in p.childNodes:
|
|
128
|
+
if n.nodeType == 3: # TEXT_NODE
|
|
129
|
+
contenuto.append(n.data)
|
|
130
|
+
if contenuto:
|
|
131
|
+
testo.append("".join(contenuto))
|
|
132
|
+
|
|
133
|
+
return "\n".join(testo)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def xtxt_html(file_buffer):
|
|
137
|
+
soup = BeautifulSoup(file_buffer.read(), "html.parser")
|
|
138
|
+
return soup.get_text(separator="\n")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
from lxml import etree
|
|
145
|
+
|
|
146
|
+
def xtxt_xml(file_buffer) -> str:
|
|
147
|
+
try:
|
|
148
|
+
file_buffer.seek(0)
|
|
149
|
+
parser = etree.XMLParser(recover=True)
|
|
150
|
+
tree = etree.parse(file_buffer, parser)
|
|
151
|
+
root = tree.getroot()
|
|
152
|
+
|
|
153
|
+
# Estrai il testo ricorsivamente da tutti i nodi
|
|
154
|
+
def get_text_recursively(elem):
|
|
155
|
+
texts = []
|
|
156
|
+
if elem.text:
|
|
157
|
+
texts.append(elem.text.strip())
|
|
158
|
+
for child in elem:
|
|
159
|
+
texts.append(get_text_recursively(child))
|
|
160
|
+
if child.tail:
|
|
161
|
+
texts.append(child.tail.strip())
|
|
162
|
+
return " ".join(filter(None, texts))
|
|
163
|
+
|
|
164
|
+
testo = get_text_recursively(root)
|
|
165
|
+
return testo.strip()
|
|
166
|
+
|
|
167
|
+
except Exception as e:
|
|
168
|
+
print(f"⚠️ Error while extracting XML : {e}")
|
|
169
|
+
return ""
|
|
170
|
+
import xlrd
|
|
171
|
+
|
|
172
|
+
def xtxt_xls(file_buffer, max_rows_per_sheet: int = 100) -> str:
|
|
173
|
+
try:
|
|
174
|
+
file_buffer.seek(0)
|
|
175
|
+
workbook = xlrd.open_workbook(file_contents=file_buffer.read())
|
|
176
|
+
testo = []
|
|
177
|
+
|
|
178
|
+
for sheet in workbook.sheets():
|
|
179
|
+
testo.append(f"# {sheet.name}")
|
|
180
|
+
for row_idx in range(min(sheet.nrows, max_rows_per_sheet)):
|
|
181
|
+
row = sheet.row(row_idx)
|
|
182
|
+
valori = [str(cell.value).strip() for cell in row if str(cell.value).strip()]
|
|
183
|
+
if valori:
|
|
184
|
+
testo.append(" | ".join(valori))
|
|
185
|
+
|
|
186
|
+
return "\n".join(testo)
|
|
187
|
+
|
|
188
|
+
except Exception as e:
|
|
189
|
+
print(f"⚠️ Error while extracting XLS: {e}")
|
|
190
|
+
return ""
|
|
191
|
+
|
|
192
|
+
# from pyth.plugins.plaintext.reader import PlaintextReader
|
|
193
|
+
# from pyth.plugins.plaintext import PlaintextFile
|
|
194
|
+
|
|
195
|
+
# def xtxt_rtf(file_buffer):
|
|
196
|
+
# try:
|
|
197
|
+
# rtf = PlaintextReader.read(file_buffer)
|
|
198
|
+
# return rtf
|
|
199
|
+
# except Exception as e:
|
|
200
|
+
# print(f"⚠️ Errore durante l'estrazione RTF: {e}")
|
|
201
|
+
# return ""
|
|
202
|
+
|
|
203
|
+
from lxml import etree
|
|
204
|
+
|
|
205
|
+
def xtxt_svg(file_buffer):
|
|
206
|
+
try:
|
|
207
|
+
tree = etree.parse(file_buffer)
|
|
208
|
+
root = tree.getroot()
|
|
209
|
+
|
|
210
|
+
# Estrai tutto il testo dai tag <text>
|
|
211
|
+
texts = [element.text for element in root.findall('.//{http://www.w3.org/2000/svg}text')]
|
|
212
|
+
return "\n".join(texts)
|
|
213
|
+
except Exception as e:
|
|
214
|
+
print(f"⚠️ Error while extracting SVG: {e}")
|
|
215
|
+
return ""
|
|
216
|
+
|
|
217
|
+
from functools import singledispatch
|
|
218
|
+
import io
|
|
219
|
+
import magic
|
|
220
|
+
|
|
221
|
+
@singledispatch
|
|
222
|
+
def xtxt(file_input):
|
|
223
|
+
raise NotImplementedError(f"Type not supported : {type(file_input)}")
|
|
224
|
+
|
|
225
|
+
# Caso 1: file path (str)
|
|
226
|
+
@xtxt.register
|
|
227
|
+
def _(file_input: str):
|
|
228
|
+
try:
|
|
229
|
+
with open(file_input, "rb") as f:
|
|
230
|
+
data = f.read()
|
|
231
|
+
buffer = io.BytesIO(data)
|
|
232
|
+
buffer.name=file_input
|
|
233
|
+
buffer.mimeType=magic.Magic(mime=True).from_file(file_input)
|
|
234
|
+
return xtxt(buffer)
|
|
235
|
+
except Exception as e:
|
|
236
|
+
print(f"⚠️ File opening error'{file_input}': {e}")
|
|
237
|
+
return None
|
|
238
|
+
|
|
239
|
+
# Caso 2: buffer (BytesIO)
|
|
240
|
+
@xtxt.register
|
|
241
|
+
def _(file_input: io.BytesIO):
|
|
242
|
+
try:
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
# Mappa dei MIME Type gestiti dagli estrattori
|
|
246
|
+
estrattori = {
|
|
247
|
+
"application/pdf": xtxt_pdf,
|
|
248
|
+
"application/vnd.openxmlformats-officedocument.wordprocessingml.document": xtxt_docx,
|
|
249
|
+
"application/vnd.openxmlformats-officedocument.presentationml.presentation": xtxt_pptx,
|
|
250
|
+
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": xtxt_xlsx,
|
|
251
|
+
"application/vnd.ms-excel": xtxt_xls,
|
|
252
|
+
"text/plain": xtxt_txt,
|
|
253
|
+
"application/vnd.oasis.opendocument.text": xtxt_odt,
|
|
254
|
+
"text/html": xtxt_html,
|
|
255
|
+
# "text/rtf": xtxt_rtf,
|
|
256
|
+
"application/xml": xtxt_xml,
|
|
257
|
+
"text/xml": xtxt_xml,
|
|
258
|
+
}
|
|
259
|
+
if hasattr(file_input,'mimeType'):
|
|
260
|
+
mime_type=file_input.mimeType
|
|
261
|
+
else:
|
|
262
|
+
mime_type = magic.Magic(mime=True).from_buffer(file_input.read(2048))
|
|
263
|
+
file_input.name='IO_buffer'
|
|
264
|
+
file_input.seek(0)
|
|
265
|
+
print(mime_type)
|
|
266
|
+
if mime_type.startswith("text/"):
|
|
267
|
+
if (mime_type != "text/html") and (mime_type != "text/xml") and (mime_type != "text/plain"):
|
|
268
|
+
print(f"📄 File recognized as text type: {mime_type}, treated as text/plain")
|
|
269
|
+
mime_type = "text/plain"
|
|
270
|
+
if mime_type not in estrattori:
|
|
271
|
+
print(f"⚠️ MIME type not supported {mime_type} ({file_input.name}) ignored.")
|
|
272
|
+
return None
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
# Estrai il testo
|
|
276
|
+
testo = estrattori[mime_type](file_input)
|
|
277
|
+
return f"{testo}"
|
|
278
|
+
except Exception as e:
|
|
279
|
+
print(f"❌ Error while reading: {e}")
|
|
280
|
+
return None
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pyxtxt
|
|
3
|
+
Version: 0.1.8
|
|
4
|
+
Summary: Una libreria Python per estrarre testo da diversi tipi di file (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
|
+
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2025 Giuseppe Levi
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
Requires-Python: >=3.7
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
Provides-Extra: pdf
|
|
31
|
+
Requires-Dist: PyMuPDF; extra == "pdf"
|
|
32
|
+
Provides-Extra: docx
|
|
33
|
+
Requires-Dist: python-docx; extra == "docx"
|
|
34
|
+
Provides-Extra: presentation
|
|
35
|
+
Requires-Dist: python-pptx; extra == "presentation"
|
|
36
|
+
Provides-Extra: spreadsheet
|
|
37
|
+
Requires-Dist: openpyxl; extra == "spreadsheet"
|
|
38
|
+
Requires-Dist: xlrd; extra == "spreadsheet"
|
|
39
|
+
Provides-Extra: odf
|
|
40
|
+
Requires-Dist: odfpy; extra == "odf"
|
|
41
|
+
Provides-Extra: html
|
|
42
|
+
Requires-Dist: beautifulsoup4; extra == "html"
|
|
43
|
+
Requires-Dist: lxml; extra == "html"
|
|
44
|
+
Provides-Extra: all
|
|
45
|
+
Requires-Dist: pytesseract; extra == "all"
|
|
46
|
+
Requires-Dist: piexif; extra == "all"
|
|
47
|
+
Requires-Dist: Pillow; extra == "all"
|
|
48
|
+
Requires-Dist: PyMuPDF; extra == "all"
|
|
49
|
+
Requires-Dist: python-docx; extra == "all"
|
|
50
|
+
Requires-Dist: python-pptx; extra == "all"
|
|
51
|
+
Requires-Dist: openpyxl; extra == "all"
|
|
52
|
+
Requires-Dist: xlrd; extra == "all"
|
|
53
|
+
Requires-Dist: odfpy; extra == "all"
|
|
54
|
+
Requires-Dist: beautifulsoup4; extra == "all"
|
|
55
|
+
Requires-Dist: lxml; extra == "all"
|
|
56
|
+
|
|
57
|
+
# PyxTxt
|
|
58
|
+
|
|
59
|
+
[](https://pypi.org/project/pyxtxt/)
|
|
60
|
+
[](https://pypi.org/project/pyxtxt/)
|
|
61
|
+
[](https://opensource.org/licenses/MIT)
|
|
62
|
+
|
|
63
|
+
**PyxTxt** is a simple and powerful Python library to extract text from various file formats.
|
|
64
|
+
It supports PDF, DOCX, XLSX, PPTX, ODT, HTML, XML, TXT, legacy XLS files, and more.
|
|
65
|
+
|
|
66
|
+
---
|
|
67
|
+
|
|
68
|
+
## ✨ Features
|
|
69
|
+
|
|
70
|
+
- Extracts text from both file paths and in-memory buffers (`io.BytesIO`).
|
|
71
|
+
- Supports multiple formats: PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT, legacy Office files (.xls, .doc, .ppt).
|
|
72
|
+
- Automatically detects MIME type using `python-magic`.
|
|
73
|
+
- Compatible with modern and legacy formats.
|
|
74
|
+
- Can handle streamed content without saving to disk (with some limitations).
|
|
75
|
+
|
|
76
|
+
---
|
|
77
|
+
|
|
78
|
+
## 📦 Installation
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
pip install pyxtxt
|
|
82
|
+
```
|
|
83
|
+
## ⚠️ Note: You must have libmagic installed on your system (required by python-magic).
|
|
84
|
+
|
|
85
|
+
**On Ubuntu/Debian:**
|
|
86
|
+
|
|
87
|
+
```bash
|
|
88
|
+
sudo apt install libmagic1
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
**On Mac (Homebrew):**
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
brew install libmagic
|
|
95
|
+
```
|
|
96
|
+
**On Windows:**
|
|
97
|
+
|
|
98
|
+
Use python-magic-bin instead of python-magic for easier installation.
|
|
99
|
+
|
|
100
|
+
## 🛠️ Dependencies
|
|
101
|
+
- PyMuPDF (fitz)
|
|
102
|
+
|
|
103
|
+
- beautifulsoup4
|
|
104
|
+
|
|
105
|
+
- python-docx
|
|
106
|
+
|
|
107
|
+
- python-pptx
|
|
108
|
+
|
|
109
|
+
- odfpy
|
|
110
|
+
|
|
111
|
+
- openpyxl
|
|
112
|
+
|
|
113
|
+
- lxml
|
|
114
|
+
|
|
115
|
+
- xlrd (<2.0.0)
|
|
116
|
+
|
|
117
|
+
- python-magic
|
|
118
|
+
|
|
119
|
+
Dependencies are automatically installed from pyproject.toml.
|
|
120
|
+
|
|
121
|
+
## 📚 Usage Example
|
|
122
|
+
Extract text from a file path:
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
from pyxtxt import xtxt
|
|
126
|
+
|
|
127
|
+
text = xtxt("document.pdf")
|
|
128
|
+
print(text)
|
|
129
|
+
```
|
|
130
|
+
Extract text from a file-like buffer:
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
import io
|
|
134
|
+
|
|
135
|
+
with open("document.docx", "rb") as f:
|
|
136
|
+
buffer = io.BytesIO(f.read())
|
|
137
|
+
|
|
138
|
+
from pyxtxt import xtxt
|
|
139
|
+
text = xtxt(buffer)
|
|
140
|
+
print(text)
|
|
141
|
+
```
|
|
142
|
+
##⚠️ Known Limitations
|
|
143
|
+
When passing a raw stream (io.BytesIO) without a filename, legacy files (.doc, .xls, .ppt) may not be correctly detected.
|
|
144
|
+
|
|
145
|
+
This is a limitation of libmagic, not of pyxtxt.
|
|
146
|
+
|
|
147
|
+
If available, passing the original filename along with the buffer is highly recommended.
|
|
148
|
+
|
|
149
|
+
## 🔒 License
|
|
150
|
+
Distributed under the MIT License.
|
|
151
|
+
|
|
152
|
+
The software is provided "as is" without any warranty of any kind.
|
|
153
|
+
|
|
154
|
+
Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
pyxtxt/__init__.py,sha256=axFUvZRxw1OJQizuGqYKWe6bdJXFnT1SSVfuYzGEovA,48
|
|
2
|
+
pyxtxt/core.py,sha256=kwGklxzc3o2tce3SOKZgwJSm0g2fKCeviYhj0p7Hgmw,2605
|
|
3
|
+
pyxtxt/pyxtxt.py,sha256=HiuoEUSr-VpCOXsk2CGgKQB4w8yfZ-6RIUAPwFbDxVg,8480
|
|
4
|
+
pyxtxt/estrattori/__init__.py,sha256=f0A9636Vi1yMgKMSxkguoYygCkjXN46aDAaKGjWy0Ow,543
|
|
5
|
+
pyxtxt/estrattori/docx.py,sha256=SRiPuW8m-w5B_mpRfYUv9P0-kA_3ubGeae6qm3NENM4,793
|
|
6
|
+
pyxtxt/estrattori/html.py,sha256=tTIKJ0xoJ44qCuUnBu5JUkBuChhB2mopbaWF_S8KIyU,262
|
|
7
|
+
pyxtxt/estrattori/odt.py,sha256=49H7PRfQ5vzKENzBS5PA0uVYaE7oawg7BiqxjkoOPg8,580
|
|
8
|
+
pyxtxt/estrattori/pdf.py,sha256=pg1lwr6vm9syUW285vNSq52KPqoNE8xnsfLjRD7wzv8,651
|
|
9
|
+
pyxtxt/estrattori/pptx.py,sha256=Xx-p9YA4-5NpDZujnu-ikGlA2hvw2SON3DCJSZFfnJg,1040
|
|
10
|
+
pyxtxt/estrattori/svg.py,sha256=Vkf8XBL9mks7gLJs-coTNP9hsjRzxdOCGroBqGeH1yA,515
|
|
11
|
+
pyxtxt/estrattori/txt.py,sha256=6lhD3ccTsExTXJvs024qZ-RCH0uDbcQHmDXf88bzXK4,193
|
|
12
|
+
pyxtxt/estrattori/xls.py,sha256=cWU5Z4XDFP4Xgz8lqqV7d6toWZTBEH03PQOEda-GneY,841
|
|
13
|
+
pyxtxt/estrattori/xlsx.py,sha256=JFglzWiYdUh6PZ_yDPwg0g4Cp4HnAvMoRJEyu5Bg53k,1360
|
|
14
|
+
pyxtxt/estrattori/xml.py,sha256=RA3x19wHS-y2hZIOXt1j7E7d2oRL1-RQBg6dnsSy86U,987
|
|
15
|
+
pyxtxt-0.1.8.dist-info/METADATA,sha256=h596qN1Hd-12an9NP-v7m59Tqedm_svD21zUIlxCws0,4733
|
|
16
|
+
pyxtxt-0.1.8.dist-info/WHEEL,sha256=SmOxYU7pzNKBqASvQJ7DjX3XGUF92lrGhMb3R6_iiqI,91
|
|
17
|
+
pyxtxt-0.1.8.dist-info/top_level.txt,sha256=kHSwy3OBOTla47zuXMsHmSaUpa0wFjEd_xNwsZLpxOc,7
|
|
18
|
+
pyxtxt-0.1.8.dist-info/RECORD,,
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
pyxtxt
|