pyxtxt 0.3.4.2__tar.gz → 0.3.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {pyxtxt-0.3.4.2/src/pyxtxt.egg-info → pyxtxt-0.3.5}/PKG-INFO +8 -1
  2. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/README.md +7 -0
  3. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/pyproject.toml +1 -1
  4. pyxtxt-0.3.5/src/pyxtxt/__init__.py +59 -0
  5. pyxtxt-0.3.5/src/pyxtxt/estrattori/exif.py +183 -0
  6. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5/src/pyxtxt.egg-info}/PKG-INFO +8 -1
  7. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt.egg-info/SOURCES.txt +1 -0
  8. pyxtxt-0.3.4.2/src/pyxtxt/__init__.py +0 -19
  9. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/LICENSE +0 -0
  10. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/MANIFEST.in +0 -0
  11. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/setup.cfg +0 -0
  12. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/core.py +0 -0
  13. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/__init__.py +0 -0
  14. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/audio.py +0 -0
  15. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/doc.py +0 -0
  16. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/docx.py +0 -0
  17. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/eml.py +0 -0
  18. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/epub.py +0 -0
  19. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/html.py +0 -0
  20. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/md.py +0 -0
  21. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/msg.py +0 -0
  22. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/ocr.py +0 -0
  23. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/ocr_ollama.py +0 -0
  24. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/odt.py +0 -0
  25. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/pdf.py +0 -0
  26. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/pptx.py +0 -0
  27. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/rtf.py +0 -0
  28. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/svg.py +0 -0
  29. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/tex.py +0 -0
  30. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/txt.py +0 -0
  31. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/xls.py +0 -0
  32. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/xlsx.py +0 -0
  33. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/estrattori/xml.py +0 -0
  34. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/examples.py +0 -0
  35. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt/pyxtxt.py +0 -0
  36. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  37. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt.egg-info/requires.txt +0 -0
  38. {pyxtxt-0.3.4.2 → pyxtxt-0.3.5}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.4.2
3
+ Version: 0.3.5
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -287,6 +287,13 @@ text = xtxt("scanned_document.png")
287
287
  text = xtxt("screenshot.jpg")
288
288
  text = xtxt("invoice.tiff")
289
289
 
290
+ # Extract EXIF metadata from photos (uses Pillow, already included)
291
+ from pyxtxt import xtxt_exif
292
+
293
+ exif_data = xtxt_exif("vacation_photo.jpg")
294
+ print(exif_data)
295
+ # Output: Camera make/model, GPS coordinates, shooting settings, datetime, etc.
296
+
290
297
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
291
298
  # Requires: ollama server running + gemma3:4b model
292
299
  from pyxtxt import (
@@ -185,6 +185,13 @@ text = xtxt("scanned_document.png")
185
185
  text = xtxt("screenshot.jpg")
186
186
  text = xtxt("invoice.tiff")
187
187
 
188
+ # Extract EXIF metadata from photos (uses Pillow, already included)
189
+ from pyxtxt import xtxt_exif
190
+
191
+ exif_data = xtxt_exif("vacation_photo.jpg")
192
+ print(exif_data)
193
+ # Output: Camera make/model, GPS coordinates, shooting settings, datetime, etc.
194
+
188
195
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
189
196
  # Requires: ollama server running + gemma3:4b model
190
197
  from pyxtxt import (
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.3.4.2"
3
+ version = "0.3.5"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -0,0 +1,59 @@
1
+ from .core import xtxt, extxt_available_formats, xtxt_from_url
2
+
3
+ # Import EXIF functions if available
4
+ try:
5
+ from .estrattori.exif import xtxt_image_exif
6
+ exif_available = True
7
+ except ImportError:
8
+ exif_available = False
9
+
10
+ # Import OCR-Ollama functions if available
11
+ try:
12
+ from .estrattori.ocr_ollama import (
13
+ set_ollama_model, get_ollama_model, xtxt_image_describe,
14
+ set_ollama_config, get_ollama_config, reset_ollama_config,
15
+ xtxt_image_with_confidence, configure_for_medical_images,
16
+ configure_for_xray_images, quick_xray_analysis
17
+ )
18
+ ollama_available = True
19
+ except ImportError:
20
+ ollama_available = False
21
+
22
+ # Build __all__ dynamically based on available features
23
+ __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
24
+
25
+ if exif_available:
26
+ __all__.extend(["xtxt_exif"])
27
+
28
+ if ollama_available:
29
+ __all__.extend([
30
+ "set_ollama_model", "get_ollama_model", "xtxt_image_describe",
31
+ "set_ollama_config", "get_ollama_config", "reset_ollama_config",
32
+ "xtxt_image_with_confidence", "configure_for_medical_images",
33
+ "configure_for_xray_images", "quick_xray_analysis"
34
+ ])
35
+
36
+ # Define EXIF wrapper function
37
+ def xtxt_exif(file_input):
38
+ """
39
+ Extract EXIF metadata from image files.
40
+
41
+ Args:
42
+ file_input: Image file path (str) or file-like object
43
+
44
+ Returns:
45
+ str: Formatted EXIF metadata or empty string if no EXIF data
46
+
47
+ Example:
48
+ exif_data = xtxt_exif("photo.jpg")
49
+ print(exif_data)
50
+ """
51
+ if not exif_available:
52
+ raise ImportError("EXIF extraction requires Pillow. Install with: pip install pyxtxt[ocr]")
53
+
54
+ # Handle file path
55
+ if isinstance(file_input, str):
56
+ with open(file_input, 'rb') as f:
57
+ return xtxt_image_exif(f)
58
+ else:
59
+ return xtxt_image_exif(file_input)
@@ -0,0 +1,183 @@
1
+ # pyxtxt/extractors/image_exif.py
2
+ from . import register_extractor
3
+ from io import BytesIO
4
+ import json
5
+
6
+ try:
7
+ from PIL import Image, ExifTags
8
+ from PIL.ExifTags import TAGS, GPSTAGS
9
+ except ImportError:
10
+ Image = None
11
+ ExifTags = None
12
+ TAGS = None
13
+ GPSTAGS = None
14
+
15
+ if Image and ExifTags and TAGS:
16
+ def xtxt_image_exif(file_buffer):
17
+ """
18
+ Extract EXIF metadata from images as human-readable text.
19
+
20
+ Returns formatted text with:
21
+ - Camera settings (make, model, lens, ISO, aperture, etc.)
22
+ - Shooting parameters (exposure, flash, focal length, etc.)
23
+ - DateTime information (creation, modification dates)
24
+ - GPS coordinates (if available)
25
+ - Technical metadata (dimensions, orientation, color space, etc.)
26
+ """
27
+ try:
28
+ # Convert buffer to PIL Image
29
+ if hasattr(file_buffer, 'seek'):
30
+ file_buffer.seek(0)
31
+ image_data = file_buffer.read()
32
+ image = Image.open(BytesIO(image_data))
33
+
34
+ # Get EXIF data
35
+ exif_data = image._getexif()
36
+
37
+ if not exif_data:
38
+ return "NO_EXIF_DATA_FOUND"
39
+
40
+ # Extract readable EXIF information
41
+ exif_text_lines = []
42
+ gps_data = {}
43
+
44
+ # Process main EXIF tags
45
+ for tag_id, value in exif_data.items():
46
+ tag_name = TAGS.get(tag_id, f"UnknownTag_{tag_id}")
47
+
48
+ # Handle special GPS data
49
+ if tag_name == "GPSInfo" and isinstance(value, dict):
50
+ gps_data = value
51
+ continue
52
+
53
+ # Format common values
54
+ if tag_name in ["DateTime", "DateTimeOriginal", "DateTimeDigitized"]:
55
+ exif_text_lines.append(f"{tag_name}: {value}")
56
+ elif tag_name in ["Make", "Model", "Software", "Artist", "Copyright"]:
57
+ exif_text_lines.append(f"{tag_name}: {value}")
58
+ elif tag_name in ["XResolution", "YResolution"]:
59
+ if isinstance(value, tuple) and len(value) == 2:
60
+ resolution = value[0] / value[1] if value[1] != 0 else value[0]
61
+ exif_text_lines.append(f"{tag_name}: {resolution:.1f} dpi")
62
+ else:
63
+ exif_text_lines.append(f"{tag_name}: {value}")
64
+ elif tag_name in ["FNumber", "FocalLength", "ExposureTime"]:
65
+ if isinstance(value, tuple) and len(value) == 2:
66
+ if value[1] != 0:
67
+ if tag_name == "FNumber":
68
+ f_value = value[0] / value[1]
69
+ exif_text_lines.append(f"Aperture: f/{f_value:.1f}")
70
+ elif tag_name == "FocalLength":
71
+ focal_mm = value[0] / value[1]
72
+ exif_text_lines.append(f"Focal Length: {focal_mm:.0f}mm")
73
+ elif tag_name == "ExposureTime":
74
+ if value[0] == 1:
75
+ exif_text_lines.append(f"Shutter Speed: 1/{value[1]}s")
76
+ else:
77
+ exp_time = value[0] / value[1]
78
+ exif_text_lines.append(f"Shutter Speed: {exp_time:.3f}s")
79
+ else:
80
+ exif_text_lines.append(f"{tag_name}: {value}")
81
+ else:
82
+ exif_text_lines.append(f"{tag_name}: {value}")
83
+ elif tag_name == "ISOSpeedRatings":
84
+ exif_text_lines.append(f"ISO: {value}")
85
+ elif tag_name == "Flash":
86
+ flash_modes = {
87
+ 0: "No Flash",
88
+ 1: "Flash Fired",
89
+ 5: "Flash Fired, Return not detected",
90
+ 7: "Flash Fired, Return detected",
91
+ 9: "Flash Fired, Compulsory",
92
+ 13: "Flash Fired, Compulsory, Return not detected",
93
+ 15: "Flash Fired, Compulsory, Return detected",
94
+ 16: "No Flash, Compulsory",
95
+ 24: "No Flash, Auto",
96
+ 25: "Flash Fired, Auto",
97
+ 29: "Flash Fired, Auto, Return not detected",
98
+ 31: "Flash Fired, Auto, Return detected",
99
+ 32: "No Flash Available"
100
+ }
101
+ flash_desc = flash_modes.get(value, f"Flash Mode {value}")
102
+ exif_text_lines.append(f"Flash: {flash_desc}")
103
+ elif tag_name in ["ExposureMode", "WhiteBalance", "SceneCaptureType", "MeteringMode"]:
104
+ exif_text_lines.append(f"{tag_name}: {value}")
105
+ elif tag_name == "Orientation":
106
+ orientations = {
107
+ 1: "Normal", 2: "Mirrored horizontal", 3: "Rotated 180°",
108
+ 4: "Mirrored vertical", 5: "Mirrored horizontal + rotated 90° CCW",
109
+ 6: "Rotated 90° CW", 7: "Mirrored horizontal + rotated 90° CW",
110
+ 8: "Rotated 90° CCW"
111
+ }
112
+ orient_desc = orientations.get(value, f"Orientation {value}")
113
+ exif_text_lines.append(f"Orientation: {orient_desc}")
114
+ elif isinstance(value, (str, int, float)):
115
+ # Include other simple values
116
+ exif_text_lines.append(f"{tag_name}: {value}")
117
+
118
+ # Process GPS data if available
119
+ if gps_data:
120
+ gps_text_lines = []
121
+ gps_info = {}
122
+
123
+ # Extract GPS coordinates
124
+ for gps_tag_id, gps_value in gps_data.items():
125
+ gps_tag_name = GPSTAGS.get(gps_tag_id, f"GPSTag_{gps_tag_id}")
126
+ gps_info[gps_tag_name] = gps_value
127
+
128
+ # Format coordinates if available
129
+ if 'GPSLatitude' in gps_info and 'GPSLatitudeRef' in gps_info:
130
+ lat_dms = gps_info['GPSLatitude']
131
+ lat_ref = gps_info['GPSLatitudeRef']
132
+ if len(lat_dms) == 3:
133
+ lat_deg = lat_dms[0] + lat_dms[1]/60 + lat_dms[2]/3600
134
+ if lat_ref == 'S':
135
+ lat_deg = -lat_deg
136
+ gps_text_lines.append(f"GPS Latitude: {lat_deg:.6f}° {lat_ref}")
137
+
138
+ if 'GPSLongitude' in gps_info and 'GPSLongitudeRef' in gps_info:
139
+ lon_dms = gps_info['GPSLongitude']
140
+ lon_ref = gps_info['GPSLongitudeRef']
141
+ if len(lon_dms) == 3:
142
+ lon_deg = lon_dms[0] + lon_dms[1]/60 + lon_dms[2]/3600
143
+ if lon_ref == 'W':
144
+ lon_deg = -lon_deg
145
+ gps_text_lines.append(f"GPS Longitude: {lon_deg:.6f}° {lon_ref}")
146
+
147
+ # Add other GPS info
148
+ for tag_name, value in gps_info.items():
149
+ if tag_name not in ['GPSLatitude', 'GPSLatitudeRef', 'GPSLongitude', 'GPSLongitudeRef']:
150
+ if isinstance(value, (str, int, float)):
151
+ gps_text_lines.append(f"{tag_name}: {value}")
152
+
153
+ if gps_text_lines:
154
+ exif_text_lines.extend(["", "=== GPS Information ==="] + gps_text_lines)
155
+
156
+ # Add image basic info
157
+ basic_info = [
158
+ "",
159
+ "=== Image Information ===",
160
+ f"Format: {image.format}",
161
+ f"Mode: {image.mode}",
162
+ f"Size: {image.width} x {image.height} pixels"
163
+ ]
164
+
165
+ # Return formatted EXIF data
166
+ if exif_text_lines:
167
+ result = "=== EXIF Metadata ===" + "\n" + "\n".join(exif_text_lines) + "\n" + "\n".join(basic_info)
168
+ return result
169
+ else:
170
+ return "NO_READABLE_EXIF_DATA"
171
+
172
+ except Exception as e:
173
+ print(f"⚠️ Error extracting EXIF from image: {e}")
174
+ return ""
175
+
176
+ # Register EXIF extractor for image formats with dedicated MIME types
177
+ # Using EXIF-specific MIME types to avoid conflicts with OCR extractors
178
+ register_extractor("image/jpeg+exif", xtxt_image_exif, name="EXIF")
179
+ register_extractor("image/jpg+exif", xtxt_image_exif, name="EXIF")
180
+ register_extractor("image/png+exif", xtxt_image_exif, name="EXIF")
181
+ register_extractor("image/tiff+exif", xtxt_image_exif, name="EXIF")
182
+ register_extractor("image/bmp+exif", xtxt_image_exif, name="EXIF")
183
+ register_extractor("image/webp+exif", xtxt_image_exif, name="EXIF")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.4.2
3
+ Version: 0.3.5
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -287,6 +287,13 @@ text = xtxt("scanned_document.png")
287
287
  text = xtxt("screenshot.jpg")
288
288
  text = xtxt("invoice.tiff")
289
289
 
290
+ # Extract EXIF metadata from photos (uses Pillow, already included)
291
+ from pyxtxt import xtxt_exif
292
+
293
+ exif_data = xtxt_exif("vacation_photo.jpg")
294
+ print(exif_data)
295
+ # Output: Camera make/model, GPS coordinates, shooting settings, datetime, etc.
296
+
290
297
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
291
298
  # Requires: ollama server running + gemma3:4b model
292
299
  from pyxtxt import (
@@ -17,6 +17,7 @@ src/pyxtxt/estrattori/doc.py
17
17
  src/pyxtxt/estrattori/docx.py
18
18
  src/pyxtxt/estrattori/eml.py
19
19
  src/pyxtxt/estrattori/epub.py
20
+ src/pyxtxt/estrattori/exif.py
20
21
  src/pyxtxt/estrattori/html.py
21
22
  src/pyxtxt/estrattori/md.py
22
23
  src/pyxtxt/estrattori/msg.py
@@ -1,19 +0,0 @@
1
- from .core import xtxt, extxt_available_formats, xtxt_from_url
2
-
3
- # Import OCR-Ollama functions if available
4
- try:
5
- from .estrattori.ocr_ollama import (
6
- set_ollama_model, get_ollama_model, xtxt_image_describe,
7
- set_ollama_config, get_ollama_config, reset_ollama_config,
8
- xtxt_image_with_confidence, configure_for_medical_images,
9
- configure_for_xray_images, quick_xray_analysis
10
- )
11
- __all__ = [
12
- "xtxt", "extxt_available_formats", "xtxt_from_url",
13
- "set_ollama_model", "get_ollama_model", "xtxt_image_describe",
14
- "set_ollama_config", "get_ollama_config", "reset_ollama_config",
15
- "xtxt_image_with_confidence", "configure_for_medical_images",
16
- "configure_for_xray_images", "quick_xray_analysis"
17
- ]
18
- except ImportError:
19
- __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes