pyxtxt 0.3.4__tar.gz → 0.3.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyxtxt-0.3.4/src/pyxtxt.egg-info → pyxtxt-0.3.4.2}/PKG-INFO +28 -3
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/README.md +27 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/pyproject.toml +1 -3
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/__init__.py +4 -2
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/ocr_ollama.py +338 -42
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2/src/pyxtxt.egg-info}/PKG-INFO +28 -3
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt.egg-info/requires.txt +0 -2
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/LICENSE +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/MANIFEST.in +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/setup.cfg +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/core.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/audio.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/doc.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/docx.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/eml.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/epub.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/md.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/msg.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/ocr.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/odt.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/pptx.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/rtf.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/svg.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/tex.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/xlsx.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/examples.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/pyxtxt.py +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.3.4
|
|
3
|
+
Version: 0.3.4.2
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -57,7 +57,6 @@ Provides-Extra: html
|
|
|
57
57
|
Requires-Dist: beautifulsoup4; extra == "html"
|
|
58
58
|
Requires-Dist: lxml; extra == "html"
|
|
59
59
|
Provides-Extra: doc
|
|
60
|
-
Requires-Dist: textract; extra == "doc"
|
|
61
60
|
Provides-Extra: markdown
|
|
62
61
|
Requires-Dist: markdown; extra == "markdown"
|
|
63
62
|
Requires-Dist: beautifulsoup4; extra == "markdown"
|
|
@@ -82,7 +81,6 @@ Provides-Extra: ocr-ollama
|
|
|
82
81
|
Requires-Dist: ollama; extra == "ocr-ollama"
|
|
83
82
|
Requires-Dist: pillow; extra == "ocr-ollama"
|
|
84
83
|
Provides-Extra: all
|
|
85
|
-
Requires-Dist: textract; extra == "all"
|
|
86
84
|
Requires-Dist: PyMuPDF; extra == "all"
|
|
87
85
|
Requires-Dist: python-docx; extra == "all"
|
|
88
86
|
Requires-Dist: python-pptx; extra == "all"
|
|
@@ -196,6 +194,33 @@ Use python-magic-bin instead of python-magic for easier installation.
|
|
|
196
194
|
|
|
197
195
|
Dependencies are automatically installed based on selected optional groups.
|
|
198
196
|
|
|
197
|
+
### System Dependencies
|
|
198
|
+
Some extractors require system-level tools to be installed:
|
|
199
|
+
|
|
200
|
+
- **Legacy DOC files**: `antiword` - Install via your package manager:
|
|
201
|
+
```bash
|
|
202
|
+
# Ubuntu/Debian
|
|
203
|
+
sudo apt install antiword
|
|
204
|
+
|
|
205
|
+
# macOS
|
|
206
|
+
brew install antiword
|
|
207
|
+
|
|
208
|
+
# CentOS/RHEL
|
|
209
|
+
sudo yum install antiword
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
- **Audio/Video transcription**: `ffmpeg` - Required for audio preprocessing:
|
|
213
|
+
```bash
|
|
214
|
+
# Ubuntu/Debian
|
|
215
|
+
sudo apt install ffmpeg
|
|
216
|
+
|
|
217
|
+
# macOS
|
|
218
|
+
brew install ffmpeg
|
|
219
|
+
|
|
220
|
+
# Windows
|
|
221
|
+
# Download from https://ffmpeg.org/download.html
|
|
222
|
+
```
|
|
223
|
+
|
|
199
224
|
## 📚 Usage Examples
|
|
200
225
|
|
|
201
226
|
### Basic Usage
|
|
@@ -92,6 +92,33 @@ Use python-magic-bin instead of python-magic for easier installation.
|
|
|
92
92
|
|
|
93
93
|
Dependencies are automatically installed based on selected optional groups.
|
|
94
94
|
|
|
95
|
+
### System Dependencies
|
|
96
|
+
Some extractors require system-level tools to be installed:
|
|
97
|
+
|
|
98
|
+
- **Legacy DOC files**: `antiword` - Install via your package manager:
|
|
99
|
+
```bash
|
|
100
|
+
# Ubuntu/Debian
|
|
101
|
+
sudo apt install antiword
|
|
102
|
+
|
|
103
|
+
# macOS
|
|
104
|
+
brew install antiword
|
|
105
|
+
|
|
106
|
+
# CentOS/RHEL
|
|
107
|
+
sudo yum install antiword
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
- **Audio/Video transcription**: `ffmpeg` - Required for audio preprocessing:
|
|
111
|
+
```bash
|
|
112
|
+
# Ubuntu/Debian
|
|
113
|
+
sudo apt install ffmpeg
|
|
114
|
+
|
|
115
|
+
# macOS
|
|
116
|
+
brew install ffmpeg
|
|
117
|
+
|
|
118
|
+
# Windows
|
|
119
|
+
# Download from https://ffmpeg.org/download.html
|
|
120
|
+
```
|
|
121
|
+
|
|
95
122
|
## 📚 Usage Examples
|
|
96
123
|
|
|
97
124
|
### Basic Usage
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "pyxtxt"
|
|
3
|
-
version = "0.3.4"
|
|
3
|
+
version = "0.3.4.2"
|
|
4
4
|
description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
|
|
5
5
|
classifiers = [
|
|
6
6
|
"Development Status :: 4 - Beta",
|
|
@@ -50,7 +50,6 @@ html = [
|
|
|
50
50
|
"lxml",
|
|
51
51
|
]
|
|
52
52
|
doc = [
|
|
53
|
-
"textract",
|
|
54
53
|
]
|
|
55
54
|
markdown = [
|
|
56
55
|
"markdown",
|
|
@@ -85,7 +84,6 @@ ocr-ollama = [
|
|
|
85
84
|
"pillow",
|
|
86
85
|
]
|
|
87
86
|
all = [
|
|
88
|
-
"textract",
|
|
89
87
|
"PyMuPDF",
|
|
90
88
|
"python-docx",
|
|
91
89
|
"python-pptx",
|
|
@@ -5,13 +5,15 @@ try:
|
|
|
5
5
|
from .estrattori.ocr_ollama import (
|
|
6
6
|
set_ollama_model, get_ollama_model, xtxt_image_describe,
|
|
7
7
|
set_ollama_config, get_ollama_config, reset_ollama_config,
|
|
8
|
-
xtxt_image_with_confidence
|
|
8
|
+
xtxt_image_with_confidence, configure_for_medical_images,
|
|
9
|
+
configure_for_xray_images, quick_xray_analysis
|
|
9
10
|
)
|
|
10
11
|
__all__ = [
|
|
11
12
|
"xtxt", "extxt_available_formats", "xtxt_from_url",
|
|
12
13
|
"set_ollama_model", "get_ollama_model", "xtxt_image_describe",
|
|
13
14
|
"set_ollama_config", "get_ollama_config", "reset_ollama_config",
|
|
14
|
-
"xtxt_image_with_confidence"
|
|
15
|
+
"xtxt_image_with_confidence", "configure_for_medical_images",
|
|
16
|
+
"configure_for_xray_images", "quick_xray_analysis"
|
|
15
17
|
]
|
|
16
18
|
except ImportError:
|
|
17
19
|
__all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
|
|
@@ -19,7 +19,12 @@ OLLAMA_CONFIG = {
|
|
|
19
19
|
'temperature': 0.1, # Response creativity (0.0-1.0)
|
|
20
20
|
'max_tokens': 1500, # Maximum response length
|
|
21
21
|
'confidence_threshold': 0.7, # Minimum confidence for text extraction
|
|
22
|
-
'context': 'general' # Context hint: general, cookbook, document, diagram,
|
|
22
|
+
'context': 'general', # Context hint: general, cookbook, document, diagram, medical, xray
|
|
23
|
+
'enhance_image': True, # Apply image enhancement preprocessing
|
|
24
|
+
'min_size': 800, # Minimum image size for processing (upscale if smaller)
|
|
25
|
+
'max_size': 2048, # Maximum image size (downscale if larger)
|
|
26
|
+
'auto_fallback': True, # Enable automatic model fallback for better results
|
|
27
|
+
'fallback_models': ['gemma3:27b', 'gemma3:12b', 'llava:13b'] # Models to try if primary fails
|
|
23
28
|
}
|
|
24
29
|
|
|
25
30
|
def set_ollama_model(model_name: str):
|
|
@@ -49,7 +54,12 @@ def set_ollama_config(**kwargs):
|
|
|
49
54
|
- language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
|
|
50
55
|
- caption_length: Caption length ('short', 'medium', 'long')
|
|
51
56
|
- style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
|
|
52
|
-
- context: Content context ('general', 'cookbook', 'document', 'handwriting', 'technical')
|
|
57
|
+
- context: Content context ('general', 'cookbook', 'document', 'handwriting', 'technical', 'medical', 'xray')
|
|
58
|
+
- enhance_image: Apply image enhancement preprocessing (default: True)
|
|
59
|
+
- min_size: Minimum image size for processing - upscale if smaller (default: 800)
|
|
60
|
+
- max_size: Maximum image size - downscale if larger (default: 2048)
|
|
61
|
+
- auto_fallback: Enable automatic model fallback for better results (default: True)
|
|
62
|
+
- fallback_models: List of models to try if primary fails (default: ['gemma3:27b', 'gemma3:12b', 'llava:13b'])
|
|
53
63
|
- temperature: Response creativity 0.0-1.0 (default: 0.1)
|
|
54
64
|
- max_tokens: Maximum response length (default: 1500)
|
|
55
65
|
- confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
|
|
@@ -58,6 +68,10 @@ def set_ollama_config(**kwargs):
|
|
|
58
68
|
set_ollama_config(language='italian', style='detailed')
|
|
59
69
|
set_ollama_config(context='document', caption_length='long')
|
|
60
70
|
set_ollama_config(context='handwriting', temperature=0.2)
|
|
71
|
+
set_ollama_config(context='medical', enhance_image=True, min_size=1024)
|
|
72
|
+
set_ollama_config(context='xray', style='technical', confidence_threshold=0.8)
|
|
73
|
+
set_ollama_config(auto_fallback=False) # Disable fallback for faster processing
|
|
74
|
+
set_ollama_config(fallback_models=['llava:13b', 'gemma3:27b']) # Custom fallback sequence
|
|
61
75
|
"""
|
|
62
76
|
global OLLAMA_CONFIG
|
|
63
77
|
for key, value in kwargs.items():
|
|
@@ -71,6 +85,105 @@ def get_ollama_config():
|
|
|
71
85
|
"""Get current Ollama configuration"""
|
|
72
86
|
return OLLAMA_CONFIG.copy()
|
|
73
87
|
|
|
88
|
+
def configure_for_medical_images():
|
|
89
|
+
"""
|
|
90
|
+
Quick configuration setup for optimal medical image processing.
|
|
91
|
+
Configures model, enhancement, and context for X-rays and medical scans.
|
|
92
|
+
"""
|
|
93
|
+
print("🏥 Configuring for medical image processing...")
|
|
94
|
+
|
|
95
|
+
# Optimal medical settings
|
|
96
|
+
global OLLAMA_CONFIG
|
|
97
|
+
OLLAMA_CONFIG.update({
|
|
98
|
+
'context': 'medical',
|
|
99
|
+
'enhance_image': True,
|
|
100
|
+
'min_size': 1024, # Higher resolution for medical details
|
|
101
|
+
'max_size': 2048, # Keep high quality
|
|
102
|
+
'style': 'technical',
|
|
103
|
+
'confidence_threshold': 0.8, # Higher threshold for medical accuracy
|
|
104
|
+
'temperature': 0.05, # Very low temperature for precision
|
|
105
|
+
'auto_fallback': True,
|
|
106
|
+
'fallback_models': ['gemma3:27b', 'llava:13b', 'gemma3:12b'] # Best models first
|
|
107
|
+
})
|
|
108
|
+
|
|
109
|
+
# Suggest high-quality model if current is default
|
|
110
|
+
current_model = get_ollama_model()
|
|
111
|
+
if current_model == "gemma3:4b":
|
|
112
|
+
print("💡 Consider upgrading to gemma3:27b for better medical image recognition")
|
|
113
|
+
print(" Run: set_ollama_model('gemma3:27b')")
|
|
114
|
+
|
|
115
|
+
print("✅ Medical configuration applied:")
|
|
116
|
+
print(f" - Enhanced image processing: {OLLAMA_CONFIG['enhance_image']}")
|
|
117
|
+
print(f" - Minimum resolution: {OLLAMA_CONFIG['min_size']}px")
|
|
118
|
+
print(f" - Confidence threshold: {OLLAMA_CONFIG['confidence_threshold']}")
|
|
119
|
+
print(f" - Fallback models: {len(OLLAMA_CONFIG['fallback_models'])} configured")
|
|
120
|
+
|
|
121
|
+
def configure_for_xray_images():
|
|
122
|
+
"""
|
|
123
|
+
Specialized configuration for X-ray and radiological image processing.
|
|
124
|
+
Optimized for detecting small text, markers, and technical annotations.
|
|
125
|
+
"""
|
|
126
|
+
print("📷 Configuring for X-ray image processing...")
|
|
127
|
+
|
|
128
|
+
# X-ray specific settings
|
|
129
|
+
global OLLAMA_CONFIG
|
|
130
|
+
OLLAMA_CONFIG.update({
|
|
131
|
+
'context': 'xray',
|
|
132
|
+
'enhance_image': True,
|
|
133
|
+
'min_size': 1200, # Even higher for X-ray details
|
|
134
|
+
'max_size': 2048,
|
|
135
|
+
'style': 'technical',
|
|
136
|
+
'confidence_threshold': 0.85, # Very high threshold for X-ray accuracy
|
|
137
|
+
'temperature': 0.02, # Minimal creativity for technical precision
|
|
138
|
+
'auto_fallback': True,
|
|
139
|
+
'fallback_models': ['gemma3:27b', 'llava:13b'] # Only best models
|
|
140
|
+
})
|
|
141
|
+
|
|
142
|
+
# Recommend best model for X-rays
|
|
143
|
+
current_model = get_ollama_model()
|
|
144
|
+
if current_model != "gemma3:27b":
|
|
145
|
+
print("🎯 For best X-ray results, use gemma3:27b model")
|
|
146
|
+
print(" Run: set_ollama_model('gemma3:27b')")
|
|
147
|
+
|
|
148
|
+
print("✅ X-ray configuration applied:")
|
|
149
|
+
print(f" - Context: Medical X-ray specialization")
|
|
150
|
+
print(f" - Enhanced processing: Advanced medical filters")
|
|
151
|
+
print(f" - High precision mode: {OLLAMA_CONFIG['confidence_threshold']} threshold")
|
|
152
|
+
print(f" - Temperature: {OLLAMA_CONFIG['temperature']} (maximum precision)")
|
|
153
|
+
|
|
154
|
+
def quick_xray_analysis(image_path: str) -> str:
|
|
155
|
+
"""
|
|
156
|
+
Convenient one-function X-ray analysis with optimal settings.
|
|
157
|
+
Automatically configures system for X-ray processing and returns detailed analysis.
|
|
158
|
+
|
|
159
|
+
Args:
|
|
160
|
+
image_path: Path to X-ray image file
|
|
161
|
+
|
|
162
|
+
Returns:
|
|
163
|
+
str: Detailed text + description analysis
|
|
164
|
+
"""
|
|
165
|
+
# Save current config
|
|
166
|
+
global OLLAMA_CONFIG
|
|
167
|
+
original_config = OLLAMA_CONFIG.copy()
|
|
168
|
+
original_model = get_ollama_model()
|
|
169
|
+
|
|
170
|
+
try:
|
|
171
|
+
# Apply X-ray configuration
|
|
172
|
+
configure_for_xray_images()
|
|
173
|
+
set_ollama_model('gemma3:27b') # Use best model
|
|
174
|
+
|
|
175
|
+
# Perform analysis
|
|
176
|
+
print("🔍 Analyzing X-ray image...")
|
|
177
|
+
result = xtxt_image_describe(image_path)
|
|
178
|
+
|
|
179
|
+
return result
|
|
180
|
+
|
|
181
|
+
finally:
|
|
182
|
+
# Restore original configuration
|
|
183
|
+
OLLAMA_CONFIG = original_config
|
|
184
|
+
set_ollama_model(original_model)
|
|
185
|
+
print("🔄 Configuration restored")
|
|
186
|
+
|
|
74
187
|
def reset_ollama_config():
|
|
75
188
|
"""Reset Ollama configuration to defaults"""
|
|
76
189
|
global OLLAMA_CONFIG
|
|
@@ -81,10 +194,125 @@ def reset_ollama_config():
|
|
|
81
194
|
'temperature': 0.1,
|
|
82
195
|
'max_tokens': 1500,
|
|
83
196
|
'confidence_threshold': 0.7,
|
|
84
|
-
'context': 'general'
|
|
197
|
+
'context': 'general',
|
|
198
|
+
'enhance_image': True,
|
|
199
|
+
'min_size': 800,
|
|
200
|
+
'max_size': 2048,
|
|
201
|
+
'auto_fallback': True,
|
|
202
|
+
'fallback_models': ['gemma3:27b', 'gemma3:12b', 'llava:13b']
|
|
85
203
|
}
|
|
86
204
|
print("✅ Ollama configuration reset to defaults")
|
|
87
205
|
|
|
206
|
+
def _enhance_image(image: Image.Image, context: str, min_size: int, max_size: int) -> Image.Image:
|
|
207
|
+
"""
|
|
208
|
+
Apply image enhancement preprocessing for better OCR results.
|
|
209
|
+
|
|
210
|
+
Args:
|
|
211
|
+
image: PIL Image object
|
|
212
|
+
context: Content context hint (medical, xray, document, etc.)
|
|
213
|
+
min_size: Minimum size threshold for upscaling
|
|
214
|
+
max_size: Maximum size threshold for downscaling
|
|
215
|
+
|
|
216
|
+
Returns:
|
|
217
|
+
Enhanced PIL Image
|
|
218
|
+
"""
|
|
219
|
+
try:
|
|
220
|
+
from PIL import ImageEnhance, ImageFilter, ImageOps
|
|
221
|
+
except ImportError:
|
|
222
|
+
# If PIL enhancements not available, return original
|
|
223
|
+
return image
|
|
224
|
+
|
|
225
|
+
enhanced = image.copy()
|
|
226
|
+
|
|
227
|
+
# Get current dimensions
|
|
228
|
+
width, height = enhanced.size
|
|
229
|
+
max_dimension = max(width, height)
|
|
230
|
+
|
|
231
|
+
# Resize if needed (quality improvement for small images, memory management for large)
|
|
232
|
+
if max_dimension < min_size:
|
|
233
|
+
# Upscale small images for better model processing
|
|
234
|
+
scale_factor = min_size / max_dimension
|
|
235
|
+
new_width = int(width * scale_factor)
|
|
236
|
+
new_height = int(height * scale_factor)
|
|
237
|
+
enhanced = enhanced.resize((new_width, new_height), Image.Resampling.LANCZOS)
|
|
238
|
+
print(f"📈 Image upscaled from {width}x{height} to {new_width}x{new_height}")
|
|
239
|
+
elif max_dimension > max_size:
|
|
240
|
+
# Downscale large images to manageable size
|
|
241
|
+
scale_factor = max_size / max_dimension
|
|
242
|
+
new_width = int(width * scale_factor)
|
|
243
|
+
new_height = int(height * scale_factor)
|
|
244
|
+
enhanced = enhanced.resize((new_width, new_height), Image.Resampling.LANCZOS)
|
|
245
|
+
print(f"📉 Image downscaled from {width}x{height} to {new_width}x{new_height}")
|
|
246
|
+
|
|
247
|
+
# Context-specific enhancements
|
|
248
|
+
if context in ['medical', 'xray']:
|
|
249
|
+
print("🏥 Applying medical image enhancements...")
|
|
250
|
+
# Medical images often benefit from:
|
|
251
|
+
# 1. Contrast enhancement to bring out subtle details
|
|
252
|
+
contrast = ImageEnhance.Contrast(enhanced)
|
|
253
|
+
enhanced = contrast.enhance(1.3)
|
|
254
|
+
|
|
255
|
+
# 2. Sharpening to improve edge definition
|
|
256
|
+
enhanced = enhanced.filter(ImageFilter.UnsharpMask(radius=1, percent=120, threshold=3))
|
|
257
|
+
|
|
258
|
+
# 3. Brightness adjustment for dark X-rays
|
|
259
|
+
brightness = ImageEnhance.Brightness(enhanced)
|
|
260
|
+
enhanced = brightness.enhance(1.1)
|
|
261
|
+
|
|
262
|
+
elif context in ['document', 'handwriting', 'technical']:
|
|
263
|
+
print("📄 Applying document enhancement...")
|
|
264
|
+
# Documents benefit from:
|
|
265
|
+
# 1. Moderate sharpening for text clarity
|
|
266
|
+
enhanced = enhanced.filter(ImageFilter.UnsharpMask(radius=0.5, percent=150, threshold=3))
|
|
267
|
+
|
|
268
|
+
# 2. Contrast boost for faded text
|
|
269
|
+
contrast = ImageEnhance.Contrast(enhanced)
|
|
270
|
+
enhanced = contrast.enhance(1.2)
|
|
271
|
+
|
|
272
|
+
elif context == 'general':
|
|
273
|
+
print("🔧 Applying general enhancements...")
|
|
274
|
+
# General purpose light enhancement
|
|
275
|
+
# 1. Slight contrast improvement
|
|
276
|
+
contrast = ImageEnhance.Contrast(enhanced)
|
|
277
|
+
enhanced = contrast.enhance(1.1)
|
|
278
|
+
|
|
279
|
+
# 2. Subtle sharpening
|
|
280
|
+
enhanced = enhanced.filter(ImageFilter.UnsharpMask(radius=0.5, percent=110, threshold=3))
|
|
281
|
+
|
|
282
|
+
return enhanced
|
|
283
|
+
|
|
284
|
+
def _try_ollama_request(prompt: str, img_base64: str, current_model: str, config: dict) -> tuple:
|
|
285
|
+
"""
|
|
286
|
+
Attempt Ollama request with a specific model.
|
|
287
|
+
|
|
288
|
+
Returns:
|
|
289
|
+
tuple: (success: bool, response: str, confidence: float)
|
|
290
|
+
"""
|
|
291
|
+
try:
|
|
292
|
+
print(f"🤖 Trying model: {current_model}")
|
|
293
|
+
|
|
294
|
+
response = ollama.generate(
|
|
295
|
+
model=current_model,
|
|
296
|
+
prompt=prompt,
|
|
297
|
+
images=[img_base64],
|
|
298
|
+
options={
|
|
299
|
+
'temperature': config['temperature'],
|
|
300
|
+
'top_p': 0.9,
|
|
301
|
+
'num_predict': config['max_tokens']
|
|
302
|
+
}
|
|
303
|
+
)
|
|
304
|
+
|
|
305
|
+
extracted_content = response.get('response', '').strip()
|
|
306
|
+
|
|
307
|
+
# Calculate confidence without printing warnings here (let parent handle it)
|
|
308
|
+
confidence_score = _calculate_confidence_score(extracted_content, "ocr" if "Extracted text:" in prompt else "describe")
|
|
309
|
+
|
|
310
|
+
return True, extracted_content, confidence_score
|
|
311
|
+
|
|
312
|
+
except Exception as e:
|
|
313
|
+
print(f"❌ Model {current_model} failed: {e}")
|
|
314
|
+
return False, "", 0.0
|
|
315
|
+
|
|
88
316
|
def _calculate_confidence_score(content: str, mode: str) -> float:
|
|
89
317
|
"""
|
|
90
318
|
Calculate confidence score (0.0-1.0) for OCR/caption quality.
|
|
@@ -261,14 +489,19 @@ if ollama and Image:
|
|
|
261
489
|
if image.mode != 'RGB':
|
|
262
490
|
image = image.convert('RGB')
|
|
263
491
|
|
|
492
|
+
# Get current configuration
|
|
493
|
+
config = OLLAMA_CONFIG
|
|
494
|
+
|
|
495
|
+
# Apply image enhancement if enabled
|
|
496
|
+
if config.get('enhance_image', True):
|
|
497
|
+
print("⚡ Enhancing image for better OCR...")
|
|
498
|
+
image = _enhance_image(image, config['context'], config['min_size'], config['max_size'])
|
|
499
|
+
|
|
264
500
|
# Convert image to base64
|
|
265
501
|
buffered = BytesIO()
|
|
266
502
|
image.save(buffered, format="PNG")
|
|
267
503
|
img_base64 = base64.b64encode(buffered.getvalue()).decode()
|
|
268
504
|
|
|
269
|
-
# Get current configuration
|
|
270
|
-
config = OLLAMA_CONFIG
|
|
271
|
-
|
|
272
505
|
# Build language hint
|
|
273
506
|
lang_hint = ""
|
|
274
507
|
if config['language'] != 'auto':
|
|
@@ -276,7 +509,43 @@ if ollama and Image:
|
|
|
276
509
|
|
|
277
510
|
# Different prompts based on mode
|
|
278
511
|
if mode == "ocr":
|
|
279
|
-
|
|
512
|
+
# Context-specific OCR prompts
|
|
513
|
+
if config.get('context') == 'xray':
|
|
514
|
+
prompt = f"""This is a medical X-ray or radiological image. Look carefully and extract ALL visible text, including:
|
|
515
|
+
- Patient identification numbers, names, or codes
|
|
516
|
+
- Date and time stamps (exam dates, birth dates)
|
|
517
|
+
- Anatomical position markers (L/R, LEFT/RIGHT, AP, LAT)
|
|
518
|
+
- Measurement scales, rulers, or calibration marks
|
|
519
|
+
- Technical annotations or radiologist markings
|
|
520
|
+
- Equipment identifiers or hospital names
|
|
521
|
+
- Any small text on borders or corners
|
|
522
|
+
|
|
523
|
+
IMPORTANT:
|
|
524
|
+
- Medical images often have small text around borders - examine carefully
|
|
525
|
+
- {lang_hint}Look for technical markings that might be faint or small
|
|
526
|
+
- Include any numbers that might be measurements or identifiers
|
|
527
|
+
- If absolutely no readable text is visible, respond with 'NO_TEXT_FOUND'
|
|
528
|
+
|
|
529
|
+
Extracted text:"""
|
|
530
|
+
elif config.get('context') == 'medical':
|
|
531
|
+
prompt = f"""This appears to be a medical document or image. Look carefully and extract ALL text, including:
|
|
532
|
+
- Patient information (names, IDs, dates of birth)
|
|
533
|
+
- Medical terminology and diagnostic information
|
|
534
|
+
- Dates, times, and timestamps
|
|
535
|
+
- Measurements, values, and test results
|
|
536
|
+
- Doctor names, hospital information, department names
|
|
537
|
+
- Small print and technical annotations
|
|
538
|
+
|
|
539
|
+
IMPORTANT:
|
|
540
|
+
- Medical documents often contain critical small text - examine thoroughly
|
|
541
|
+
- {lang_hint}Include all numerical values as they may be measurements
|
|
542
|
+
- Preserve formatting for medical data accuracy
|
|
543
|
+
- If no readable text is found, respond with 'NO_TEXT_FOUND'
|
|
544
|
+
|
|
545
|
+
Extracted text:"""
|
|
546
|
+
else:
|
|
547
|
+
# General OCR prompt
|
|
548
|
+
prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
|
|
280
549
|
- Titles, headings, and main text content
|
|
281
550
|
- Small print, captions, labels, and annotations
|
|
282
551
|
- Numbers, measurements, quantities, and symbols
|
|
@@ -329,6 +598,17 @@ Extracted text:"""
|
|
|
329
598
|
context_hint = """
|
|
330
599
|
- Focus on technical elements: labels, measurements, specifications, diagrams
|
|
331
600
|
- Include any mathematical formulas, technical symbols, or engineering notations"""
|
|
601
|
+
elif context == 'medical':
|
|
602
|
+
context_hint = """
|
|
603
|
+
- Focus on medical content: patient data, measurements, anatomical labels, medical terminology
|
|
604
|
+
- Look for dates, patient IDs, measurement values, diagnostic information
|
|
605
|
+
- Note any visible text on medical equipment or instrumentation"""
|
|
606
|
+
elif context == 'xray':
|
|
607
|
+
context_hint = """
|
|
608
|
+
- This appears to be a medical X-ray or radiological image
|
|
609
|
+
- Look for: anatomical markers, measurement scales, patient information, timestamps
|
|
610
|
+
- Focus on any visible text annotations, labels, or technical markings
|
|
611
|
+
- Note positioning indicators (L/R, anterior/posterior) or measurement rulers"""
|
|
332
612
|
|
|
333
613
|
prompt = f"""Analyze this image and provide:
|
|
334
614
|
1. All visible text exactly as written (preserve formatting, line breaks, bullet points)
|
|
@@ -341,23 +621,31 @@ Format:
|
|
|
341
621
|
TEXT: [all visible text here, or NO_TEXT_FOUND if none]
|
|
342
622
|
DESCRIPTION: [image description following the guidelines above]"""
|
|
343
623
|
|
|
344
|
-
#
|
|
345
|
-
|
|
346
|
-
model=current_model,
|
|
347
|
-
prompt=prompt,
|
|
348
|
-
images=[img_base64],
|
|
349
|
-
options={
|
|
350
|
-
'temperature': config['temperature'],
|
|
351
|
-
'top_p': 0.9,
|
|
352
|
-
'num_predict': config['max_tokens']
|
|
353
|
-
}
|
|
354
|
-
)
|
|
355
|
-
|
|
356
|
-
# Extract and clean response
|
|
357
|
-
extracted_content = response.get('response', '').strip()
|
|
624
|
+
# Try primary model first
|
|
625
|
+
success, extracted_content, confidence_score = _try_ollama_request(prompt, img_base64, current_model, config)
|
|
358
626
|
|
|
359
|
-
#
|
|
360
|
-
|
|
627
|
+
# If primary model failed or confidence is very low, try fallback models
|
|
628
|
+
if config.get('auto_fallback', True) and (not success or confidence_score < 0.3):
|
|
629
|
+
print("🔄 Primary model result unsatisfactory, trying fallback models...")
|
|
630
|
+
|
|
631
|
+
fallback_models = config.get('fallback_models', [])
|
|
632
|
+
best_result = (extracted_content, confidence_score) if success else ("", 0.0)
|
|
633
|
+
|
|
634
|
+
for fallback_model in fallback_models:
|
|
635
|
+
if fallback_model == current_model:
|
|
636
|
+
continue # Skip if same as primary
|
|
637
|
+
|
|
638
|
+
success, fallback_content, fallback_confidence = _try_ollama_request(prompt, img_base64, fallback_model, config)
|
|
639
|
+
|
|
640
|
+
if success and fallback_confidence > best_result[1]:
|
|
641
|
+
print(f"✨ Better result from {fallback_model} (confidence: {fallback_confidence:.2f} vs {best_result[1]:.2f})")
|
|
642
|
+
best_result = (fallback_content, fallback_confidence)
|
|
643
|
+
|
|
644
|
+
# Stop if we found a good enough result
|
|
645
|
+
if fallback_confidence >= config['confidence_threshold']:
|
|
646
|
+
break
|
|
647
|
+
|
|
648
|
+
extracted_content, confidence_score = best_result
|
|
361
649
|
|
|
362
650
|
# Check confidence threshold
|
|
363
651
|
if confidence_score < config['confidence_threshold']:
|
|
@@ -401,14 +689,19 @@ DESCRIPTION: [image description following the guidelines above]"""
|
|
|
401
689
|
if image.mode != 'RGB':
|
|
402
690
|
image = image.convert('RGB')
|
|
403
691
|
|
|
692
|
+
# Get current configuration
|
|
693
|
+
config = OLLAMA_CONFIG
|
|
694
|
+
|
|
695
|
+
# Apply image enhancement if enabled
|
|
696
|
+
if config.get('enhance_image', True):
|
|
697
|
+
print("⚡ Enhancing image for better OCR...")
|
|
698
|
+
image = _enhance_image(image, config['context'], config['min_size'], config['max_size'])
|
|
699
|
+
|
|
404
700
|
# Convert image to base64
|
|
405
701
|
buffered = BytesIO()
|
|
406
702
|
image.save(buffered, format="PNG")
|
|
407
703
|
img_base64 = base64.b64encode(buffered.getvalue()).decode()
|
|
408
704
|
|
|
409
|
-
# Get current configuration
|
|
410
|
-
config = OLLAMA_CONFIG
|
|
411
|
-
|
|
412
705
|
# Build prompts (same logic as main function)
|
|
413
706
|
lang_hint = ""
|
|
414
707
|
if config['language'] != 'auto':
|
|
@@ -457,23 +750,26 @@ Format:
|
|
|
457
750
|
TEXT: [all visible text here, or NO_TEXT_FOUND if none]
|
|
458
751
|
DESCRIPTION: [image description following the guidelines above]"""
|
|
459
752
|
|
|
460
|
-
#
|
|
461
|
-
|
|
462
|
-
model=current_model,
|
|
463
|
-
prompt=prompt,
|
|
464
|
-
images=[img_base64],
|
|
465
|
-
options={
|
|
466
|
-
'temperature': config['temperature'],
|
|
467
|
-
'top_p': 0.9,
|
|
468
|
-
'num_predict': config['max_tokens']
|
|
469
|
-
}
|
|
470
|
-
)
|
|
471
|
-
|
|
472
|
-
# Extract response
|
|
473
|
-
extracted_content = response.get('response', '').strip()
|
|
753
|
+
# Try primary model first
|
|
754
|
+
success, extracted_content, confidence_score = _try_ollama_request(prompt, img_base64, current_model, config)
|
|
474
755
|
|
|
475
|
-
#
|
|
476
|
-
|
|
756
|
+
# Try fallback if enabled and primary result is poor
|
|
757
|
+
if config.get('auto_fallback', True) and (not success or confidence_score < 0.3):
|
|
758
|
+
fallback_models = config.get('fallback_models', [])
|
|
759
|
+
best_result = (extracted_content, confidence_score) if success else ("", 0.0)
|
|
760
|
+
|
|
761
|
+
for fallback_model in fallback_models:
|
|
762
|
+
if fallback_model == current_model:
|
|
763
|
+
continue
|
|
764
|
+
|
|
765
|
+
success, fallback_content, fallback_confidence = _try_ollama_request(prompt, img_base64, fallback_model, config)
|
|
766
|
+
|
|
767
|
+
if success and fallback_confidence > best_result[1]:
|
|
768
|
+
best_result = (fallback_content, fallback_confidence)
|
|
769
|
+
if fallback_confidence >= config['confidence_threshold']:
|
|
770
|
+
break
|
|
771
|
+
|
|
772
|
+
extracted_content, confidence_score = best_result
|
|
477
773
|
|
|
478
774
|
# Return both text and confidence
|
|
479
775
|
return extracted_content, confidence_score
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.3.4
|
|
3
|
+
Version: 0.3.4.2
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -57,7 +57,6 @@ Provides-Extra: html
|
|
|
57
57
|
Requires-Dist: beautifulsoup4; extra == "html"
|
|
58
58
|
Requires-Dist: lxml; extra == "html"
|
|
59
59
|
Provides-Extra: doc
|
|
60
|
-
Requires-Dist: textract; extra == "doc"
|
|
61
60
|
Provides-Extra: markdown
|
|
62
61
|
Requires-Dist: markdown; extra == "markdown"
|
|
63
62
|
Requires-Dist: beautifulsoup4; extra == "markdown"
|
|
@@ -82,7 +81,6 @@ Provides-Extra: ocr-ollama
|
|
|
82
81
|
Requires-Dist: ollama; extra == "ocr-ollama"
|
|
83
82
|
Requires-Dist: pillow; extra == "ocr-ollama"
|
|
84
83
|
Provides-Extra: all
|
|
85
|
-
Requires-Dist: textract; extra == "all"
|
|
86
84
|
Requires-Dist: PyMuPDF; extra == "all"
|
|
87
85
|
Requires-Dist: python-docx; extra == "all"
|
|
88
86
|
Requires-Dist: python-pptx; extra == "all"
|
|
@@ -196,6 +194,33 @@ Use python-magic-bin instead of python-magic for easier installation.
|
|
|
196
194
|
|
|
197
195
|
Dependencies are automatically installed based on selected optional groups.
|
|
198
196
|
|
|
197
|
+
### System Dependencies
|
|
198
|
+
Some extractors require system-level tools to be installed:
|
|
199
|
+
|
|
200
|
+
- **Legacy DOC files**: `antiword` - Install via your package manager:
|
|
201
|
+
```bash
|
|
202
|
+
# Ubuntu/Debian
|
|
203
|
+
sudo apt install antiword
|
|
204
|
+
|
|
205
|
+
# macOS
|
|
206
|
+
brew install antiword
|
|
207
|
+
|
|
208
|
+
# CentOS/RHEL
|
|
209
|
+
sudo yum install antiword
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
- **Audio/Video transcription**: `ffmpeg` - Required for audio preprocessing:
|
|
213
|
+
```bash
|
|
214
|
+
# Ubuntu/Debian
|
|
215
|
+
sudo apt install ffmpeg
|
|
216
|
+
|
|
217
|
+
# macOS
|
|
218
|
+
brew install ffmpeg
|
|
219
|
+
|
|
220
|
+
# Windows
|
|
221
|
+
# Download from https://ffmpeg.org/download.html
|
|
222
|
+
```
|
|
223
|
+
|
|
199
224
|
## 📚 Usage Examples
|
|
200
225
|
|
|
201
226
|
### Basic Usage
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|