pyxtxt 0.3.4__tar.gz → 0.3.4.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. {pyxtxt-0.3.4/src/pyxtxt.egg-info → pyxtxt-0.3.4.2}/PKG-INFO +28 -3
  2. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/README.md +27 -0
  3. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/pyproject.toml +1 -3
  4. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/__init__.py +4 -2
  5. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/ocr_ollama.py +338 -42
  6. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2/src/pyxtxt.egg-info}/PKG-INFO +28 -3
  7. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt.egg-info/requires.txt +0 -2
  8. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/LICENSE +0 -0
  9. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/MANIFEST.in +0 -0
  10. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/setup.cfg +0 -0
  11. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/core.py +0 -0
  12. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/__init__.py +0 -0
  13. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/audio.py +0 -0
  14. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/doc.py +0 -0
  15. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/docx.py +0 -0
  16. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/eml.py +0 -0
  17. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/epub.py +0 -0
  18. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/html.py +0 -0
  19. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/md.py +0 -0
  20. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/msg.py +0 -0
  21. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/ocr.py +0 -0
  22. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/odt.py +0 -0
  23. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/pdf.py +0 -0
  24. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/pptx.py +0 -0
  25. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/rtf.py +0 -0
  26. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/svg.py +0 -0
  27. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/tex.py +0 -0
  28. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/txt.py +0 -0
  29. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/xls.py +0 -0
  30. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/xlsx.py +0 -0
  31. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/estrattori/xml.py +0 -0
  32. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/examples.py +0 -0
  33. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt/pyxtxt.py +0 -0
  34. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
  35. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  36. {pyxtxt-0.3.4 → pyxtxt-0.3.4.2}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.4
3
+ Version: 0.3.4.2
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -57,7 +57,6 @@ Provides-Extra: html
57
57
  Requires-Dist: beautifulsoup4; extra == "html"
58
58
  Requires-Dist: lxml; extra == "html"
59
59
  Provides-Extra: doc
60
- Requires-Dist: textract; extra == "doc"
61
60
  Provides-Extra: markdown
62
61
  Requires-Dist: markdown; extra == "markdown"
63
62
  Requires-Dist: beautifulsoup4; extra == "markdown"
@@ -82,7 +81,6 @@ Provides-Extra: ocr-ollama
82
81
  Requires-Dist: ollama; extra == "ocr-ollama"
83
82
  Requires-Dist: pillow; extra == "ocr-ollama"
84
83
  Provides-Extra: all
85
- Requires-Dist: textract; extra == "all"
86
84
  Requires-Dist: PyMuPDF; extra == "all"
87
85
  Requires-Dist: python-docx; extra == "all"
88
86
  Requires-Dist: python-pptx; extra == "all"
@@ -196,6 +194,33 @@ Use python-magic-bin instead of python-magic for easier installation.
196
194
 
197
195
  Dependencies are automatically installed based on selected optional groups.
198
196
 
197
+ ### System Dependencies
198
+ Some extractors require system-level tools to be installed:
199
+
200
+ - **Legacy DOC files**: `antiword` - Install via your package manager:
201
+ ```bash
202
+ # Ubuntu/Debian
203
+ sudo apt install antiword
204
+
205
+ # macOS
206
+ brew install antiword
207
+
208
+ # CentOS/RHEL
209
+ sudo yum install antiword
210
+ ```
211
+
212
+ - **Audio/Video transcription**: `ffmpeg` - Required for audio preprocessing:
213
+ ```bash
214
+ # Ubuntu/Debian
215
+ sudo apt install ffmpeg
216
+
217
+ # macOS
218
+ brew install ffmpeg
219
+
220
+ # Windows
221
+ # Download from https://ffmpeg.org/download.html
222
+ ```
223
+
199
224
  ## 📚 Usage Examples
200
225
 
201
226
  ### Basic Usage
@@ -92,6 +92,33 @@ Use python-magic-bin instead of python-magic for easier installation.
92
92
 
93
93
  Dependencies are automatically installed based on selected optional groups.
94
94
 
95
+ ### System Dependencies
96
+ Some extractors require system-level tools to be installed:
97
+
98
+ - **Legacy DOC files**: `antiword` - Install via your package manager:
99
+ ```bash
100
+ # Ubuntu/Debian
101
+ sudo apt install antiword
102
+
103
+ # macOS
104
+ brew install antiword
105
+
106
+ # CentOS/RHEL
107
+ sudo yum install antiword
108
+ ```
109
+
110
+ - **Audio/Video transcription**: `ffmpeg` - Required for audio preprocessing:
111
+ ```bash
112
+ # Ubuntu/Debian
113
+ sudo apt install ffmpeg
114
+
115
+ # macOS
116
+ brew install ffmpeg
117
+
118
+ # Windows
119
+ # Download from https://ffmpeg.org/download.html
120
+ ```
121
+
95
122
  ## 📚 Usage Examples
96
123
 
97
124
  ### Basic Usage
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.3.4"
3
+ version = "0.3.4.2"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -50,7 +50,6 @@ html = [
50
50
  "lxml",
51
51
  ]
52
52
  doc = [
53
- "textract",
54
53
  ]
55
54
  markdown = [
56
55
  "markdown",
@@ -85,7 +84,6 @@ ocr-ollama = [
85
84
  "pillow",
86
85
  ]
87
86
  all = [
88
- "textract",
89
87
  "PyMuPDF",
90
88
  "python-docx",
91
89
  "python-pptx",
@@ -5,13 +5,15 @@ try:
5
5
  from .estrattori.ocr_ollama import (
6
6
  set_ollama_model, get_ollama_model, xtxt_image_describe,
7
7
  set_ollama_config, get_ollama_config, reset_ollama_config,
8
- xtxt_image_with_confidence
8
+ xtxt_image_with_confidence, configure_for_medical_images,
9
+ configure_for_xray_images, quick_xray_analysis
9
10
  )
10
11
  __all__ = [
11
12
  "xtxt", "extxt_available_formats", "xtxt_from_url",
12
13
  "set_ollama_model", "get_ollama_model", "xtxt_image_describe",
13
14
  "set_ollama_config", "get_ollama_config", "reset_ollama_config",
14
- "xtxt_image_with_confidence"
15
+ "xtxt_image_with_confidence", "configure_for_medical_images",
16
+ "configure_for_xray_images", "quick_xray_analysis"
15
17
  ]
16
18
  except ImportError:
17
19
  __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
@@ -19,7 +19,12 @@ OLLAMA_CONFIG = {
19
19
  'temperature': 0.1, # Response creativity (0.0-1.0)
20
20
  'max_tokens': 1500, # Maximum response length
21
21
  'confidence_threshold': 0.7, # Minimum confidence for text extraction
22
- 'context': 'general' # Context hint: general, cookbook, document, diagram, etc.
22
+ 'context': 'general', # Context hint: general, cookbook, document, diagram, medical, xray
23
+ 'enhance_image': True, # Apply image enhancement preprocessing
24
+ 'min_size': 800, # Minimum image size for processing (upscale if smaller)
25
+ 'max_size': 2048, # Maximum image size (downscale if larger)
26
+ 'auto_fallback': True, # Enable automatic model fallback for better results
27
+ 'fallback_models': ['gemma3:27b', 'gemma3:12b', 'llava:13b'] # Models to try if primary fails
23
28
  }
24
29
 
25
30
  def set_ollama_model(model_name: str):
@@ -49,7 +54,12 @@ def set_ollama_config(**kwargs):
49
54
  - language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
50
55
  - caption_length: Caption length ('short', 'medium', 'long')
51
56
  - style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
52
- - context: Content context ('general', 'cookbook', 'document', 'handwriting', 'technical')
57
+ - context: Content context ('general', 'cookbook', 'document', 'handwriting', 'technical', 'medical', 'xray')
58
+ - enhance_image: Apply image enhancement preprocessing (default: True)
59
+ - min_size: Minimum image size for processing - upscale if smaller (default: 800)
60
+ - max_size: Maximum image size - downscale if larger (default: 2048)
61
+ - auto_fallback: Enable automatic model fallback for better results (default: True)
62
+ - fallback_models: List of models to try if primary fails (default: ['gemma3:27b', 'gemma3:12b', 'llava:13b'])
53
63
  - temperature: Response creativity 0.0-1.0 (default: 0.1)
54
64
  - max_tokens: Maximum response length (default: 1500)
55
65
  - confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
@@ -58,6 +68,10 @@ def set_ollama_config(**kwargs):
58
68
  set_ollama_config(language='italian', style='detailed')
59
69
  set_ollama_config(context='document', caption_length='long')
60
70
  set_ollama_config(context='handwriting', temperature=0.2)
71
+ set_ollama_config(context='medical', enhance_image=True, min_size=1024)
72
+ set_ollama_config(context='xray', style='technical', confidence_threshold=0.8)
73
+ set_ollama_config(auto_fallback=False) # Disable fallback for faster processing
74
+ set_ollama_config(fallback_models=['llava:13b', 'gemma3:27b']) # Custom fallback sequence
61
75
  """
62
76
  global OLLAMA_CONFIG
63
77
  for key, value in kwargs.items():
@@ -71,6 +85,105 @@ def get_ollama_config():
71
85
  """Get current Ollama configuration"""
72
86
  return OLLAMA_CONFIG.copy()
73
87
 
88
+ def configure_for_medical_images():
89
+ """
90
+ Quick configuration setup for optimal medical image processing.
91
+ Configures model, enhancement, and context for X-rays and medical scans.
92
+ """
93
+ print("🏥 Configuring for medical image processing...")
94
+
95
+ # Optimal medical settings
96
+ global OLLAMA_CONFIG
97
+ OLLAMA_CONFIG.update({
98
+ 'context': 'medical',
99
+ 'enhance_image': True,
100
+ 'min_size': 1024, # Higher resolution for medical details
101
+ 'max_size': 2048, # Keep high quality
102
+ 'style': 'technical',
103
+ 'confidence_threshold': 0.8, # Higher threshold for medical accuracy
104
+ 'temperature': 0.05, # Very low temperature for precision
105
+ 'auto_fallback': True,
106
+ 'fallback_models': ['gemma3:27b', 'llava:13b', 'gemma3:12b'] # Best models first
107
+ })
108
+
109
+ # Suggest high-quality model if current is default
110
+ current_model = get_ollama_model()
111
+ if current_model == "gemma3:4b":
112
+ print("💡 Consider upgrading to gemma3:27b for better medical image recognition")
113
+ print(" Run: set_ollama_model('gemma3:27b')")
114
+
115
+ print("✅ Medical configuration applied:")
116
+ print(f" - Enhanced image processing: {OLLAMA_CONFIG['enhance_image']}")
117
+ print(f" - Minimum resolution: {OLLAMA_CONFIG['min_size']}px")
118
+ print(f" - Confidence threshold: {OLLAMA_CONFIG['confidence_threshold']}")
119
+ print(f" - Fallback models: {len(OLLAMA_CONFIG['fallback_models'])} configured")
120
+
121
+ def configure_for_xray_images():
122
+ """
123
+ Specialized configuration for X-ray and radiological image processing.
124
+ Optimized for detecting small text, markers, and technical annotations.
125
+ """
126
+ print("📷 Configuring for X-ray image processing...")
127
+
128
+ # X-ray specific settings
129
+ global OLLAMA_CONFIG
130
+ OLLAMA_CONFIG.update({
131
+ 'context': 'xray',
132
+ 'enhance_image': True,
133
+ 'min_size': 1200, # Even higher for X-ray details
134
+ 'max_size': 2048,
135
+ 'style': 'technical',
136
+ 'confidence_threshold': 0.85, # Very high threshold for X-ray accuracy
137
+ 'temperature': 0.02, # Minimal creativity for technical precision
138
+ 'auto_fallback': True,
139
+ 'fallback_models': ['gemma3:27b', 'llava:13b'] # Only best models
140
+ })
141
+
142
+ # Recommend best model for X-rays
143
+ current_model = get_ollama_model()
144
+ if current_model != "gemma3:27b":
145
+ print("🎯 For best X-ray results, use gemma3:27b model")
146
+ print(" Run: set_ollama_model('gemma3:27b')")
147
+
148
+ print("✅ X-ray configuration applied:")
149
+ print(f" - Context: Medical X-ray specialization")
150
+ print(f" - Enhanced processing: Advanced medical filters")
151
+ print(f" - High precision mode: {OLLAMA_CONFIG['confidence_threshold']} threshold")
152
+ print(f" - Temperature: {OLLAMA_CONFIG['temperature']} (maximum precision)")
153
+
154
+ def quick_xray_analysis(image_path: str) -> str:
155
+ """
156
+ Convenient one-function X-ray analysis with optimal settings.
157
+ Automatically configures system for X-ray processing and returns detailed analysis.
158
+
159
+ Args:
160
+ image_path: Path to X-ray image file
161
+
162
+ Returns:
163
+ str: Detailed text + description analysis
164
+ """
165
+ # Save current config
166
+ global OLLAMA_CONFIG
167
+ original_config = OLLAMA_CONFIG.copy()
168
+ original_model = get_ollama_model()
169
+
170
+ try:
171
+ # Apply X-ray configuration
172
+ configure_for_xray_images()
173
+ set_ollama_model('gemma3:27b') # Use best model
174
+
175
+ # Perform analysis
176
+ print("🔍 Analyzing X-ray image...")
177
+ result = xtxt_image_describe(image_path)
178
+
179
+ return result
180
+
181
+ finally:
182
+ # Restore original configuration
183
+ OLLAMA_CONFIG = original_config
184
+ set_ollama_model(original_model)
185
+ print("🔄 Configuration restored")
186
+
74
187
  def reset_ollama_config():
75
188
  """Reset Ollama configuration to defaults"""
76
189
  global OLLAMA_CONFIG
@@ -81,10 +194,125 @@ def reset_ollama_config():
81
194
  'temperature': 0.1,
82
195
  'max_tokens': 1500,
83
196
  'confidence_threshold': 0.7,
84
- 'context': 'general'
197
+ 'context': 'general',
198
+ 'enhance_image': True,
199
+ 'min_size': 800,
200
+ 'max_size': 2048,
201
+ 'auto_fallback': True,
202
+ 'fallback_models': ['gemma3:27b', 'gemma3:12b', 'llava:13b']
85
203
  }
86
204
  print("✅ Ollama configuration reset to defaults")
87
205
 
206
+ def _enhance_image(image: Image.Image, context: str, min_size: int, max_size: int) -> Image.Image:
207
+ """
208
+ Apply image enhancement preprocessing for better OCR results.
209
+
210
+ Args:
211
+ image: PIL Image object
212
+ context: Content context hint (medical, xray, document, etc.)
213
+ min_size: Minimum size threshold for upscaling
214
+ max_size: Maximum size threshold for downscaling
215
+
216
+ Returns:
217
+ Enhanced PIL Image
218
+ """
219
+ try:
220
+ from PIL import ImageEnhance, ImageFilter, ImageOps
221
+ except ImportError:
222
+ # If PIL enhancements not available, return original
223
+ return image
224
+
225
+ enhanced = image.copy()
226
+
227
+ # Get current dimensions
228
+ width, height = enhanced.size
229
+ max_dimension = max(width, height)
230
+
231
+ # Resize if needed (quality improvement for small images, memory management for large)
232
+ if max_dimension < min_size:
233
+ # Upscale small images for better model processing
234
+ scale_factor = min_size / max_dimension
235
+ new_width = int(width * scale_factor)
236
+ new_height = int(height * scale_factor)
237
+ enhanced = enhanced.resize((new_width, new_height), Image.Resampling.LANCZOS)
238
+ print(f"📈 Image upscaled from {width}x{height} to {new_width}x{new_height}")
239
+ elif max_dimension > max_size:
240
+ # Downscale large images to manageable size
241
+ scale_factor = max_size / max_dimension
242
+ new_width = int(width * scale_factor)
243
+ new_height = int(height * scale_factor)
244
+ enhanced = enhanced.resize((new_width, new_height), Image.Resampling.LANCZOS)
245
+ print(f"📉 Image downscaled from {width}x{height} to {new_width}x{new_height}")
246
+
247
+ # Context-specific enhancements
248
+ if context in ['medical', 'xray']:
249
+ print("🏥 Applying medical image enhancements...")
250
+ # Medical images often benefit from:
251
+ # 1. Contrast enhancement to bring out subtle details
252
+ contrast = ImageEnhance.Contrast(enhanced)
253
+ enhanced = contrast.enhance(1.3)
254
+
255
+ # 2. Sharpening to improve edge definition
256
+ enhanced = enhanced.filter(ImageFilter.UnsharpMask(radius=1, percent=120, threshold=3))
257
+
258
+ # 3. Brightness adjustment for dark X-rays
259
+ brightness = ImageEnhance.Brightness(enhanced)
260
+ enhanced = brightness.enhance(1.1)
261
+
262
+ elif context in ['document', 'handwriting', 'technical']:
263
+ print("📄 Applying document enhancement...")
264
+ # Documents benefit from:
265
+ # 1. Moderate sharpening for text clarity
266
+ enhanced = enhanced.filter(ImageFilter.UnsharpMask(radius=0.5, percent=150, threshold=3))
267
+
268
+ # 2. Contrast boost for faded text
269
+ contrast = ImageEnhance.Contrast(enhanced)
270
+ enhanced = contrast.enhance(1.2)
271
+
272
+ elif context == 'general':
273
+ print("🔧 Applying general enhancements...")
274
+ # General purpose light enhancement
275
+ # 1. Slight contrast improvement
276
+ contrast = ImageEnhance.Contrast(enhanced)
277
+ enhanced = contrast.enhance(1.1)
278
+
279
+ # 2. Subtle sharpening
280
+ enhanced = enhanced.filter(ImageFilter.UnsharpMask(radius=0.5, percent=110, threshold=3))
281
+
282
+ return enhanced
283
+
284
+ def _try_ollama_request(prompt: str, img_base64: str, current_model: str, config: dict) -> tuple:
285
+ """
286
+ Attempt Ollama request with a specific model.
287
+
288
+ Returns:
289
+ tuple: (success: bool, response: str, confidence: float)
290
+ """
291
+ try:
292
+ print(f"🤖 Trying model: {current_model}")
293
+
294
+ response = ollama.generate(
295
+ model=current_model,
296
+ prompt=prompt,
297
+ images=[img_base64],
298
+ options={
299
+ 'temperature': config['temperature'],
300
+ 'top_p': 0.9,
301
+ 'num_predict': config['max_tokens']
302
+ }
303
+ )
304
+
305
+ extracted_content = response.get('response', '').strip()
306
+
307
+ # Calculate confidence without printing warnings here (let parent handle it)
308
+ confidence_score = _calculate_confidence_score(extracted_content, "ocr" if "Extracted text:" in prompt else "describe")
309
+
310
+ return True, extracted_content, confidence_score
311
+
312
+ except Exception as e:
313
+ print(f"❌ Model {current_model} failed: {e}")
314
+ return False, "", 0.0
315
+
88
316
  def _calculate_confidence_score(content: str, mode: str) -> float:
89
317
  """
90
318
  Calculate confidence score (0.0-1.0) for OCR/caption quality.
@@ -261,14 +489,19 @@ if ollama and Image:
261
489
  if image.mode != 'RGB':
262
490
  image = image.convert('RGB')
263
491
 
492
+ # Get current configuration
493
+ config = OLLAMA_CONFIG
494
+
495
+ # Apply image enhancement if enabled
496
+ if config.get('enhance_image', True):
497
+ print("⚡ Enhancing image for better OCR...")
498
+ image = _enhance_image(image, config['context'], config['min_size'], config['max_size'])
499
+
264
500
  # Convert image to base64
265
501
  buffered = BytesIO()
266
502
  image.save(buffered, format="PNG")
267
503
  img_base64 = base64.b64encode(buffered.getvalue()).decode()
268
504
 
269
- # Get current configuration
270
- config = OLLAMA_CONFIG
271
-
272
505
  # Build language hint
273
506
  lang_hint = ""
274
507
  if config['language'] != 'auto':
@@ -276,7 +509,43 @@ if ollama and Image:
276
509
 
277
510
  # Different prompts based on mode
278
511
  if mode == "ocr":
279
- prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
512
+ # Context-specific OCR prompts
513
+ if config.get('context') == 'xray':
514
+ prompt = f"""This is a medical X-ray or radiological image. Look carefully and extract ALL visible text, including:
515
+ - Patient identification numbers, names, or codes
516
+ - Date and time stamps (exam dates, birth dates)
517
+ - Anatomical position markers (L/R, LEFT/RIGHT, AP, LAT)
518
+ - Measurement scales, rulers, or calibration marks
519
+ - Technical annotations or radiologist markings
520
+ - Equipment identifiers or hospital names
521
+ - Any small text on borders or corners
522
+
523
+ IMPORTANT:
524
+ - Medical images often have small text around borders - examine carefully
525
+ - {lang_hint}Look for technical markings that might be faint or small
526
+ - Include any numbers that might be measurements or identifiers
527
+ - If absolutely no readable text is visible, respond with 'NO_TEXT_FOUND'
528
+
529
+ Extracted text:"""
530
+ elif config.get('context') == 'medical':
531
+ prompt = f"""This appears to be a medical document or image. Look carefully and extract ALL text, including:
532
+ - Patient information (names, IDs, dates of birth)
533
+ - Medical terminology and diagnostic information
534
+ - Dates, times, and timestamps
535
+ - Measurements, values, and test results
536
+ - Doctor names, hospital information, department names
537
+ - Small print and technical annotations
538
+
539
+ IMPORTANT:
540
+ - Medical documents often contain critical small text - examine thoroughly
541
+ - {lang_hint}Include all numerical values as they may be measurements
542
+ - Preserve formatting for medical data accuracy
543
+ - If no readable text is found, respond with 'NO_TEXT_FOUND'
544
+
545
+ Extracted text:"""
546
+ else:
547
+ # General OCR prompt
548
+ prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
280
549
  - Titles, headings, and main text content
281
550
  - Small print, captions, labels, and annotations
282
551
  - Numbers, measurements, quantities, and symbols
@@ -329,6 +598,17 @@ Extracted text:"""
329
598
  context_hint = """
330
599
  - Focus on technical elements: labels, measurements, specifications, diagrams
331
600
  - Include any mathematical formulas, technical symbols, or engineering notations"""
601
+ elif context == 'medical':
602
+ context_hint = """
603
+ - Focus on medical content: patient data, measurements, anatomical labels, medical terminology
604
+ - Look for dates, patient IDs, measurement values, diagnostic information
605
+ - Note any visible text on medical equipment or instrumentation"""
606
+ elif context == 'xray':
607
+ context_hint = """
608
+ - This appears to be a medical X-ray or radiological image
609
+ - Look for: anatomical markers, measurement scales, patient information, timestamps
610
+ - Focus on any visible text annotations, labels, or technical markings
611
+ - Note positioning indicators (L/R, anterior/posterior) or measurement rulers"""
332
612
 
333
613
  prompt = f"""Analyze this image and provide:
334
614
  1. All visible text exactly as written (preserve formatting, line breaks, bullet points)
@@ -341,23 +621,31 @@ Format:
341
621
  TEXT: [all visible text here, or NO_TEXT_FOUND if none]
342
622
  DESCRIPTION: [image description following the guidelines above]"""
343
623
 
344
- # Send request to Ollama with configured parameters
345
- response = ollama.generate(
346
- model=current_model,
347
- prompt=prompt,
348
- images=[img_base64],
349
- options={
350
- 'temperature': config['temperature'],
351
- 'top_p': 0.9,
352
- 'num_predict': config['max_tokens']
353
- }
354
- )
355
-
356
- # Extract and clean response
357
- extracted_content = response.get('response', '').strip()
624
+ # Try primary model first
625
+ success, extracted_content, confidence_score = _try_ollama_request(prompt, img_base64, current_model, config)
358
626
 
359
- # Calculate confidence score based on response quality indicators
360
- confidence_score = _calculate_confidence_score(extracted_content, mode)
627
+ # If primary model failed or confidence is very low, try fallback models
628
+ if config.get('auto_fallback', True) and (not success or confidence_score < 0.3):
629
+ print("🔄 Primary model result unsatisfactory, trying fallback models...")
630
+
631
+ fallback_models = config.get('fallback_models', [])
632
+ best_result = (extracted_content, confidence_score) if success else ("", 0.0)
633
+
634
+ for fallback_model in fallback_models:
635
+ if fallback_model == current_model:
636
+ continue # Skip if same as primary
637
+
638
+ success, fallback_content, fallback_confidence = _try_ollama_request(prompt, img_base64, fallback_model, config)
639
+
640
+ if success and fallback_confidence > best_result[1]:
641
+ print(f"✨ Better result from {fallback_model} (confidence: {fallback_confidence:.2f} vs {best_result[1]:.2f})")
642
+ best_result = (fallback_content, fallback_confidence)
643
+
644
+ # Stop if we found a good enough result
645
+ if fallback_confidence >= config['confidence_threshold']:
646
+ break
647
+
648
+ extracted_content, confidence_score = best_result
361
649
 
362
650
  # Check confidence threshold
363
651
  if confidence_score < config['confidence_threshold']:
@@ -401,14 +689,19 @@ DESCRIPTION: [image description following the guidelines above]"""
401
689
  if image.mode != 'RGB':
402
690
  image = image.convert('RGB')
403
691
 
692
+ # Get current configuration
693
+ config = OLLAMA_CONFIG
694
+
695
+ # Apply image enhancement if enabled
696
+ if config.get('enhance_image', True):
697
+ print("⚡ Enhancing image for better OCR...")
698
+ image = _enhance_image(image, config['context'], config['min_size'], config['max_size'])
699
+
404
700
  # Convert image to base64
405
701
  buffered = BytesIO()
406
702
  image.save(buffered, format="PNG")
407
703
  img_base64 = base64.b64encode(buffered.getvalue()).decode()
408
704
 
409
- # Get current configuration
410
- config = OLLAMA_CONFIG
411
-
412
705
  # Build prompts (same logic as main function)
413
706
  lang_hint = ""
414
707
  if config['language'] != 'auto':
@@ -457,23 +750,26 @@ Format:
457
750
  TEXT: [all visible text here, or NO_TEXT_FOUND if none]
458
751
  DESCRIPTION: [image description following the guidelines above]"""
459
752
 
460
- # Send request to Ollama
461
- response = ollama.generate(
462
- model=current_model,
463
- prompt=prompt,
464
- images=[img_base64],
465
- options={
466
- 'temperature': config['temperature'],
467
- 'top_p': 0.9,
468
- 'num_predict': config['max_tokens']
469
- }
470
- )
471
-
472
- # Extract response
473
- extracted_content = response.get('response', '').strip()
753
+ # Try primary model first
754
+ success, extracted_content, confidence_score = _try_ollama_request(prompt, img_base64, current_model, config)
474
755
 
475
- # Calculate confidence without threshold filtering
476
- confidence_score = _calculate_confidence_score(extracted_content, mode)
756
+ # Try fallback if enabled and primary result is poor
757
+ if config.get('auto_fallback', True) and (not success or confidence_score < 0.3):
758
+ fallback_models = config.get('fallback_models', [])
759
+ best_result = (extracted_content, confidence_score) if success else ("", 0.0)
760
+
761
+ for fallback_model in fallback_models:
762
+ if fallback_model == current_model:
763
+ continue
764
+
765
+ success, fallback_content, fallback_confidence = _try_ollama_request(prompt, img_base64, fallback_model, config)
766
+
767
+ if success and fallback_confidence > best_result[1]:
768
+ best_result = (fallback_content, fallback_confidence)
769
+ if fallback_confidence >= config['confidence_threshold']:
770
+ break
771
+
772
+ extracted_content, confidence_score = best_result
477
773
 
478
774
  # Return both text and confidence
479
775
  return extracted_content, confidence_score
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.4
3
+ Version: 0.3.4.2
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -57,7 +57,6 @@ Provides-Extra: html
57
57
  Requires-Dist: beautifulsoup4; extra == "html"
58
58
  Requires-Dist: lxml; extra == "html"
59
59
  Provides-Extra: doc
60
- Requires-Dist: textract; extra == "doc"
61
60
  Provides-Extra: markdown
62
61
  Requires-Dist: markdown; extra == "markdown"
63
62
  Requires-Dist: beautifulsoup4; extra == "markdown"
@@ -82,7 +81,6 @@ Provides-Extra: ocr-ollama
82
81
  Requires-Dist: ollama; extra == "ocr-ollama"
83
82
  Requires-Dist: pillow; extra == "ocr-ollama"
84
83
  Provides-Extra: all
85
- Requires-Dist: textract; extra == "all"
86
84
  Requires-Dist: PyMuPDF; extra == "all"
87
85
  Requires-Dist: python-docx; extra == "all"
88
86
  Requires-Dist: python-pptx; extra == "all"
@@ -196,6 +194,33 @@ Use python-magic-bin instead of python-magic for easier installation.
196
194
 
197
195
  Dependencies are automatically installed based on selected optional groups.
198
196
 
197
+ ### System Dependencies
198
+ Some extractors require system-level tools to be installed:
199
+
200
+ - **Legacy DOC files**: `antiword` - Install via your package manager:
201
+ ```bash
202
+ # Ubuntu/Debian
203
+ sudo apt install antiword
204
+
205
+ # macOS
206
+ brew install antiword
207
+
208
+ # CentOS/RHEL
209
+ sudo yum install antiword
210
+ ```
211
+
212
+ - **Audio/Video transcription**: `ffmpeg` - Required for audio preprocessing:
213
+ ```bash
214
+ # Ubuntu/Debian
215
+ sudo apt install ffmpeg
216
+
217
+ # macOS
218
+ brew install ffmpeg
219
+
220
+ # Windows
221
+ # Download from https://ffmpeg.org/download.html
222
+ ```
223
+
199
224
  ## 📚 Usage Examples
200
225
 
201
226
  ### Basic Usage
@@ -6,7 +6,6 @@ python-magic
6
6
  python-magic-bin
7
7
 
8
8
  [all]
9
- textract
10
9
  PyMuPDF
11
10
  python-docx
12
11
  python-pptx
@@ -29,7 +28,6 @@ ollama
29
28
  openai-whisper
30
29
 
31
30
  [doc]
32
- textract
33
31
 
34
32
  [docx]
35
33
  python-docx
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes