pyxtxt 0.3.2__tar.gz → 0.3.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {pyxtxt-0.3.2/src/pyxtxt.egg-info → pyxtxt-0.3.4}/PKG-INFO +89 -1
  2. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/README.md +88 -0
  3. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/pyproject.toml +1 -1
  4. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/__init__.py +4 -2
  5. pyxtxt-0.3.4/src/pyxtxt/estrattori/ocr_ollama.py +520 -0
  6. {pyxtxt-0.3.2 → pyxtxt-0.3.4/src/pyxtxt.egg-info}/PKG-INFO +89 -1
  7. pyxtxt-0.3.2/src/pyxtxt/estrattori/ocr_ollama.py +0 -202
  8. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/LICENSE +0 -0
  9. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/MANIFEST.in +0 -0
  10. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/setup.cfg +0 -0
  11. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/core.py +0 -0
  12. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/__init__.py +0 -0
  13. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/audio.py +0 -0
  14. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/doc.py +0 -0
  15. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/docx.py +0 -0
  16. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/eml.py +0 -0
  17. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/epub.py +0 -0
  18. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/html.py +0 -0
  19. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/md.py +0 -0
  20. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/msg.py +0 -0
  21. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/ocr.py +0 -0
  22. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/odt.py +0 -0
  23. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/pdf.py +0 -0
  24. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/pptx.py +0 -0
  25. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/rtf.py +0 -0
  26. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/svg.py +0 -0
  27. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/tex.py +0 -0
  28. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/txt.py +0 -0
  29. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/xls.py +0 -0
  30. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/xlsx.py +0 -0
  31. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/xml.py +0 -0
  32. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/examples.py +0 -0
  33. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/pyxtxt.py +0 -0
  34. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
  35. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  36. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt.egg-info/requires.txt +0 -0
  37. {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.2
3
+ Version: 0.3.4
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -308,6 +308,45 @@ image_response = requests.get("https://example.com/document.png")
308
308
  text = xtxt(image_response.content)
309
309
  ```
310
310
 
311
+ ### AI OCR Confidence Scoring (NEW)
312
+
313
+ ⚠️ **IMPORTANT**: AI-powered OCR is experimental technology that may produce errors, hallucinations, or misinterpretations. Always validate results for critical applications.
314
+
315
+ The OCR-Ollama system includes confidence scoring to help identify unreliable results:
316
+
317
+ ```python
318
+ from pyxtxt import xtxt_image_with_confidence, set_ollama_config
319
+
320
+ # Configure confidence threshold (0.0-1.0, default: 0.7)
321
+ set_ollama_config(confidence_threshold=0.8) # More restrictive
322
+
323
+ # Get text with confidence score
324
+ text, confidence = xtxt_image_with_confidence("document.png", mode="ocr")
325
+ print(f"Confidence: {confidence:.2f} ({confidence*100:.1f}%)")
326
+ print(f"Text: {text}")
327
+
328
+ # Check if reliable
329
+ if confidence < 0.7:
330
+ print("⚠️ Low confidence - result may be unreliable")
331
+ print("Consider using traditional OCR or manual verification")
332
+ else:
333
+ print("✅ Good confidence - result likely reliable")
334
+ ```
335
+
336
+ #### Confidence Scoring Features
337
+
338
+ - **Hallucination Detection**: Penalizes historical/cultural references that often indicate AI misinterpretation
339
+ - **Pattern Recognition**: Rewards structured text (numbers, punctuation, proper formatting)
340
+ - **Uncertainty Detection**: Flags vague language ("appears to be", "seems like", etc.)
341
+ - **Quality Assessment**: Considers content length, repetition, and coherence
342
+
343
+ #### Common Hallucination Patterns (Automatically Detected)
344
+ - **Historical Content**: "ancient", "medieval", "Egyptian papyrus", "hieroglyphs"
345
+ - **Artistic Interpretations**: "painting", "artwork", "masterpiece", "Renaissance"
346
+ - **Fantasy Content**: "mystical", "magical", "dragon", "wizard"
347
+ - **Scientific Misinterpretation**: "fossil", "geological formation", "crystal structure"
348
+ - **Vague Language**: "unclear", "difficult to read", "appears to be"
349
+
311
350
  ### Command-Line OCR Example
312
351
 
313
352
  A complete example script for command-line usage is available:
@@ -324,6 +363,11 @@ with open("ocr_example.py", "wb") as f:
324
363
  # python ocr_example.py document.png
325
364
  # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
326
365
  # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
366
+
367
+ # NEW: Confidence scoring examples
368
+ # python ocr_example.py suspicious.png --show-confidence
369
+ # python ocr_example.py medical.png --confidence=0.9 --show-confidence
370
+ # python ocr_example.py diagram.png --confidence=0.5 --mode=describe --show-confidence
327
371
  ```
328
372
 
329
373
  The script supports:
@@ -333,6 +377,8 @@ The script supports:
333
377
  - **Style control**: descriptive, technical, simple, detailed
334
378
  - **Length control**: short, medium, long captions
335
379
  - **Temperature**: Adjust LLM creativity (0.0-1.0)
380
+ - **Confidence scoring**: Set threshold and display confidence scores
381
+ - **Quality filtering**: Automatically reject low-confidence results
336
382
 
337
383
  ### Show Available Formats
338
384
  ```python
@@ -376,6 +422,7 @@ text = xtxt(attachment_bytes)
376
422
 
377
423
  ## ⚠️ Known Limitations
378
424
 
425
+ ### General Limitations
379
426
  - **Legacy file detection**: When using raw streams without filenames, legacy files (.doc, .xls, .ppt) may not be correctly detected due to identical file signatures in libmagic
380
427
  - **Filename hints recommended**: When available, providing original filenames improves detection accuracy
381
428
  - **MSWrite .doc files**: Require `antiword` installation:
@@ -383,6 +430,41 @@ text = xtxt(attachment_bytes)
383
430
  sudo apt-get update && sudo apt-get install antiword
384
431
  ```
385
432
 
433
+ ### 🤖 AI-Powered Features - Important Warnings
434
+
435
+ **⚠️ EXPERIMENTAL TECHNOLOGY**: AI-powered features (OCR-Ollama, audio transcription) are based on machine learning models and may produce:
436
+
437
+ #### Potential Issues:
438
+ - **Hallucinations**: AI may "see" or "hear" content that isn't actually present
439
+ - **Misinterpretations**: Complex images may be incorrectly identified (e.g., X-ray images mistaken for historical artifacts)
440
+ - **Language Errors**: Transcription accuracy depends on audio quality, accents, and background noise
441
+ - **Context Confusion**: AI may apply inappropriate cultural/historical context to technical content
442
+ - **Model Dependence**: Results vary significantly between different AI models (gemma3, llava, whisper versions)
443
+ - **Bias and Inconsistency**: Models may exhibit cultural, linguistic, or domain-specific biases
444
+
445
+ #### Critical Applications Warning:
446
+ **🚨 DO NOT USE for critical applications** such as:
447
+ - Medical diagnosis or medical image interpretation
448
+ - Legal document analysis requiring perfect accuracy
449
+ - Financial data extraction where errors have monetary impact
450
+ - Security/safety systems where false positives/negatives are dangerous
451
+ - Academic research requiring citation-quality accuracy
452
+
453
+ #### Best Practices:
454
+ - **Always validate AI results** against source material when accuracy matters
455
+ - **Use confidence scoring** to identify potentially unreliable results
456
+ - **Cross-reference** with traditional OCR/transcription tools for important content
457
+ - **Human review** recommended for any production use case
458
+ - **Test thoroughly** with your specific content types and use cases
459
+ - **Fallback options**: Keep traditional OCR (EasyOCR) available as backup
460
+
461
+ #### Recommended Use Cases:
462
+ ✅ Content discovery and initial text extraction
463
+ ✅ Batch processing of low-stakes content
464
+ ✅ Development and prototyping workflows
465
+ ✅ Personal document organization
466
+ ✅ Educational and learning projects
467
+
386
468
  ## 📖 Full Examples
387
469
 
388
470
  ### Accessing Examples After Installation
@@ -427,7 +509,13 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
427
509
  - ✅ **NEW**: Caption length control (short/medium/long)
428
510
  - ✅ **NEW**: Temperature and token limit configuration
429
511
  - ✅ **NEW**: Command-line OCR example script with full parameter support
512
+ - ✅ **NEW**: **Confidence Scoring System** - Advanced AI hallucination detection
513
+ - ✅ **NEW**: `xtxt_image_with_confidence()` - Returns text + confidence score
514
+ - ✅ **NEW**: Automatic rejection of low-confidence results with configurable thresholds
515
+ - ✅ **NEW**: 60+ hallucination patterns detected (historical, artistic, fantasy, scientific)
516
+ - ✅ **NEW**: `--show-confidence` and `--confidence` CLI parameters
430
517
  - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
518
+ - ✅ **ENHANCED**: Comprehensive AI safety warnings and best practices documentation
431
519
  - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
432
520
 
433
521
  ### v0.2.4
@@ -204,6 +204,45 @@ image_response = requests.get("https://example.com/document.png")
204
204
  text = xtxt(image_response.content)
205
205
  ```
206
206
 
207
+ ### AI OCR Confidence Scoring (NEW)
208
+
209
+ ⚠️ **IMPORTANT**: AI-powered OCR is experimental technology that may produce errors, hallucinations, or misinterpretations. Always validate results for critical applications.
210
+
211
+ The OCR-Ollama system includes confidence scoring to help identify unreliable results:
212
+
213
+ ```python
214
+ from pyxtxt import xtxt_image_with_confidence, set_ollama_config
215
+
216
+ # Configure confidence threshold (0.0-1.0, default: 0.7)
217
+ set_ollama_config(confidence_threshold=0.8) # More restrictive
218
+
219
+ # Get text with confidence score
220
+ text, confidence = xtxt_image_with_confidence("document.png", mode="ocr")
221
+ print(f"Confidence: {confidence:.2f} ({confidence*100:.1f}%)")
222
+ print(f"Text: {text}")
223
+
224
+ # Check if reliable
225
+ if confidence < 0.7:
226
+ print("⚠️ Low confidence - result may be unreliable")
227
+ print("Consider using traditional OCR or manual verification")
228
+ else:
229
+ print("✅ Good confidence - result likely reliable")
230
+ ```
231
+
232
+ #### Confidence Scoring Features
233
+
234
+ - **Hallucination Detection**: Penalizes historical/cultural references that often indicate AI misinterpretation
235
+ - **Pattern Recognition**: Rewards structured text (numbers, punctuation, proper formatting)
236
+ - **Uncertainty Detection**: Flags vague language ("appears to be", "seems like", etc.)
237
+ - **Quality Assessment**: Considers content length, repetition, and coherence
238
+
239
+ #### Common Hallucination Patterns (Automatically Detected)
240
+ - **Historical Content**: "ancient", "medieval", "Egyptian papyrus", "hieroglyphs"
241
+ - **Artistic Interpretations**: "painting", "artwork", "masterpiece", "Renaissance"
242
+ - **Fantasy Content**: "mystical", "magical", "dragon", "wizard"
243
+ - **Scientific Misinterpretation**: "fossil", "geological formation", "crystal structure"
244
+ - **Vague Language**: "unclear", "difficult to read", "appears to be"
245
+
207
246
  ### Command-Line OCR Example
208
247
 
209
248
  A complete example script for command-line usage is available:
@@ -220,6 +259,11 @@ with open("ocr_example.py", "wb") as f:
220
259
  # python ocr_example.py document.png
221
260
  # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
222
261
  # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
262
+
263
+ # NEW: Confidence scoring examples
264
+ # python ocr_example.py suspicious.png --show-confidence
265
+ # python ocr_example.py medical.png --confidence=0.9 --show-confidence
266
+ # python ocr_example.py diagram.png --confidence=0.5 --mode=describe --show-confidence
223
267
  ```
224
268
 
225
269
  The script supports:
@@ -229,6 +273,8 @@ The script supports:
229
273
  - **Style control**: descriptive, technical, simple, detailed
230
274
  - **Length control**: short, medium, long captions
231
275
  - **Temperature**: Adjust LLM creativity (0.0-1.0)
276
+ - **Confidence scoring**: Set threshold and display confidence scores
277
+ - **Quality filtering**: Automatically reject low-confidence results
232
278
 
233
279
  ### Show Available Formats
234
280
  ```python
@@ -272,6 +318,7 @@ text = xtxt(attachment_bytes)
272
318
 
273
319
  ## ⚠️ Known Limitations
274
320
 
321
+ ### General Limitations
275
322
  - **Legacy file detection**: When using raw streams without filenames, legacy files (.doc, .xls, .ppt) may not be correctly detected due to identical file signatures in libmagic
276
323
  - **Filename hints recommended**: When available, providing original filenames improves detection accuracy
277
324
  - **MSWrite .doc files**: Require `antiword` installation:
@@ -279,6 +326,41 @@ text = xtxt(attachment_bytes)
279
326
  sudo apt-get update && sudo apt-get install antiword
280
327
  ```
281
328
 
329
+ ### 🤖 AI-Powered Features - Important Warnings
330
+
331
+ **⚠️ EXPERIMENTAL TECHNOLOGY**: AI-powered features (OCR-Ollama, audio transcription) are based on machine learning models and may produce:
332
+
333
+ #### Potential Issues:
334
+ - **Hallucinations**: AI may "see" or "hear" content that isn't actually present
335
+ - **Misinterpretations**: Complex images may be incorrectly identified (e.g., X-ray images mistaken for historical artifacts)
336
+ - **Language Errors**: Transcription accuracy depends on audio quality, accents, and background noise
337
+ - **Context Confusion**: AI may apply inappropriate cultural/historical context to technical content
338
+ - **Model Dependence**: Results vary significantly between different AI models (gemma3, llava, whisper versions)
339
+ - **Bias and Inconsistency**: Models may exhibit cultural, linguistic, or domain-specific biases
340
+
341
+ #### Critical Applications Warning:
342
+ **🚨 DO NOT USE for critical applications** such as:
343
+ - Medical diagnosis or medical image interpretation
344
+ - Legal document analysis requiring perfect accuracy
345
+ - Financial data extraction where errors have monetary impact
346
+ - Security/safety systems where false positives/negatives are dangerous
347
+ - Academic research requiring citation-quality accuracy
348
+
349
+ #### Best Practices:
350
+ - **Always validate AI results** against source material when accuracy matters
351
+ - **Use confidence scoring** to identify potentially unreliable results
352
+ - **Cross-reference** with traditional OCR/transcription tools for important content
353
+ - **Human review** recommended for any production use case
354
+ - **Test thoroughly** with your specific content types and use cases
355
+ - **Fallback options**: Keep traditional OCR (EasyOCR) available as backup
356
+
357
+ #### Recommended Use Cases:
358
+ ✅ Content discovery and initial text extraction
359
+ ✅ Batch processing of low-stakes content
360
+ ✅ Development and prototyping workflows
361
+ ✅ Personal document organization
362
+ ✅ Educational and learning projects
363
+
282
364
  ## 📖 Full Examples
283
365
 
284
366
  ### Accessing Examples After Installation
@@ -323,7 +405,13 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
323
405
  - ✅ **NEW**: Caption length control (short/medium/long)
324
406
  - ✅ **NEW**: Temperature and token limit configuration
325
407
  - ✅ **NEW**: Command-line OCR example script with full parameter support
408
+ - ✅ **NEW**: **Confidence Scoring System** - Advanced AI hallucination detection
409
+ - ✅ **NEW**: `xtxt_image_with_confidence()` - Returns text + confidence score
410
+ - ✅ **NEW**: Automatic rejection of low-confidence results with configurable thresholds
411
+ - ✅ **NEW**: 60+ hallucination patterns detected (historical, artistic, fantasy, scientific)
412
+ - ✅ **NEW**: `--show-confidence` and `--confidence` CLI parameters
326
413
  - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
414
+ - ✅ **ENHANCED**: Comprehensive AI safety warnings and best practices documentation
327
415
  - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
328
416
 
329
417
  ### v0.2.4
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.3.2"
3
+ version = "0.3.4"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -4,12 +4,14 @@ from .core import xtxt, extxt_available_formats, xtxt_from_url
4
4
  try:
5
5
  from .estrattori.ocr_ollama import (
6
6
  set_ollama_model, get_ollama_model, xtxt_image_describe,
7
- set_ollama_config, get_ollama_config, reset_ollama_config
7
+ set_ollama_config, get_ollama_config, reset_ollama_config,
8
+ xtxt_image_with_confidence
8
9
  )
9
10
  __all__ = [
10
11
  "xtxt", "extxt_available_formats", "xtxt_from_url",
11
12
  "set_ollama_model", "get_ollama_model", "xtxt_image_describe",
12
- "set_ollama_config", "get_ollama_config", "reset_ollama_config"
13
+ "set_ollama_config", "get_ollama_config", "reset_ollama_config",
14
+ "xtxt_image_with_confidence"
13
15
  ]
14
16
  except ImportError:
15
17
  __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
@@ -0,0 +1,520 @@
1
+ # pyxtxt/extractors/image_ocr_ollama.py
2
+ from . import register_extractor
3
+ from io import BytesIO
4
+ import base64
5
+
6
+ try:
7
+ import ollama
8
+ from PIL import Image
9
+ except ImportError:
10
+ ollama = None
11
+ Image = None
12
+
13
+ # Global configuration for Ollama model and parameters
14
+ OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
15
+ OLLAMA_CONFIG = {
16
+ 'language': 'auto', # Language hint for caption generation
17
+ 'caption_length': 'medium', # short, medium, long
18
+ 'style': 'descriptive', # descriptive, technical, simple, detailed
19
+ 'temperature': 0.1, # Response creativity (0.0-1.0)
20
+ 'max_tokens': 1500, # Maximum response length
21
+ 'confidence_threshold': 0.7, # Minimum confidence for text extraction
22
+ 'context': 'general' # Context hint: general, cookbook, document, diagram, etc.
23
+ }
24
+
25
+ def set_ollama_model(model_name: str):
26
+ """
27
+ Set the Ollama model to use for OCR.
28
+
29
+ Recommended multimodal models:
30
+ - gemma3:4b (default, balanced speed/quality)
31
+ - gemma3:12b (higher quality, slower)
32
+ - gemma3:27b (best quality, very slow)
33
+ - llava:7b (alternative vision model)
34
+ - llava:13b (higher quality LLAVA)
35
+ """
36
+ global OLLAMA_MODEL
37
+ OLLAMA_MODEL = model_name
38
+ print(f"✅ Ollama OCR model set to: {model_name}")
39
+
40
+ def get_ollama_model():
41
+ """Get current Ollama model name"""
42
+ return OLLAMA_MODEL
43
+
44
+ def set_ollama_config(**kwargs):
45
+ """
46
+ Configure Ollama LLM parameters for better caption generation.
47
+
48
+ Parameters:
49
+ - language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
50
+ - caption_length: Caption length ('short', 'medium', 'long')
51
+ - style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
52
+ - context: Content context ('general', 'cookbook', 'document', 'handwriting', 'technical')
53
+ - temperature: Response creativity 0.0-1.0 (default: 0.1)
54
+ - max_tokens: Maximum response length (default: 1500)
55
+ - confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
56
+
57
+ Examples:
58
+ set_ollama_config(language='italian', style='detailed')
59
+ set_ollama_config(context='document', caption_length='long')
60
+ set_ollama_config(context='handwriting', temperature=0.2)
61
+ """
62
+ global OLLAMA_CONFIG
63
+ for key, value in kwargs.items():
64
+ if key in OLLAMA_CONFIG:
65
+ OLLAMA_CONFIG[key] = value
66
+ print(f"✅ Ollama config updated: {key} = {value}")
67
+ else:
68
+ print(f"⚠️ Unknown config parameter: {key}")
69
+
70
+ def get_ollama_config():
71
+ """Get current Ollama configuration"""
72
+ return OLLAMA_CONFIG.copy()
73
+
74
+ def reset_ollama_config():
75
+ """Reset Ollama configuration to defaults"""
76
+ global OLLAMA_CONFIG
77
+ OLLAMA_CONFIG = {
78
+ 'language': 'auto',
79
+ 'caption_length': 'medium',
80
+ 'style': 'descriptive',
81
+ 'temperature': 0.1,
82
+ 'max_tokens': 1500,
83
+ 'confidence_threshold': 0.7,
84
+ 'context': 'general'
85
+ }
86
+ print("✅ Ollama configuration reset to defaults")
87
+
88
+ def _calculate_confidence_score(content: str, mode: str) -> float:
89
+ """
90
+ Calculate confidence score (0.0-1.0) for OCR/caption quality.
91
+
92
+ Evaluates response based on:
93
+ - Content length and structure
94
+ - Presence of meaningful text patterns
95
+ - Absence of hallucination indicators
96
+ - Appropriate response format
97
+ """
98
+ if not content or len(content.strip()) < 3:
99
+ return 0.0
100
+
101
+ content_lower = content.lower()
102
+ score = 0.5 # Base score
103
+
104
+ # Positive indicators (add to confidence)
105
+ positive_patterns = [
106
+ # Structured text indicators
107
+ (r'\b\d+\b', 0.1, "Contains numbers/measurements"),
108
+ (r'[.!?]', 0.1, "Has sentence punctuation"),
109
+ (r'\b(the|and|of|in|to|for|with|on|at|by)\b', 0.1, "Common words present"),
110
+ (r'\b[A-Z][a-z]+\b', 0.1, "Proper capitalization"),
111
+ (r'[,;:]', 0.05, "Has proper punctuation"),
112
+
113
+ # Content-specific indicators
114
+ (r'\b(text|document|page|title|heading)\b', 0.1, "Document references"),
115
+ (r'\b(image|photo|picture|shows|contains)\b', 0.1, "Visual descriptions"),
116
+ (r'\b\d+[%$€£¥]\b|\b[%$€£¥]\d+\b', 0.1, "Currency/percentages"),
117
+ (r'\b\d{1,2}[:/]\d{1,2}\b|\b\d{4}\b', 0.05, "Times/years"),
118
+ ]
119
+
120
+ # Negative indicators (reduce confidence)
121
+ negative_patterns = [
122
+ # Historical/Archaeological hallucinations (common with X-rays, scans)
123
+ ("ancient", 0.2, "May be hallucinating historical content"),
124
+ ("egypt", 0.15, "Egyptian hallucination"),
125
+ ("hieroglyph", 0.25, "Hieroglyphic hallucination"),
126
+ ("papyrus", 0.2, "Papyrus hallucination"),
127
+ ("scroll", 0.1, "Ancient scroll hallucination"),
128
+ ("medieval", 0.15, "Medieval hallucination"),
129
+ ("manuscript", 0.1, "Manuscript hallucination"),
130
+ ("archaeological", 0.15, "Archaeological hallucination"),
131
+ ("artifact", 0.12, "Artifact hallucination"),
132
+ ("roman", 0.1, "Roman historical hallucination"),
133
+ ("greek", 0.1, "Greek historical hallucination"),
134
+ ("biblical", 0.15, "Religious historical hallucination"),
135
+ ("stone tablet", 0.2, "Stone tablet hallucination"),
136
+ ("carved", 0.1, "Carving hallucination"),
137
+
138
+ # Art/Cultural hallucinations (common with medical images, diagrams)
139
+ ("painting", 0.1, "Artistic interpretation hallucination"),
140
+ ("artwork", 0.12, "Artwork hallucination"),
141
+ ("masterpiece", 0.15, "Art masterpiece hallucination"),
142
+ ("renaissance", 0.15, "Renaissance art hallucination"),
143
+ ("portrait", 0.08, "Portrait hallucination"),
144
+ ("landscape", 0.08, "Landscape hallucination"),
145
+ ("abstract art", 0.12, "Abstract art hallucination"),
146
+
147
+ # Fantasy/Fictional content (can occur with any unclear image)
148
+ ("mystical", 0.15, "Fantasy content hallucination"),
149
+ ("magical", 0.15, "Magical content hallucination"),
150
+ ("mythological", 0.15, "Mythology hallucination"),
151
+ ("fairy tale", 0.15, "Fairy tale hallucination"),
152
+ ("legend", 0.1, "Legend/folklore hallucination"),
153
+ ("dragon", 0.2, "Dragon hallucination"),
154
+ ("wizard", 0.15, "Fantasy character hallucination"),
155
+
156
+ # Scientific misinterpretations (X-rays as geological, etc.)
157
+ ("fossil", 0.15, "Fossil hallucination"),
158
+ ("geological", 0.1, "Geological misinterpretation"),
159
+ ("mineral", 0.1, "Mineral hallucination"),
160
+ ("crystal", 0.1, "Crystal hallucination"),
161
+ ("rock formation", 0.12, "Rock formation hallucination"),
162
+ ("sediment", 0.1, "Sediment hallucination"),
163
+
164
+ # Vague/uncertain language
165
+ ("unclear", 0.1, "Uncertainty indicator"),
166
+ ("difficult to read", 0.1, "Reading difficulty"),
167
+ ("appears to be", 0.05, "Tentative language"),
168
+ ("seems to", 0.05, "Uncertain language"),
169
+ ("might be", 0.1, "Possibility language"),
170
+ ("possibly", 0.1, "Uncertain language"),
171
+ ("probably", 0.08, "Probability language"),
172
+ ("looks like", 0.08, "Appearance-based guess"),
173
+ ("reminds me of", 0.1, "Subjective association"),
174
+ ("similar to", 0.05, "Similarity guess"),
175
+
176
+ # Quality/visibility issues (legitimate but indicate uncertainty)
177
+ ("blurry", 0.05, "Image quality issue"),
178
+ ("faded", 0.05, "Faded content"),
179
+ ("damaged", 0.05, "Damaged content"),
180
+ ("corrupted", 0.1, "Corrupted content"),
181
+ ("low resolution", 0.08, "Resolution issue"),
182
+ ("hard to make out", 0.1, "Visibility issue"),
183
+
184
+ # Generic/evasive responses
185
+ ("sorry, i cannot", 0.3, "Refusal response"),
186
+ ("cannot determine", 0.2, "Unable to process"),
187
+ ("unable to identify", 0.15, "Identification failure"),
188
+ ("not clear enough", 0.1, "Clarity issue"),
189
+ ("too dark", 0.08, "Darkness issue"),
190
+ ("too bright", 0.08, "Brightness issue"),
191
+ ("no visible text", 0.0, "No text found - appropriate response"),
192
+
193
+ # Repetitive/nonsensical content (AI breakdown indicators)
194
+ ("lorem ipsum", 0.2, "Placeholder text hallucination"),
195
+ ("test test test", 0.25, "Repetitive test pattern"),
196
+ ("abc abc abc", 0.2, "Repetitive pattern"),
197
+ ("error error", 0.15, "Error message hallucination"),
198
+ ]
199
+
200
+ import re
201
+
202
+ # Apply positive pattern scoring
203
+ for pattern, bonus, description in positive_patterns:
204
+ if re.search(pattern, content_lower):
205
+ score += bonus
206
+
207
+ # Apply negative pattern scoring
208
+ for pattern, penalty, description in negative_patterns:
209
+ if pattern in content_lower:
210
+ score -= penalty
211
+ print(f"⚠️ Confidence penalty ({penalty}): {description}")
212
+
213
+ # Length-based adjustments
214
+ if len(content) > 200: # Substantial content
215
+ score += 0.1
216
+ elif len(content) < 20: # Very short responses are suspicious
217
+ score -= 0.2
218
+
219
+ # Mode-specific adjustments
220
+ if mode == "describe":
221
+ # Check for proper formatting in describe mode
222
+ if "TEXT:" in content and "DESCRIPTION:" in content:
223
+ score += 0.1 # Proper format bonus
224
+ elif len(content) > 50: # Has substantial descriptive content
225
+ score += 0.05
226
+
227
+ # Context coherence check - very basic
228
+ words = content_lower.split()
229
+ if len(words) > 5:
230
+ # Check for repetitive patterns (hallucination indicator)
231
+ unique_words = set(words)
232
+ repetition_ratio = len(unique_words) / len(words)
233
+ if repetition_ratio < 0.3: # Too repetitive
234
+ score -= 0.2
235
+
236
+ # Clamp to valid range
237
+ return max(0.0, min(1.0, score))
238
+
239
+ if ollama and Image:
240
+ def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
241
+ """
242
+ Extract text from images using Ollama with multimodal models.
243
+
244
+ Args:
245
+ file_buffer: Image file buffer
246
+ mode: "ocr" (text only) or "describe" (text + description)
247
+ model: Override default model (optional)
248
+ """
249
+ try:
250
+ # Use specified model or global default
251
+ current_model = model or OLLAMA_MODEL
252
+
253
+ # Convert buffer to PIL Image
254
+ # Reset buffer position if it has read method
255
+ if hasattr(file_buffer, 'seek'):
256
+ file_buffer.seek(0)
257
+ image_data = file_buffer.read()
258
+ image = Image.open(BytesIO(image_data))
259
+
260
+ # Convert to RGB if needed
261
+ if image.mode != 'RGB':
262
+ image = image.convert('RGB')
263
+
264
+ # Convert image to base64
265
+ buffered = BytesIO()
266
+ image.save(buffered, format="PNG")
267
+ img_base64 = base64.b64encode(buffered.getvalue()).decode()
268
+
269
+ # Get current configuration
270
+ config = OLLAMA_CONFIG
271
+
272
+ # Build language hint
273
+ lang_hint = ""
274
+ if config['language'] != 'auto':
275
+ lang_hint = f"Text language: {config['language']}. "
276
+
277
+ # Different prompts based on mode
278
+ if mode == "ocr":
279
+ prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
280
+ - Titles, headings, and main text content
281
+ - Small print, captions, labels, and annotations
282
+ - Numbers, measurements, quantities, and symbols
283
+ - Menu items, ingredient lists, cooking instructions
284
+ - Any text in boxes, speech bubbles, or decorative elements
285
+
286
+ IMPORTANT:
287
+ - Read carefully and include even small or partially visible text
288
+ - Preserve the original formatting and line breaks where possible
289
+ - {lang_hint}Process the text from left to right, top to bottom
290
+ - If you cannot find any readable text at all, respond with 'NO_TEXT_FOUND'
291
+
292
+ Extracted text:"""
293
+
294
+ else: # mode == "describe"
295
+ # Build style-specific prompts
296
+ style_prompts = {
297
+ 'descriptive': "Provide a clear, descriptive explanation",
298
+ 'technical': "Use technical terminology and precise descriptions",
299
+ 'simple': "Use simple, easy-to-understand language",
300
+ 'detailed': "Provide comprehensive details about all visual elements"
301
+ }
302
+
303
+ length_hints = {
304
+ 'short': "Keep descriptions brief (1-2 sentences)",
305
+ 'medium': "Provide moderate detail (2-4 sentences)",
306
+ 'long': "Give comprehensive descriptions (4-8 sentences)"
307
+ }
308
+
309
+ style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
310
+ length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
311
+
312
+ # Context-specific hints (only when explicitly set)
313
+ context_hint = ""
314
+ context = config.get('context', 'general').lower()
315
+ if context == 'cookbook' or context == 'recipe':
316
+ context_hint = """
317
+ - If this appears to be a recipe/cookbook page, include: ingredients, cooking steps, quantities, cooking times
318
+ - Mention any photos of prepared dishes or cooking techniques shown
319
+ - Note any special formatting like ingredient lists, step numbers, or cooking tips"""
320
+ elif context == 'document':
321
+ context_hint = """
322
+ - Focus on document structure: headers, paragraphs, sections, page numbers
323
+ - Note any official formatting, letterheads, signatures, or stamps"""
324
+ elif context == 'handwriting' or context == 'notes':
325
+ context_hint = """
326
+ - Pay special attention to handwritten text which may be harder to read
327
+ - Note any sketches, diagrams, or informal formatting typical of personal notes"""
328
+ elif context == 'technical' or context == 'diagram':
329
+ context_hint = """
330
+ - Focus on technical elements: labels, measurements, specifications, diagrams
331
+ - Include any mathematical formulas, technical symbols, or engineering notations"""
332
+
333
+ prompt = f"""Analyze this image and provide:
334
+ 1. All visible text exactly as written (preserve formatting, line breaks, bullet points)
335
+ 2. Image description following these guidelines:
336
+ - {style_instruction}
337
+ - {length_instruction}
338
+ - {lang_hint}Focus on key visual elements, layout, and context{context_hint}
339
+
340
+ Format:
341
+ TEXT: [all visible text here, or NO_TEXT_FOUND if none]
342
+ DESCRIPTION: [image description following the guidelines above]"""
343
+
344
+ # Send request to Ollama with configured parameters
345
+ response = ollama.generate(
346
+ model=current_model,
347
+ prompt=prompt,
348
+ images=[img_base64],
349
+ options={
350
+ 'temperature': config['temperature'],
351
+ 'top_p': 0.9,
352
+ 'num_predict': config['max_tokens']
353
+ }
354
+ )
355
+
356
+ # Extract and clean response
357
+ extracted_content = response.get('response', '').strip()
358
+
359
+ # Calculate confidence score based on response quality indicators
360
+ confidence_score = _calculate_confidence_score(extracted_content, mode)
361
+
362
+ # Check confidence threshold
363
+ if confidence_score < config['confidence_threshold']:
364
+ print(f"⚠️ Low confidence ({confidence_score:.2f} < {config['confidence_threshold']}): {extracted_content[:100]}...")
365
+ if mode == "ocr":
366
+ return "" # Return empty for OCR if below threshold
367
+ else:
368
+ # For describe mode, add warning prefix
369
+ extracted_content = f"[LOW_CONFIDENCE_{confidence_score:.2f}] {extracted_content}"
370
+ else:
371
+ print(f"✅ Good confidence ({confidence_score:.2f}): Processing successful")
372
+
373
+ # Handle no-text case for OCR mode
374
+ if mode == "ocr" and ('NO_TEXT_FOUND' in extracted_content or len(extracted_content) < 3):
375
+ return ""
376
+
377
+ return extracted_content
378
+
379
+ except Exception as e:
380
+ print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
381
+ return ""
382
+
383
+ def xtxt_image_ocr_ollama_with_confidence(file_buffer, mode="ocr", model=None):
384
+ """
385
+ Version of OCR-Ollama that returns both text and confidence score.
386
+
387
+ Returns:
388
+ tuple: (extracted_text, confidence_score)
389
+ """
390
+ try:
391
+ # Use specified model or global default
392
+ current_model = model or OLLAMA_MODEL
393
+
394
+ # Convert buffer to PIL Image
395
+ if hasattr(file_buffer, 'seek'):
396
+ file_buffer.seek(0)
397
+ image_data = file_buffer.read()
398
+ image = Image.open(BytesIO(image_data))
399
+
400
+ # Convert to RGB if needed
401
+ if image.mode != 'RGB':
402
+ image = image.convert('RGB')
403
+
404
+ # Convert image to base64
405
+ buffered = BytesIO()
406
+ image.save(buffered, format="PNG")
407
+ img_base64 = base64.b64encode(buffered.getvalue()).decode()
408
+
409
+ # Get current configuration
410
+ config = OLLAMA_CONFIG
411
+
412
+ # Build prompts (same logic as main function)
413
+ lang_hint = ""
414
+ if config['language'] != 'auto':
415
+ lang_hint = f"Text language: {config['language']}. "
416
+
417
+ if mode == "ocr":
418
+ prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
419
+ - Titles, headings, and main text content
420
+ - Small print, captions, labels, and annotations
421
+ - Numbers, measurements, quantities, and symbols
422
+ - Menu items, ingredient lists, cooking instructions
423
+ - Any text in boxes, speech bubbles, or decorative elements
424
+
425
+ IMPORTANT:
426
+ - Read carefully and include even small or partially visible text
427
+ - Preserve the original formatting and line breaks where possible
428
+ - {lang_hint}Process the text from left to right, top to bottom
429
+ - If you cannot find any readable text at all, respond with 'NO_TEXT_FOUND'
430
+
431
+ Extracted text:"""
432
+ else: # describe mode
433
+ style_prompts = {
434
+ 'descriptive': "Provide a clear, descriptive explanation",
435
+ 'technical': "Use technical terminology and precise descriptions",
436
+ 'simple': "Use simple, easy-to-understand language",
437
+ 'detailed': "Provide comprehensive details about all visual elements"
438
+ }
439
+
440
+ length_hints = {
441
+ 'short': "Keep descriptions brief (1-2 sentences)",
442
+ 'medium': "Provide moderate detail (2-4 sentences)",
443
+ 'long': "Give comprehensive descriptions (4-8 sentences)"
444
+ }
445
+
446
+ style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
447
+ length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
448
+
449
+ prompt = f"""Analyze this image and provide:
450
+ 1. All visible text exactly as written (preserve formatting, line breaks, bullet points)
451
+ 2. Image description following these guidelines:
452
+ - {style_instruction}
453
+ - {length_instruction}
454
+ - {lang_hint}Focus on key visual elements, layout, and context
455
+
456
+ Format:
457
+ TEXT: [all visible text here, or NO_TEXT_FOUND if none]
458
+ DESCRIPTION: [image description following the guidelines above]"""
459
+
460
+ # Send request to Ollama
461
+ response = ollama.generate(
462
+ model=current_model,
463
+ prompt=prompt,
464
+ images=[img_base64],
465
+ options={
466
+ 'temperature': config['temperature'],
467
+ 'top_p': 0.9,
468
+ 'num_predict': config['max_tokens']
469
+ }
470
+ )
471
+
472
+ # Extract response
473
+ extracted_content = response.get('response', '').strip()
474
+
475
+ # Calculate confidence without threshold filtering
476
+ confidence_score = _calculate_confidence_score(extracted_content, mode)
477
+
478
+ # Return both text and confidence
479
+ return extracted_content, confidence_score
480
+
481
+ except Exception as e:
482
+ print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
483
+ return "", 0.0
484
+
485
+ # Wrapper functions for each mode
486
+ def xtxt_image_ocr_only(file_input):
487
+ """Traditional OCR: extract only visible text using Ollama"""
488
+ # Handle both file paths and buffers
489
+ if isinstance(file_input, str):
490
+ with open(file_input, 'rb') as f:
491
+ return xtxt_image_ocr_ollama(f, mode="ocr")
492
+ else:
493
+ return xtxt_image_ocr_ollama(file_input, mode="ocr")
494
+
495
+ def xtxt_image_describe(file_input):
496
+ """OCR + Description: text + image context using Ollama"""
497
+ # Handle both file paths and buffers
498
+ if isinstance(file_input, str):
499
+ with open(file_input, 'rb') as f:
500
+ return xtxt_image_ocr_ollama(f, mode="describe")
501
+ else:
502
+ return xtxt_image_ocr_ollama(file_input, mode="describe")
503
+
504
+ def xtxt_image_with_confidence(file_input, mode="ocr"):
505
+ """Get both text and confidence score from image OCR"""
506
+ if isinstance(file_input, str):
507
+ with open(file_input, 'rb') as f:
508
+ return xtxt_image_ocr_ollama_with_confidence(f, mode=mode)
509
+ else:
510
+ return xtxt_image_ocr_ollama_with_confidence(file_input, mode=mode)
511
+
512
+ # Register OCR-only version as default
513
+ # Note: Will override traditional EasyOCR if both modules are loaded
514
+ image_formats = [
515
+ "image/jpeg", "image/jpg", "image/png",
516
+ "image/bmp", "image/tiff", "image/webp"
517
+ ]
518
+
519
+ for format_type in image_formats:
520
+ register_extractor(format_type, xtxt_image_ocr_only, name="OCR-Ollama")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.2
3
+ Version: 0.3.4
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -308,6 +308,45 @@ image_response = requests.get("https://example.com/document.png")
308
308
  text = xtxt(image_response.content)
309
309
  ```
310
310
 
311
+ ### AI OCR Confidence Scoring (NEW)
312
+
313
+ ⚠️ **IMPORTANT**: AI-powered OCR is experimental technology that may produce errors, hallucinations, or misinterpretations. Always validate results for critical applications.
314
+
315
+ The OCR-Ollama system includes confidence scoring to help identify unreliable results:
316
+
317
+ ```python
318
+ from pyxtxt import xtxt_image_with_confidence, set_ollama_config
319
+
320
+ # Configure confidence threshold (0.0-1.0, default: 0.7)
321
+ set_ollama_config(confidence_threshold=0.8) # More restrictive
322
+
323
+ # Get text with confidence score
324
+ text, confidence = xtxt_image_with_confidence("document.png", mode="ocr")
325
+ print(f"Confidence: {confidence:.2f} ({confidence*100:.1f}%)")
326
+ print(f"Text: {text}")
327
+
328
+ # Check if reliable
329
+ if confidence < 0.7:
330
+ print("⚠️ Low confidence - result may be unreliable")
331
+ print("Consider using traditional OCR or manual verification")
332
+ else:
333
+ print("✅ Good confidence - result likely reliable")
334
+ ```
335
+
336
+ #### Confidence Scoring Features
337
+
338
+ - **Hallucination Detection**: Penalizes historical/cultural references that often indicate AI misinterpretation
339
+ - **Pattern Recognition**: Rewards structured text (numbers, punctuation, proper formatting)
340
+ - **Uncertainty Detection**: Flags vague language ("appears to be", "seems like", etc.)
341
+ - **Quality Assessment**: Considers content length, repetition, and coherence
342
+
343
+ #### Common Hallucination Patterns (Automatically Detected)
344
+ - **Historical Content**: "ancient", "medieval", "Egyptian papyrus", "hieroglyphs"
345
+ - **Artistic Interpretations**: "painting", "artwork", "masterpiece", "Renaissance"
346
+ - **Fantasy Content**: "mystical", "magical", "dragon", "wizard"
347
+ - **Scientific Misinterpretation**: "fossil", "geological formation", "crystal structure"
348
+ - **Vague Language**: "unclear", "difficult to read", "appears to be"
349
+
311
350
  ### Command-Line OCR Example
312
351
 
313
352
  A complete example script for command-line usage is available:
@@ -324,6 +363,11 @@ with open("ocr_example.py", "wb") as f:
324
363
  # python ocr_example.py document.png
325
364
  # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
326
365
  # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
366
+
367
+ # NEW: Confidence scoring examples
368
+ # python ocr_example.py suspicious.png --show-confidence
369
+ # python ocr_example.py medical.png --confidence=0.9 --show-confidence
370
+ # python ocr_example.py diagram.png --confidence=0.5 --mode=describe --show-confidence
327
371
  ```
328
372
 
329
373
  The script supports:
@@ -333,6 +377,8 @@ The script supports:
333
377
  - **Style control**: descriptive, technical, simple, detailed
334
378
  - **Length control**: short, medium, long captions
335
379
  - **Temperature**: Adjust LLM creativity (0.0-1.0)
380
+ - **Confidence scoring**: Set threshold and display confidence scores
381
+ - **Quality filtering**: Automatically reject low-confidence results
336
382
 
337
383
  ### Show Available Formats
338
384
  ```python
@@ -376,6 +422,7 @@ text = xtxt(attachment_bytes)
376
422
 
377
423
  ## ⚠️ Known Limitations
378
424
 
425
+ ### General Limitations
379
426
  - **Legacy file detection**: When using raw streams without filenames, legacy files (.doc, .xls, .ppt) may not be correctly detected due to identical file signatures in libmagic
380
427
  - **Filename hints recommended**: When available, providing original filenames improves detection accuracy
381
428
  - **MSWrite .doc files**: Require `antiword` installation:
@@ -383,6 +430,41 @@ text = xtxt(attachment_bytes)
383
430
  sudo apt-get update && sudo apt-get install antiword
384
431
  ```
385
432
 
433
+ ### 🤖 AI-Powered Features - Important Warnings
434
+
435
+ **⚠️ EXPERIMENTAL TECHNOLOGY**: AI-powered features (OCR-Ollama, audio transcription) are based on machine learning models and may produce:
436
+
437
+ #### Potential Issues:
438
+ - **Hallucinations**: AI may "see" or "hear" content that isn't actually present
439
+ - **Misinterpretations**: Complex images may be incorrectly identified (e.g., X-ray images mistaken for historical artifacts)
440
+ - **Language Errors**: Transcription accuracy depends on audio quality, accents, and background noise
441
+ - **Context Confusion**: AI may apply inappropriate cultural/historical context to technical content
442
+ - **Model Dependence**: Results vary significantly between different AI models (gemma3, llava, whisper versions)
443
+ - **Bias and Inconsistency**: Models may exhibit cultural, linguistic, or domain-specific biases
444
+
445
+ #### Critical Applications Warning:
446
+ **🚨 DO NOT USE for critical applications** such as:
447
+ - Medical diagnosis or medical image interpretation
448
+ - Legal document analysis requiring perfect accuracy
449
+ - Financial data extraction where errors have monetary impact
450
+ - Security/safety systems where false positives/negatives are dangerous
451
+ - Academic research requiring citation-quality accuracy
452
+
453
+ #### Best Practices:
454
+ - **Always validate AI results** against source material when accuracy matters
455
+ - **Use confidence scoring** to identify potentially unreliable results
456
+ - **Cross-reference** with traditional OCR/transcription tools for important content
457
+ - **Human review** recommended for any production use case
458
+ - **Test thoroughly** with your specific content types and use cases
459
+ - **Fallback options**: Keep traditional OCR (EasyOCR) available as backup
460
+
461
+ #### Recommended Use Cases:
462
+ ✅ Content discovery and initial text extraction
463
+ ✅ Batch processing of low-stakes content
464
+ ✅ Development and prototyping workflows
465
+ ✅ Personal document organization
466
+ ✅ Educational and learning projects
467
+
386
468
  ## 📖 Full Examples
387
469
 
388
470
  ### Accessing Examples After Installation
@@ -427,7 +509,13 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
427
509
  - ✅ **NEW**: Caption length control (short/medium/long)
428
510
  - ✅ **NEW**: Temperature and token limit configuration
429
511
  - ✅ **NEW**: Command-line OCR example script with full parameter support
512
+ - ✅ **NEW**: **Confidence Scoring System** - Advanced AI hallucination detection
513
+ - ✅ **NEW**: `xtxt_image_with_confidence()` - Returns text + confidence score
514
+ - ✅ **NEW**: Automatic rejection of low-confidence results with configurable thresholds
515
+ - ✅ **NEW**: 60+ hallucination patterns detected (historical, artistic, fantasy, scientific)
516
+ - ✅ **NEW**: `--show-confidence` and `--confidence` CLI parameters
430
517
  - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
518
+ - ✅ **ENHANCED**: Comprehensive AI safety warnings and best practices documentation
431
519
  - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
432
520
 
433
521
  ### v0.2.4
@@ -1,202 +0,0 @@
1
- # pyxtxt/extractors/image_ocr_ollama.py
2
- from . import register_extractor
3
- from io import BytesIO
4
- import base64
5
-
6
- try:
7
- import ollama
8
- from PIL import Image
9
- except ImportError:
10
- ollama = None
11
- Image = None
12
-
13
- # Global configuration for Ollama model and parameters
14
- OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
15
- OLLAMA_CONFIG = {
16
- 'language': 'auto', # Language hint for caption generation
17
- 'caption_length': 'medium', # short, medium, long
18
- 'style': 'descriptive', # descriptive, technical, simple, detailed
19
- 'temperature': 0.1, # Response creativity (0.0-1.0)
20
- 'max_tokens': 1500, # Maximum response length
21
- 'confidence_threshold': 0.7 # Minimum confidence for text extraction
22
- }
23
-
24
- def set_ollama_model(model_name: str):
25
- """
26
- Set the Ollama model to use for OCR.
27
-
28
- Recommended multimodal models:
29
- - gemma3:4b (default, balanced speed/quality)
30
- - gemma3:12b (higher quality, slower)
31
- - gemma3:27b (best quality, very slow)
32
- - llava:7b (alternative vision model)
33
- - llava:13b (higher quality LLAVA)
34
- """
35
- global OLLAMA_MODEL
36
- OLLAMA_MODEL = model_name
37
- print(f"✅ Ollama OCR model set to: {model_name}")
38
-
39
- def get_ollama_model():
40
- """Get current Ollama model name"""
41
- return OLLAMA_MODEL
42
-
43
- def set_ollama_config(**kwargs):
44
- """
45
- Configure Ollama LLM parameters for better caption generation.
46
-
47
- Parameters:
48
- - language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
49
- - caption_length: Caption length ('short', 'medium', 'long')
50
- - style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
51
- - temperature: Response creativity 0.0-1.0 (default: 0.1)
52
- - max_tokens: Maximum response length (default: 1500)
53
- - confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
54
-
55
- Examples:
56
- set_ollama_config(language='italian', style='detailed')
57
- set_ollama_config(caption_length='long', temperature=0.3)
58
- """
59
- global OLLAMA_CONFIG
60
- for key, value in kwargs.items():
61
- if key in OLLAMA_CONFIG:
62
- OLLAMA_CONFIG[key] = value
63
- print(f"✅ Ollama config updated: {key} = {value}")
64
- else:
65
- print(f"⚠️ Unknown config parameter: {key}")
66
-
67
- def get_ollama_config():
68
- """Get current Ollama configuration"""
69
- return OLLAMA_CONFIG.copy()
70
-
71
- def reset_ollama_config():
72
- """Reset Ollama configuration to defaults"""
73
- global OLLAMA_CONFIG
74
- OLLAMA_CONFIG = {
75
- 'language': 'auto',
76
- 'caption_length': 'medium',
77
- 'style': 'descriptive',
78
- 'temperature': 0.1,
79
- 'max_tokens': 1500,
80
- 'confidence_threshold': 0.7
81
- }
82
- print("✅ Ollama configuration reset to defaults")
83
-
84
- if ollama and Image:
85
- def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
86
- """
87
- Extract text from images using Ollama with multimodal models.
88
-
89
- Args:
90
- file_buffer: Image file buffer
91
- mode: "ocr" (text only) or "describe" (text + description)
92
- model: Override default model (optional)
93
- """
94
- try:
95
- # Use specified model or global default
96
- current_model = model or OLLAMA_MODEL
97
-
98
- # Convert buffer to PIL Image
99
- image = Image.open(BytesIO(file_buffer.read()))
100
-
101
- # Convert to RGB if needed
102
- if image.mode != 'RGB':
103
- image = image.convert('RGB')
104
-
105
- # Convert image to base64
106
- buffered = BytesIO()
107
- image.save(buffered, format="PNG")
108
- img_base64 = base64.b64encode(buffered.getvalue()).decode()
109
-
110
- # Get current configuration
111
- config = OLLAMA_CONFIG
112
-
113
- # Build language hint
114
- lang_hint = ""
115
- if config['language'] != 'auto':
116
- lang_hint = f"Text language: {config['language']}. "
117
-
118
- # Different prompts based on mode
119
- if mode == "ocr":
120
- prompt = f"""Extract ALL visible text from this image exactly as it appears.
121
- Rules:
122
- - Only return text that is actually written/printed in the image
123
- - Preserve reading order (left to right, top to bottom)
124
- - Maintain line breaks and formatting
125
- - Include numbers, symbols, special characters
126
- - Do NOT add descriptions, interpretations, or context
127
- - {lang_hint}If no text is visible, return 'NO_TEXT_FOUND'
128
-
129
- Extracted text:"""
130
-
131
- else: # mode == "describe"
132
- # Build style-specific prompts
133
- style_prompts = {
134
- 'descriptive': "Provide a clear, descriptive explanation",
135
- 'technical': "Use technical terminology and precise descriptions",
136
- 'simple': "Use simple, easy-to-understand language",
137
- 'detailed': "Provide comprehensive details about all visual elements"
138
- }
139
-
140
- length_hints = {
141
- 'short': "Keep descriptions brief (1-2 sentences)",
142
- 'medium': "Provide moderate detail (2-4 sentences)",
143
- 'long': "Give comprehensive descriptions (4-8 sentences)"
144
- }
145
-
146
- style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
147
- length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
148
-
149
- prompt = f"""Analyze this image and provide:
150
- 1. All visible text exactly as written
151
- 2. Image description following these guidelines:
152
- - {style_instruction}
153
- - {length_instruction}
154
- - {lang_hint}Focus on key visual elements, layout, and context
155
-
156
- Format:
157
- TEXT: [all visible text here, or NO_TEXT_FOUND if none]
158
- DESCRIPTION: [image description following the guidelines above]"""
159
-
160
- # Send request to Ollama with configured parameters
161
- response = ollama.generate(
162
- model=current_model,
163
- prompt=prompt,
164
- images=[img_base64],
165
- options={
166
- 'temperature': config['temperature'],
167
- 'top_p': 0.9,
168
- 'num_predict': config['max_tokens']
169
- }
170
- )
171
-
172
- # Extract and clean response
173
- extracted_content = response.get('response', '').strip()
174
-
175
- # Handle no-text case for OCR mode
176
- if mode == "ocr" and ('NO_TEXT_FOUND' in extracted_content or len(extracted_content) < 3):
177
- return ""
178
-
179
- return extracted_content
180
-
181
- except Exception as e:
182
- print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
183
- return ""
184
-
185
- # Wrapper functions for each mode
186
- def xtxt_image_ocr_only(file_buffer):
187
- """Traditional OCR: extract only visible text using Ollama"""
188
- return xtxt_image_ocr_ollama(file_buffer, mode="ocr")
189
-
190
- def xtxt_image_describe(file_buffer):
191
- """OCR + Description: text + image context using Ollama"""
192
- return xtxt_image_ocr_ollama(file_buffer, mode="describe")
193
-
194
- # Register OCR-only version as default
195
- # Note: Will override traditional EasyOCR if both modules are loaded
196
- image_formats = [
197
- "image/jpeg", "image/jpg", "image/png",
198
- "image/bmp", "image/tiff", "image/webp"
199
- ]
200
-
201
- for format_type in image_formats:
202
- register_extractor(format_type, xtxt_image_ocr_only, name="OCR-Ollama")
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes