pyxtxt 0.3.2__tar.gz → 0.3.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyxtxt-0.3.2/src/pyxtxt.egg-info → pyxtxt-0.3.4}/PKG-INFO +89 -1
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/README.md +88 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/pyproject.toml +1 -1
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/__init__.py +4 -2
- pyxtxt-0.3.4/src/pyxtxt/estrattori/ocr_ollama.py +520 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4/src/pyxtxt.egg-info}/PKG-INFO +89 -1
- pyxtxt-0.3.2/src/pyxtxt/estrattori/ocr_ollama.py +0 -202
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/LICENSE +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/MANIFEST.in +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/setup.cfg +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/core.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/audio.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/doc.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/docx.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/eml.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/epub.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/md.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/msg.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/ocr.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/odt.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/pptx.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/rtf.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/svg.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/tex.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/xlsx.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/examples.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt/pyxtxt.py +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt.egg-info/requires.txt +0 -0
- {pyxtxt-0.3.2 → pyxtxt-0.3.4}/src/pyxtxt.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.4
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -308,6 +308,45 @@ image_response = requests.get("https://example.com/document.png")
|
|
|
308
308
|
text = xtxt(image_response.content)
|
|
309
309
|
```
|
|
310
310
|
|
|
311
|
+
### AI OCR Confidence Scoring (NEW)
|
|
312
|
+
|
|
313
|
+
⚠️ **IMPORTANT**: AI-powered OCR is experimental technology that may produce errors, hallucinations, or misinterpretations. Always validate results for critical applications.
|
|
314
|
+
|
|
315
|
+
The OCR-Ollama system includes confidence scoring to help identify unreliable results:
|
|
316
|
+
|
|
317
|
+
```python
|
|
318
|
+
from pyxtxt import xtxt_image_with_confidence, set_ollama_config
|
|
319
|
+
|
|
320
|
+
# Configure confidence threshold (0.0-1.0, default: 0.7)
|
|
321
|
+
set_ollama_config(confidence_threshold=0.8) # More restrictive
|
|
322
|
+
|
|
323
|
+
# Get text with confidence score
|
|
324
|
+
text, confidence = xtxt_image_with_confidence("document.png", mode="ocr")
|
|
325
|
+
print(f"Confidence: {confidence:.2f} ({confidence*100:.1f}%)")
|
|
326
|
+
print(f"Text: {text}")
|
|
327
|
+
|
|
328
|
+
# Check if reliable
|
|
329
|
+
if confidence < 0.7:
|
|
330
|
+
print("⚠️ Low confidence - result may be unreliable")
|
|
331
|
+
print("Consider using traditional OCR or manual verification")
|
|
332
|
+
else:
|
|
333
|
+
print("✅ Good confidence - result likely reliable")
|
|
334
|
+
```
|
|
335
|
+
|
|
336
|
+
#### Confidence Scoring Features
|
|
337
|
+
|
|
338
|
+
- **Hallucination Detection**: Penalizes historical/cultural references that often indicate AI misinterpretation
|
|
339
|
+
- **Pattern Recognition**: Rewards structured text (numbers, punctuation, proper formatting)
|
|
340
|
+
- **Uncertainty Detection**: Flags vague language ("appears to be", "seems like", etc.)
|
|
341
|
+
- **Quality Assessment**: Considers content length, repetition, and coherence
|
|
342
|
+
|
|
343
|
+
#### Common Hallucination Patterns (Automatically Detected)
|
|
344
|
+
- **Historical Content**: "ancient", "medieval", "Egyptian papyrus", "hieroglyphs"
|
|
345
|
+
- **Artistic Interpretations**: "painting", "artwork", "masterpiece", "Renaissance"
|
|
346
|
+
- **Fantasy Content**: "mystical", "magical", "dragon", "wizard"
|
|
347
|
+
- **Scientific Misinterpretation**: "fossil", "geological formation", "crystal structure"
|
|
348
|
+
- **Vague Language**: "unclear", "difficult to read", "appears to be"
|
|
349
|
+
|
|
311
350
|
### Command-Line OCR Example
|
|
312
351
|
|
|
313
352
|
A complete example script for command-line usage is available:
|
|
@@ -324,6 +363,11 @@ with open("ocr_example.py", "wb") as f:
|
|
|
324
363
|
# python ocr_example.py document.png
|
|
325
364
|
# python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
|
|
326
365
|
# python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
|
|
366
|
+
|
|
367
|
+
# NEW: Confidence scoring examples
|
|
368
|
+
# python ocr_example.py suspicious.png --show-confidence
|
|
369
|
+
# python ocr_example.py medical.png --confidence=0.9 --show-confidence
|
|
370
|
+
# python ocr_example.py diagram.png --confidence=0.5 --mode=describe --show-confidence
|
|
327
371
|
```
|
|
328
372
|
|
|
329
373
|
The script supports:
|
|
@@ -333,6 +377,8 @@ The script supports:
|
|
|
333
377
|
- **Style control**: descriptive, technical, simple, detailed
|
|
334
378
|
- **Length control**: short, medium, long captions
|
|
335
379
|
- **Temperature**: Adjust LLM creativity (0.0-1.0)
|
|
380
|
+
- **Confidence scoring**: Set threshold and display confidence scores
|
|
381
|
+
- **Quality filtering**: Automatically reject low-confidence results
|
|
336
382
|
|
|
337
383
|
### Show Available Formats
|
|
338
384
|
```python
|
|
@@ -376,6 +422,7 @@ text = xtxt(attachment_bytes)
|
|
|
376
422
|
|
|
377
423
|
## ⚠️ Known Limitations
|
|
378
424
|
|
|
425
|
+
### General Limitations
|
|
379
426
|
- **Legacy file detection**: When using raw streams without filenames, legacy files (.doc, .xls, .ppt) may not be correctly detected due to identical file signatures in libmagic
|
|
380
427
|
- **Filename hints recommended**: When available, providing original filenames improves detection accuracy
|
|
381
428
|
- **MSWrite .doc files**: Require `antiword` installation:
|
|
@@ -383,6 +430,41 @@ text = xtxt(attachment_bytes)
|
|
|
383
430
|
sudo apt-get update && sudo apt-get install antiword
|
|
384
431
|
```
|
|
385
432
|
|
|
433
|
+
### 🤖 AI-Powered Features - Important Warnings
|
|
434
|
+
|
|
435
|
+
**⚠️ EXPERIMENTAL TECHNOLOGY**: AI-powered features (OCR-Ollama, audio transcription) are based on machine learning models and may produce:
|
|
436
|
+
|
|
437
|
+
#### Potential Issues:
|
|
438
|
+
- **Hallucinations**: AI may "see" or "hear" content that isn't actually present
|
|
439
|
+
- **Misinterpretations**: Complex images may be incorrectly identified (e.g., X-ray images mistaken for historical artifacts)
|
|
440
|
+
- **Language Errors**: Transcription accuracy depends on audio quality, accents, and background noise
|
|
441
|
+
- **Context Confusion**: AI may apply inappropriate cultural/historical context to technical content
|
|
442
|
+
- **Model Dependence**: Results vary significantly between different AI models (gemma3, llava, whisper versions)
|
|
443
|
+
- **Bias and Inconsistency**: Models may exhibit cultural, linguistic, or domain-specific biases
|
|
444
|
+
|
|
445
|
+
#### Critical Applications Warning:
|
|
446
|
+
**🚨 DO NOT USE for critical applications** such as:
|
|
447
|
+
- Medical diagnosis or medical image interpretation
|
|
448
|
+
- Legal document analysis requiring perfect accuracy
|
|
449
|
+
- Financial data extraction where errors have monetary impact
|
|
450
|
+
- Security/safety systems where false positives/negatives are dangerous
|
|
451
|
+
- Academic research requiring citation-quality accuracy
|
|
452
|
+
|
|
453
|
+
#### Best Practices:
|
|
454
|
+
- **Always validate AI results** against source material when accuracy matters
|
|
455
|
+
- **Use confidence scoring** to identify potentially unreliable results
|
|
456
|
+
- **Cross-reference** with traditional OCR/transcription tools for important content
|
|
457
|
+
- **Human review** recommended for any production use case
|
|
458
|
+
- **Test thoroughly** with your specific content types and use cases
|
|
459
|
+
- **Fallback options**: Keep traditional OCR (EasyOCR) available as backup
|
|
460
|
+
|
|
461
|
+
#### Recommended Use Cases:
|
|
462
|
+
✅ Content discovery and initial text extraction
|
|
463
|
+
✅ Batch processing of low-stakes content
|
|
464
|
+
✅ Development and prototyping workflows
|
|
465
|
+
✅ Personal document organization
|
|
466
|
+
✅ Educational and learning projects
|
|
467
|
+
|
|
386
468
|
## 📖 Full Examples
|
|
387
469
|
|
|
388
470
|
### Accessing Examples After Installation
|
|
@@ -427,7 +509,13 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
427
509
|
- ✅ **NEW**: Caption length control (short/medium/long)
|
|
428
510
|
- ✅ **NEW**: Temperature and token limit configuration
|
|
429
511
|
- ✅ **NEW**: Command-line OCR example script with full parameter support
|
|
512
|
+
- ✅ **NEW**: **Confidence Scoring System** - Advanced AI hallucination detection
|
|
513
|
+
- ✅ **NEW**: `xtxt_image_with_confidence()` - Returns text + confidence score
|
|
514
|
+
- ✅ **NEW**: Automatic rejection of low-confidence results with configurable thresholds
|
|
515
|
+
- ✅ **NEW**: 60+ hallucination patterns detected (historical, artistic, fantasy, scientific)
|
|
516
|
+
- ✅ **NEW**: `--show-confidence` and `--confidence` CLI parameters
|
|
430
517
|
- ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
|
|
518
|
+
- ✅ **ENHANCED**: Comprehensive AI safety warnings and best practices documentation
|
|
431
519
|
- ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
|
|
432
520
|
|
|
433
521
|
### v0.2.4
|
|
@@ -204,6 +204,45 @@ image_response = requests.get("https://example.com/document.png")
|
|
|
204
204
|
text = xtxt(image_response.content)
|
|
205
205
|
```
|
|
206
206
|
|
|
207
|
+
### AI OCR Confidence Scoring (NEW)
|
|
208
|
+
|
|
209
|
+
⚠️ **IMPORTANT**: AI-powered OCR is experimental technology that may produce errors, hallucinations, or misinterpretations. Always validate results for critical applications.
|
|
210
|
+
|
|
211
|
+
The OCR-Ollama system includes confidence scoring to help identify unreliable results:
|
|
212
|
+
|
|
213
|
+
```python
|
|
214
|
+
from pyxtxt import xtxt_image_with_confidence, set_ollama_config
|
|
215
|
+
|
|
216
|
+
# Configure confidence threshold (0.0-1.0, default: 0.7)
|
|
217
|
+
set_ollama_config(confidence_threshold=0.8) # More restrictive
|
|
218
|
+
|
|
219
|
+
# Get text with confidence score
|
|
220
|
+
text, confidence = xtxt_image_with_confidence("document.png", mode="ocr")
|
|
221
|
+
print(f"Confidence: {confidence:.2f} ({confidence*100:.1f}%)")
|
|
222
|
+
print(f"Text: {text}")
|
|
223
|
+
|
|
224
|
+
# Check if reliable
|
|
225
|
+
if confidence < 0.7:
|
|
226
|
+
print("⚠️ Low confidence - result may be unreliable")
|
|
227
|
+
print("Consider using traditional OCR or manual verification")
|
|
228
|
+
else:
|
|
229
|
+
print("✅ Good confidence - result likely reliable")
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
#### Confidence Scoring Features
|
|
233
|
+
|
|
234
|
+
- **Hallucination Detection**: Penalizes historical/cultural references that often indicate AI misinterpretation
|
|
235
|
+
- **Pattern Recognition**: Rewards structured text (numbers, punctuation, proper formatting)
|
|
236
|
+
- **Uncertainty Detection**: Flags vague language ("appears to be", "seems like", etc.)
|
|
237
|
+
- **Quality Assessment**: Considers content length, repetition, and coherence
|
|
238
|
+
|
|
239
|
+
#### Common Hallucination Patterns (Automatically Detected)
|
|
240
|
+
- **Historical Content**: "ancient", "medieval", "Egyptian papyrus", "hieroglyphs"
|
|
241
|
+
- **Artistic Interpretations**: "painting", "artwork", "masterpiece", "Renaissance"
|
|
242
|
+
- **Fantasy Content**: "mystical", "magical", "dragon", "wizard"
|
|
243
|
+
- **Scientific Misinterpretation**: "fossil", "geological formation", "crystal structure"
|
|
244
|
+
- **Vague Language**: "unclear", "difficult to read", "appears to be"
|
|
245
|
+
|
|
207
246
|
### Command-Line OCR Example
|
|
208
247
|
|
|
209
248
|
A complete example script for command-line usage is available:
|
|
@@ -220,6 +259,11 @@ with open("ocr_example.py", "wb") as f:
|
|
|
220
259
|
# python ocr_example.py document.png
|
|
221
260
|
# python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
|
|
222
261
|
# python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
|
|
262
|
+
|
|
263
|
+
# NEW: Confidence scoring examples
|
|
264
|
+
# python ocr_example.py suspicious.png --show-confidence
|
|
265
|
+
# python ocr_example.py medical.png --confidence=0.9 --show-confidence
|
|
266
|
+
# python ocr_example.py diagram.png --confidence=0.5 --mode=describe --show-confidence
|
|
223
267
|
```
|
|
224
268
|
|
|
225
269
|
The script supports:
|
|
@@ -229,6 +273,8 @@ The script supports:
|
|
|
229
273
|
- **Style control**: descriptive, technical, simple, detailed
|
|
230
274
|
- **Length control**: short, medium, long captions
|
|
231
275
|
- **Temperature**: Adjust LLM creativity (0.0-1.0)
|
|
276
|
+
- **Confidence scoring**: Set threshold and display confidence scores
|
|
277
|
+
- **Quality filtering**: Automatically reject low-confidence results
|
|
232
278
|
|
|
233
279
|
### Show Available Formats
|
|
234
280
|
```python
|
|
@@ -272,6 +318,7 @@ text = xtxt(attachment_bytes)
|
|
|
272
318
|
|
|
273
319
|
## ⚠️ Known Limitations
|
|
274
320
|
|
|
321
|
+
### General Limitations
|
|
275
322
|
- **Legacy file detection**: When using raw streams without filenames, legacy files (.doc, .xls, .ppt) may not be correctly detected due to identical file signatures in libmagic
|
|
276
323
|
- **Filename hints recommended**: When available, providing original filenames improves detection accuracy
|
|
277
324
|
- **MSWrite .doc files**: Require `antiword` installation:
|
|
@@ -279,6 +326,41 @@ text = xtxt(attachment_bytes)
|
|
|
279
326
|
sudo apt-get update && sudo apt-get install antiword
|
|
280
327
|
```
|
|
281
328
|
|
|
329
|
+
### 🤖 AI-Powered Features - Important Warnings
|
|
330
|
+
|
|
331
|
+
**⚠️ EXPERIMENTAL TECHNOLOGY**: AI-powered features (OCR-Ollama, audio transcription) are based on machine learning models and may produce:
|
|
332
|
+
|
|
333
|
+
#### Potential Issues:
|
|
334
|
+
- **Hallucinations**: AI may "see" or "hear" content that isn't actually present
|
|
335
|
+
- **Misinterpretations**: Complex images may be incorrectly identified (e.g., X-ray images mistaken for historical artifacts)
|
|
336
|
+
- **Language Errors**: Transcription accuracy depends on audio quality, accents, and background noise
|
|
337
|
+
- **Context Confusion**: AI may apply inappropriate cultural/historical context to technical content
|
|
338
|
+
- **Model Dependence**: Results vary significantly between different AI models (gemma3, llava, whisper versions)
|
|
339
|
+
- **Bias and Inconsistency**: Models may exhibit cultural, linguistic, or domain-specific biases
|
|
340
|
+
|
|
341
|
+
#### Critical Applications Warning:
|
|
342
|
+
**🚨 DO NOT USE for critical applications** such as:
|
|
343
|
+
- Medical diagnosis or medical image interpretation
|
|
344
|
+
- Legal document analysis requiring perfect accuracy
|
|
345
|
+
- Financial data extraction where errors have monetary impact
|
|
346
|
+
- Security/safety systems where false positives/negatives are dangerous
|
|
347
|
+
- Academic research requiring citation-quality accuracy
|
|
348
|
+
|
|
349
|
+
#### Best Practices:
|
|
350
|
+
- **Always validate AI results** against source material when accuracy matters
|
|
351
|
+
- **Use confidence scoring** to identify potentially unreliable results
|
|
352
|
+
- **Cross-reference** with traditional OCR/transcription tools for important content
|
|
353
|
+
- **Human review** recommended for any production use case
|
|
354
|
+
- **Test thoroughly** with your specific content types and use cases
|
|
355
|
+
- **Fallback options**: Keep traditional OCR (EasyOCR) available as backup
|
|
356
|
+
|
|
357
|
+
#### Recommended Use Cases:
|
|
358
|
+
✅ Content discovery and initial text extraction
|
|
359
|
+
✅ Batch processing of low-stakes content
|
|
360
|
+
✅ Development and prototyping workflows
|
|
361
|
+
✅ Personal document organization
|
|
362
|
+
✅ Educational and learning projects
|
|
363
|
+
|
|
282
364
|
## 📖 Full Examples
|
|
283
365
|
|
|
284
366
|
### Accessing Examples After Installation
|
|
@@ -323,7 +405,13 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
323
405
|
- ✅ **NEW**: Caption length control (short/medium/long)
|
|
324
406
|
- ✅ **NEW**: Temperature and token limit configuration
|
|
325
407
|
- ✅ **NEW**: Command-line OCR example script with full parameter support
|
|
408
|
+
- ✅ **NEW**: **Confidence Scoring System** - Advanced AI hallucination detection
|
|
409
|
+
- ✅ **NEW**: `xtxt_image_with_confidence()` - Returns text + confidence score
|
|
410
|
+
- ✅ **NEW**: Automatic rejection of low-confidence results with configurable thresholds
|
|
411
|
+
- ✅ **NEW**: 60+ hallucination patterns detected (historical, artistic, fantasy, scientific)
|
|
412
|
+
- ✅ **NEW**: `--show-confidence` and `--confidence` CLI parameters
|
|
326
413
|
- ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
|
|
414
|
+
- ✅ **ENHANCED**: Comprehensive AI safety warnings and best practices documentation
|
|
327
415
|
- ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
|
|
328
416
|
|
|
329
417
|
### v0.2.4
|
|
@@ -4,12 +4,14 @@ from .core import xtxt, extxt_available_formats, xtxt_from_url
|
|
|
4
4
|
try:
|
|
5
5
|
from .estrattori.ocr_ollama import (
|
|
6
6
|
set_ollama_model, get_ollama_model, xtxt_image_describe,
|
|
7
|
-
set_ollama_config, get_ollama_config, reset_ollama_config
|
|
7
|
+
set_ollama_config, get_ollama_config, reset_ollama_config,
|
|
8
|
+
xtxt_image_with_confidence
|
|
8
9
|
)
|
|
9
10
|
__all__ = [
|
|
10
11
|
"xtxt", "extxt_available_formats", "xtxt_from_url",
|
|
11
12
|
"set_ollama_model", "get_ollama_model", "xtxt_image_describe",
|
|
12
|
-
"set_ollama_config", "get_ollama_config", "reset_ollama_config"
|
|
13
|
+
"set_ollama_config", "get_ollama_config", "reset_ollama_config",
|
|
14
|
+
"xtxt_image_with_confidence"
|
|
13
15
|
]
|
|
14
16
|
except ImportError:
|
|
15
17
|
__all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
|
|
@@ -0,0 +1,520 @@
|
|
|
1
|
+
# pyxtxt/extractors/image_ocr_ollama.py
|
|
2
|
+
from . import register_extractor
|
|
3
|
+
from io import BytesIO
|
|
4
|
+
import base64
|
|
5
|
+
|
|
6
|
+
try:
|
|
7
|
+
import ollama
|
|
8
|
+
from PIL import Image
|
|
9
|
+
except ImportError:
|
|
10
|
+
ollama = None
|
|
11
|
+
Image = None
|
|
12
|
+
|
|
13
|
+
# Global configuration for Ollama model and parameters
|
|
14
|
+
OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
|
|
15
|
+
OLLAMA_CONFIG = {
|
|
16
|
+
'language': 'auto', # Language hint for caption generation
|
|
17
|
+
'caption_length': 'medium', # short, medium, long
|
|
18
|
+
'style': 'descriptive', # descriptive, technical, simple, detailed
|
|
19
|
+
'temperature': 0.1, # Response creativity (0.0-1.0)
|
|
20
|
+
'max_tokens': 1500, # Maximum response length
|
|
21
|
+
'confidence_threshold': 0.7, # Minimum confidence for text extraction
|
|
22
|
+
'context': 'general' # Context hint: general, cookbook, document, diagram, etc.
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
def set_ollama_model(model_name: str):
|
|
26
|
+
"""
|
|
27
|
+
Set the Ollama model to use for OCR.
|
|
28
|
+
|
|
29
|
+
Recommended multimodal models:
|
|
30
|
+
- gemma3:4b (default, balanced speed/quality)
|
|
31
|
+
- gemma3:12b (higher quality, slower)
|
|
32
|
+
- gemma3:27b (best quality, very slow)
|
|
33
|
+
- llava:7b (alternative vision model)
|
|
34
|
+
- llava:13b (higher quality LLAVA)
|
|
35
|
+
"""
|
|
36
|
+
global OLLAMA_MODEL
|
|
37
|
+
OLLAMA_MODEL = model_name
|
|
38
|
+
print(f"✅ Ollama OCR model set to: {model_name}")
|
|
39
|
+
|
|
40
|
+
def get_ollama_model():
|
|
41
|
+
"""Get current Ollama model name"""
|
|
42
|
+
return OLLAMA_MODEL
|
|
43
|
+
|
|
44
|
+
def set_ollama_config(**kwargs):
|
|
45
|
+
"""
|
|
46
|
+
Configure Ollama LLM parameters for better caption generation.
|
|
47
|
+
|
|
48
|
+
Parameters:
|
|
49
|
+
- language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
|
|
50
|
+
- caption_length: Caption length ('short', 'medium', 'long')
|
|
51
|
+
- style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
|
|
52
|
+
- context: Content context ('general', 'cookbook', 'document', 'handwriting', 'technical')
|
|
53
|
+
- temperature: Response creativity 0.0-1.0 (default: 0.1)
|
|
54
|
+
- max_tokens: Maximum response length (default: 1500)
|
|
55
|
+
- confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
|
|
56
|
+
|
|
57
|
+
Examples:
|
|
58
|
+
set_ollama_config(language='italian', style='detailed')
|
|
59
|
+
set_ollama_config(context='document', caption_length='long')
|
|
60
|
+
set_ollama_config(context='handwriting', temperature=0.2)
|
|
61
|
+
"""
|
|
62
|
+
global OLLAMA_CONFIG
|
|
63
|
+
for key, value in kwargs.items():
|
|
64
|
+
if key in OLLAMA_CONFIG:
|
|
65
|
+
OLLAMA_CONFIG[key] = value
|
|
66
|
+
print(f"✅ Ollama config updated: {key} = {value}")
|
|
67
|
+
else:
|
|
68
|
+
print(f"⚠️ Unknown config parameter: {key}")
|
|
69
|
+
|
|
70
|
+
def get_ollama_config():
|
|
71
|
+
"""Get current Ollama configuration"""
|
|
72
|
+
return OLLAMA_CONFIG.copy()
|
|
73
|
+
|
|
74
|
+
def reset_ollama_config():
|
|
75
|
+
"""Reset Ollama configuration to defaults"""
|
|
76
|
+
global OLLAMA_CONFIG
|
|
77
|
+
OLLAMA_CONFIG = {
|
|
78
|
+
'language': 'auto',
|
|
79
|
+
'caption_length': 'medium',
|
|
80
|
+
'style': 'descriptive',
|
|
81
|
+
'temperature': 0.1,
|
|
82
|
+
'max_tokens': 1500,
|
|
83
|
+
'confidence_threshold': 0.7,
|
|
84
|
+
'context': 'general'
|
|
85
|
+
}
|
|
86
|
+
print("✅ Ollama configuration reset to defaults")
|
|
87
|
+
|
|
88
|
+
def _calculate_confidence_score(content: str, mode: str) -> float:
|
|
89
|
+
"""
|
|
90
|
+
Calculate confidence score (0.0-1.0) for OCR/caption quality.
|
|
91
|
+
|
|
92
|
+
Evaluates response based on:
|
|
93
|
+
- Content length and structure
|
|
94
|
+
- Presence of meaningful text patterns
|
|
95
|
+
- Absence of hallucination indicators
|
|
96
|
+
- Appropriate response format
|
|
97
|
+
"""
|
|
98
|
+
if not content or len(content.strip()) < 3:
|
|
99
|
+
return 0.0
|
|
100
|
+
|
|
101
|
+
content_lower = content.lower()
|
|
102
|
+
score = 0.5 # Base score
|
|
103
|
+
|
|
104
|
+
# Positive indicators (add to confidence)
|
|
105
|
+
positive_patterns = [
|
|
106
|
+
# Structured text indicators
|
|
107
|
+
(r'\b\d+\b', 0.1, "Contains numbers/measurements"),
|
|
108
|
+
(r'[.!?]', 0.1, "Has sentence punctuation"),
|
|
109
|
+
(r'\b(the|and|of|in|to|for|with|on|at|by)\b', 0.1, "Common words present"),
|
|
110
|
+
(r'\b[A-Z][a-z]+\b', 0.1, "Proper capitalization"),
|
|
111
|
+
(r'[,;:]', 0.05, "Has proper punctuation"),
|
|
112
|
+
|
|
113
|
+
# Content-specific indicators
|
|
114
|
+
(r'\b(text|document|page|title|heading)\b', 0.1, "Document references"),
|
|
115
|
+
(r'\b(image|photo|picture|shows|contains)\b', 0.1, "Visual descriptions"),
|
|
116
|
+
(r'\b\d+[%$€£¥]\b|\b[%$€£¥]\d+\b', 0.1, "Currency/percentages"),
|
|
117
|
+
(r'\b\d{1,2}[:/]\d{1,2}\b|\b\d{4}\b', 0.05, "Times/years"),
|
|
118
|
+
]
|
|
119
|
+
|
|
120
|
+
# Negative indicators (reduce confidence)
|
|
121
|
+
negative_patterns = [
|
|
122
|
+
# Historical/Archaeological hallucinations (common with X-rays, scans)
|
|
123
|
+
("ancient", 0.2, "May be hallucinating historical content"),
|
|
124
|
+
("egypt", 0.15, "Egyptian hallucination"),
|
|
125
|
+
("hieroglyph", 0.25, "Hieroglyphic hallucination"),
|
|
126
|
+
("papyrus", 0.2, "Papyrus hallucination"),
|
|
127
|
+
("scroll", 0.1, "Ancient scroll hallucination"),
|
|
128
|
+
("medieval", 0.15, "Medieval hallucination"),
|
|
129
|
+
("manuscript", 0.1, "Manuscript hallucination"),
|
|
130
|
+
("archaeological", 0.15, "Archaeological hallucination"),
|
|
131
|
+
("artifact", 0.12, "Artifact hallucination"),
|
|
132
|
+
("roman", 0.1, "Roman historical hallucination"),
|
|
133
|
+
("greek", 0.1, "Greek historical hallucination"),
|
|
134
|
+
("biblical", 0.15, "Religious historical hallucination"),
|
|
135
|
+
("stone tablet", 0.2, "Stone tablet hallucination"),
|
|
136
|
+
("carved", 0.1, "Carving hallucination"),
|
|
137
|
+
|
|
138
|
+
# Art/Cultural hallucinations (common with medical images, diagrams)
|
|
139
|
+
("painting", 0.1, "Artistic interpretation hallucination"),
|
|
140
|
+
("artwork", 0.12, "Artwork hallucination"),
|
|
141
|
+
("masterpiece", 0.15, "Art masterpiece hallucination"),
|
|
142
|
+
("renaissance", 0.15, "Renaissance art hallucination"),
|
|
143
|
+
("portrait", 0.08, "Portrait hallucination"),
|
|
144
|
+
("landscape", 0.08, "Landscape hallucination"),
|
|
145
|
+
("abstract art", 0.12, "Abstract art hallucination"),
|
|
146
|
+
|
|
147
|
+
# Fantasy/Fictional content (can occur with any unclear image)
|
|
148
|
+
("mystical", 0.15, "Fantasy content hallucination"),
|
|
149
|
+
("magical", 0.15, "Magical content hallucination"),
|
|
150
|
+
("mythological", 0.15, "Mythology hallucination"),
|
|
151
|
+
("fairy tale", 0.15, "Fairy tale hallucination"),
|
|
152
|
+
("legend", 0.1, "Legend/folklore hallucination"),
|
|
153
|
+
("dragon", 0.2, "Dragon hallucination"),
|
|
154
|
+
("wizard", 0.15, "Fantasy character hallucination"),
|
|
155
|
+
|
|
156
|
+
# Scientific misinterpretations (X-rays as geological, etc.)
|
|
157
|
+
("fossil", 0.15, "Fossil hallucination"),
|
|
158
|
+
("geological", 0.1, "Geological misinterpretation"),
|
|
159
|
+
("mineral", 0.1, "Mineral hallucination"),
|
|
160
|
+
("crystal", 0.1, "Crystal hallucination"),
|
|
161
|
+
("rock formation", 0.12, "Rock formation hallucination"),
|
|
162
|
+
("sediment", 0.1, "Sediment hallucination"),
|
|
163
|
+
|
|
164
|
+
# Vague/uncertain language
|
|
165
|
+
("unclear", 0.1, "Uncertainty indicator"),
|
|
166
|
+
("difficult to read", 0.1, "Reading difficulty"),
|
|
167
|
+
("appears to be", 0.05, "Tentative language"),
|
|
168
|
+
("seems to", 0.05, "Uncertain language"),
|
|
169
|
+
("might be", 0.1, "Possibility language"),
|
|
170
|
+
("possibly", 0.1, "Uncertain language"),
|
|
171
|
+
("probably", 0.08, "Probability language"),
|
|
172
|
+
("looks like", 0.08, "Appearance-based guess"),
|
|
173
|
+
("reminds me of", 0.1, "Subjective association"),
|
|
174
|
+
("similar to", 0.05, "Similarity guess"),
|
|
175
|
+
|
|
176
|
+
# Quality/visibility issues (legitimate but indicate uncertainty)
|
|
177
|
+
("blurry", 0.05, "Image quality issue"),
|
|
178
|
+
("faded", 0.05, "Faded content"),
|
|
179
|
+
("damaged", 0.05, "Damaged content"),
|
|
180
|
+
("corrupted", 0.1, "Corrupted content"),
|
|
181
|
+
("low resolution", 0.08, "Resolution issue"),
|
|
182
|
+
("hard to make out", 0.1, "Visibility issue"),
|
|
183
|
+
|
|
184
|
+
# Generic/evasive responses
|
|
185
|
+
("sorry, i cannot", 0.3, "Refusal response"),
|
|
186
|
+
("cannot determine", 0.2, "Unable to process"),
|
|
187
|
+
("unable to identify", 0.15, "Identification failure"),
|
|
188
|
+
("not clear enough", 0.1, "Clarity issue"),
|
|
189
|
+
("too dark", 0.08, "Darkness issue"),
|
|
190
|
+
("too bright", 0.08, "Brightness issue"),
|
|
191
|
+
("no visible text", 0.0, "No text found - appropriate response"),
|
|
192
|
+
|
|
193
|
+
# Repetitive/nonsensical content (AI breakdown indicators)
|
|
194
|
+
("lorem ipsum", 0.2, "Placeholder text hallucination"),
|
|
195
|
+
("test test test", 0.25, "Repetitive test pattern"),
|
|
196
|
+
("abc abc abc", 0.2, "Repetitive pattern"),
|
|
197
|
+
("error error", 0.15, "Error message hallucination"),
|
|
198
|
+
]
|
|
199
|
+
|
|
200
|
+
import re
|
|
201
|
+
|
|
202
|
+
# Apply positive pattern scoring
|
|
203
|
+
for pattern, bonus, description in positive_patterns:
|
|
204
|
+
if re.search(pattern, content_lower):
|
|
205
|
+
score += bonus
|
|
206
|
+
|
|
207
|
+
# Apply negative pattern scoring
|
|
208
|
+
for pattern, penalty, description in negative_patterns:
|
|
209
|
+
if pattern in content_lower:
|
|
210
|
+
score -= penalty
|
|
211
|
+
print(f"⚠️ Confidence penalty ({penalty}): {description}")
|
|
212
|
+
|
|
213
|
+
# Length-based adjustments
|
|
214
|
+
if len(content) > 200: # Substantial content
|
|
215
|
+
score += 0.1
|
|
216
|
+
elif len(content) < 20: # Very short responses are suspicious
|
|
217
|
+
score -= 0.2
|
|
218
|
+
|
|
219
|
+
# Mode-specific adjustments
|
|
220
|
+
if mode == "describe":
|
|
221
|
+
# Check for proper formatting in describe mode
|
|
222
|
+
if "TEXT:" in content and "DESCRIPTION:" in content:
|
|
223
|
+
score += 0.1 # Proper format bonus
|
|
224
|
+
elif len(content) > 50: # Has substantial descriptive content
|
|
225
|
+
score += 0.05
|
|
226
|
+
|
|
227
|
+
# Context coherence check - very basic
|
|
228
|
+
words = content_lower.split()
|
|
229
|
+
if len(words) > 5:
|
|
230
|
+
# Check for repetitive patterns (hallucination indicator)
|
|
231
|
+
unique_words = set(words)
|
|
232
|
+
repetition_ratio = len(unique_words) / len(words)
|
|
233
|
+
if repetition_ratio < 0.3: # Too repetitive
|
|
234
|
+
score -= 0.2
|
|
235
|
+
|
|
236
|
+
# Clamp to valid range
|
|
237
|
+
return max(0.0, min(1.0, score))
|
|
238
|
+
|
|
239
|
+
if ollama and Image:
|
|
240
|
+
def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
|
|
241
|
+
"""
|
|
242
|
+
Extract text from images using Ollama with multimodal models.
|
|
243
|
+
|
|
244
|
+
Args:
|
|
245
|
+
file_buffer: Image file buffer
|
|
246
|
+
mode: "ocr" (text only) or "describe" (text + description)
|
|
247
|
+
model: Override default model (optional)
|
|
248
|
+
"""
|
|
249
|
+
try:
|
|
250
|
+
# Use specified model or global default
|
|
251
|
+
current_model = model or OLLAMA_MODEL
|
|
252
|
+
|
|
253
|
+
# Convert buffer to PIL Image
|
|
254
|
+
# Reset buffer position if it has read method
|
|
255
|
+
if hasattr(file_buffer, 'seek'):
|
|
256
|
+
file_buffer.seek(0)
|
|
257
|
+
image_data = file_buffer.read()
|
|
258
|
+
image = Image.open(BytesIO(image_data))
|
|
259
|
+
|
|
260
|
+
# Convert to RGB if needed
|
|
261
|
+
if image.mode != 'RGB':
|
|
262
|
+
image = image.convert('RGB')
|
|
263
|
+
|
|
264
|
+
# Convert image to base64
|
|
265
|
+
buffered = BytesIO()
|
|
266
|
+
image.save(buffered, format="PNG")
|
|
267
|
+
img_base64 = base64.b64encode(buffered.getvalue()).decode()
|
|
268
|
+
|
|
269
|
+
# Get current configuration
|
|
270
|
+
config = OLLAMA_CONFIG
|
|
271
|
+
|
|
272
|
+
# Build language hint
|
|
273
|
+
lang_hint = ""
|
|
274
|
+
if config['language'] != 'auto':
|
|
275
|
+
lang_hint = f"Text language: {config['language']}. "
|
|
276
|
+
|
|
277
|
+
# Different prompts based on mode
|
|
278
|
+
if mode == "ocr":
|
|
279
|
+
prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
|
|
280
|
+
- Titles, headings, and main text content
|
|
281
|
+
- Small print, captions, labels, and annotations
|
|
282
|
+
- Numbers, measurements, quantities, and symbols
|
|
283
|
+
- Menu items, ingredient lists, cooking instructions
|
|
284
|
+
- Any text in boxes, speech bubbles, or decorative elements
|
|
285
|
+
|
|
286
|
+
IMPORTANT:
|
|
287
|
+
- Read carefully and include even small or partially visible text
|
|
288
|
+
- Preserve the original formatting and line breaks where possible
|
|
289
|
+
- {lang_hint}Process the text from left to right, top to bottom
|
|
290
|
+
- If you cannot find any readable text at all, respond with 'NO_TEXT_FOUND'
|
|
291
|
+
|
|
292
|
+
Extracted text:"""
|
|
293
|
+
|
|
294
|
+
else: # mode == "describe"
|
|
295
|
+
# Build style-specific prompts
|
|
296
|
+
style_prompts = {
|
|
297
|
+
'descriptive': "Provide a clear, descriptive explanation",
|
|
298
|
+
'technical': "Use technical terminology and precise descriptions",
|
|
299
|
+
'simple': "Use simple, easy-to-understand language",
|
|
300
|
+
'detailed': "Provide comprehensive details about all visual elements"
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
length_hints = {
|
|
304
|
+
'short': "Keep descriptions brief (1-2 sentences)",
|
|
305
|
+
'medium': "Provide moderate detail (2-4 sentences)",
|
|
306
|
+
'long': "Give comprehensive descriptions (4-8 sentences)"
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
|
|
310
|
+
length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
|
|
311
|
+
|
|
312
|
+
# Context-specific hints (only when explicitly set)
|
|
313
|
+
context_hint = ""
|
|
314
|
+
context = config.get('context', 'general').lower()
|
|
315
|
+
if context == 'cookbook' or context == 'recipe':
|
|
316
|
+
context_hint = """
|
|
317
|
+
- If this appears to be a recipe/cookbook page, include: ingredients, cooking steps, quantities, cooking times
|
|
318
|
+
- Mention any photos of prepared dishes or cooking techniques shown
|
|
319
|
+
- Note any special formatting like ingredient lists, step numbers, or cooking tips"""
|
|
320
|
+
elif context == 'document':
|
|
321
|
+
context_hint = """
|
|
322
|
+
- Focus on document structure: headers, paragraphs, sections, page numbers
|
|
323
|
+
- Note any official formatting, letterheads, signatures, or stamps"""
|
|
324
|
+
elif context == 'handwriting' or context == 'notes':
|
|
325
|
+
context_hint = """
|
|
326
|
+
- Pay special attention to handwritten text which may be harder to read
|
|
327
|
+
- Note any sketches, diagrams, or informal formatting typical of personal notes"""
|
|
328
|
+
elif context == 'technical' or context == 'diagram':
|
|
329
|
+
context_hint = """
|
|
330
|
+
- Focus on technical elements: labels, measurements, specifications, diagrams
|
|
331
|
+
- Include any mathematical formulas, technical symbols, or engineering notations"""
|
|
332
|
+
|
|
333
|
+
prompt = f"""Analyze this image and provide:
|
|
334
|
+
1. All visible text exactly as written (preserve formatting, line breaks, bullet points)
|
|
335
|
+
2. Image description following these guidelines:
|
|
336
|
+
- {style_instruction}
|
|
337
|
+
- {length_instruction}
|
|
338
|
+
- {lang_hint}Focus on key visual elements, layout, and context{context_hint}
|
|
339
|
+
|
|
340
|
+
Format:
|
|
341
|
+
TEXT: [all visible text here, or NO_TEXT_FOUND if none]
|
|
342
|
+
DESCRIPTION: [image description following the guidelines above]"""
|
|
343
|
+
|
|
344
|
+
# Send request to Ollama with configured parameters
|
|
345
|
+
response = ollama.generate(
|
|
346
|
+
model=current_model,
|
|
347
|
+
prompt=prompt,
|
|
348
|
+
images=[img_base64],
|
|
349
|
+
options={
|
|
350
|
+
'temperature': config['temperature'],
|
|
351
|
+
'top_p': 0.9,
|
|
352
|
+
'num_predict': config['max_tokens']
|
|
353
|
+
}
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
# Extract and clean response
|
|
357
|
+
extracted_content = response.get('response', '').strip()
|
|
358
|
+
|
|
359
|
+
# Calculate confidence score based on response quality indicators
|
|
360
|
+
confidence_score = _calculate_confidence_score(extracted_content, mode)
|
|
361
|
+
|
|
362
|
+
# Check confidence threshold
|
|
363
|
+
if confidence_score < config['confidence_threshold']:
|
|
364
|
+
print(f"⚠️ Low confidence ({confidence_score:.2f} < {config['confidence_threshold']}): {extracted_content[:100]}...")
|
|
365
|
+
if mode == "ocr":
|
|
366
|
+
return "" # Return empty for OCR if below threshold
|
|
367
|
+
else:
|
|
368
|
+
# For describe mode, add warning prefix
|
|
369
|
+
extracted_content = f"[LOW_CONFIDENCE_{confidence_score:.2f}] {extracted_content}"
|
|
370
|
+
else:
|
|
371
|
+
print(f"✅ Good confidence ({confidence_score:.2f}): Processing successful")
|
|
372
|
+
|
|
373
|
+
# Handle no-text case for OCR mode
|
|
374
|
+
if mode == "ocr" and ('NO_TEXT_FOUND' in extracted_content or len(extracted_content) < 3):
|
|
375
|
+
return ""
|
|
376
|
+
|
|
377
|
+
return extracted_content
|
|
378
|
+
|
|
379
|
+
except Exception as e:
|
|
380
|
+
print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
|
|
381
|
+
return ""
|
|
382
|
+
|
|
383
|
+
def xtxt_image_ocr_ollama_with_confidence(file_buffer, mode="ocr", model=None):
|
|
384
|
+
"""
|
|
385
|
+
Version of OCR-Ollama that returns both text and confidence score.
|
|
386
|
+
|
|
387
|
+
Returns:
|
|
388
|
+
tuple: (extracted_text, confidence_score)
|
|
389
|
+
"""
|
|
390
|
+
try:
|
|
391
|
+
# Use specified model or global default
|
|
392
|
+
current_model = model or OLLAMA_MODEL
|
|
393
|
+
|
|
394
|
+
# Convert buffer to PIL Image
|
|
395
|
+
if hasattr(file_buffer, 'seek'):
|
|
396
|
+
file_buffer.seek(0)
|
|
397
|
+
image_data = file_buffer.read()
|
|
398
|
+
image = Image.open(BytesIO(image_data))
|
|
399
|
+
|
|
400
|
+
# Convert to RGB if needed
|
|
401
|
+
if image.mode != 'RGB':
|
|
402
|
+
image = image.convert('RGB')
|
|
403
|
+
|
|
404
|
+
# Convert image to base64
|
|
405
|
+
buffered = BytesIO()
|
|
406
|
+
image.save(buffered, format="PNG")
|
|
407
|
+
img_base64 = base64.b64encode(buffered.getvalue()).decode()
|
|
408
|
+
|
|
409
|
+
# Get current configuration
|
|
410
|
+
config = OLLAMA_CONFIG
|
|
411
|
+
|
|
412
|
+
# Build prompts (same logic as main function)
|
|
413
|
+
lang_hint = ""
|
|
414
|
+
if config['language'] != 'auto':
|
|
415
|
+
lang_hint = f"Text language: {config['language']}. "
|
|
416
|
+
|
|
417
|
+
if mode == "ocr":
|
|
418
|
+
prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
|
|
419
|
+
- Titles, headings, and main text content
|
|
420
|
+
- Small print, captions, labels, and annotations
|
|
421
|
+
- Numbers, measurements, quantities, and symbols
|
|
422
|
+
- Menu items, ingredient lists, cooking instructions
|
|
423
|
+
- Any text in boxes, speech bubbles, or decorative elements
|
|
424
|
+
|
|
425
|
+
IMPORTANT:
|
|
426
|
+
- Read carefully and include even small or partially visible text
|
|
427
|
+
- Preserve the original formatting and line breaks where possible
|
|
428
|
+
- {lang_hint}Process the text from left to right, top to bottom
|
|
429
|
+
- If you cannot find any readable text at all, respond with 'NO_TEXT_FOUND'
|
|
430
|
+
|
|
431
|
+
Extracted text:"""
|
|
432
|
+
else: # describe mode
|
|
433
|
+
style_prompts = {
|
|
434
|
+
'descriptive': "Provide a clear, descriptive explanation",
|
|
435
|
+
'technical': "Use technical terminology and precise descriptions",
|
|
436
|
+
'simple': "Use simple, easy-to-understand language",
|
|
437
|
+
'detailed': "Provide comprehensive details about all visual elements"
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
length_hints = {
|
|
441
|
+
'short': "Keep descriptions brief (1-2 sentences)",
|
|
442
|
+
'medium': "Provide moderate detail (2-4 sentences)",
|
|
443
|
+
'long': "Give comprehensive descriptions (4-8 sentences)"
|
|
444
|
+
}
|
|
445
|
+
|
|
446
|
+
style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
|
|
447
|
+
length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
|
|
448
|
+
|
|
449
|
+
prompt = f"""Analyze this image and provide:
|
|
450
|
+
1. All visible text exactly as written (preserve formatting, line breaks, bullet points)
|
|
451
|
+
2. Image description following these guidelines:
|
|
452
|
+
- {style_instruction}
|
|
453
|
+
- {length_instruction}
|
|
454
|
+
- {lang_hint}Focus on key visual elements, layout, and context
|
|
455
|
+
|
|
456
|
+
Format:
|
|
457
|
+
TEXT: [all visible text here, or NO_TEXT_FOUND if none]
|
|
458
|
+
DESCRIPTION: [image description following the guidelines above]"""
|
|
459
|
+
|
|
460
|
+
# Send request to Ollama
|
|
461
|
+
response = ollama.generate(
|
|
462
|
+
model=current_model,
|
|
463
|
+
prompt=prompt,
|
|
464
|
+
images=[img_base64],
|
|
465
|
+
options={
|
|
466
|
+
'temperature': config['temperature'],
|
|
467
|
+
'top_p': 0.9,
|
|
468
|
+
'num_predict': config['max_tokens']
|
|
469
|
+
}
|
|
470
|
+
)
|
|
471
|
+
|
|
472
|
+
# Extract response
|
|
473
|
+
extracted_content = response.get('response', '').strip()
|
|
474
|
+
|
|
475
|
+
# Calculate confidence without threshold filtering
|
|
476
|
+
confidence_score = _calculate_confidence_score(extracted_content, mode)
|
|
477
|
+
|
|
478
|
+
# Return both text and confidence
|
|
479
|
+
return extracted_content, confidence_score
|
|
480
|
+
|
|
481
|
+
except Exception as e:
|
|
482
|
+
print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
|
|
483
|
+
return "", 0.0
|
|
484
|
+
|
|
485
|
+
# Wrapper functions for each mode
|
|
486
|
+
def xtxt_image_ocr_only(file_input):
|
|
487
|
+
"""Traditional OCR: extract only visible text using Ollama"""
|
|
488
|
+
# Handle both file paths and buffers
|
|
489
|
+
if isinstance(file_input, str):
|
|
490
|
+
with open(file_input, 'rb') as f:
|
|
491
|
+
return xtxt_image_ocr_ollama(f, mode="ocr")
|
|
492
|
+
else:
|
|
493
|
+
return xtxt_image_ocr_ollama(file_input, mode="ocr")
|
|
494
|
+
|
|
495
|
+
def xtxt_image_describe(file_input):
|
|
496
|
+
"""OCR + Description: text + image context using Ollama"""
|
|
497
|
+
# Handle both file paths and buffers
|
|
498
|
+
if isinstance(file_input, str):
|
|
499
|
+
with open(file_input, 'rb') as f:
|
|
500
|
+
return xtxt_image_ocr_ollama(f, mode="describe")
|
|
501
|
+
else:
|
|
502
|
+
return xtxt_image_ocr_ollama(file_input, mode="describe")
|
|
503
|
+
|
|
504
|
+
def xtxt_image_with_confidence(file_input, mode="ocr"):
|
|
505
|
+
"""Get both text and confidence score from image OCR"""
|
|
506
|
+
if isinstance(file_input, str):
|
|
507
|
+
with open(file_input, 'rb') as f:
|
|
508
|
+
return xtxt_image_ocr_ollama_with_confidence(f, mode=mode)
|
|
509
|
+
else:
|
|
510
|
+
return xtxt_image_ocr_ollama_with_confidence(file_input, mode=mode)
|
|
511
|
+
|
|
512
|
+
# Register OCR-only version as default
|
|
513
|
+
# Note: Will override traditional EasyOCR if both modules are loaded
|
|
514
|
+
image_formats = [
|
|
515
|
+
"image/jpeg", "image/jpg", "image/png",
|
|
516
|
+
"image/bmp", "image/tiff", "image/webp"
|
|
517
|
+
]
|
|
518
|
+
|
|
519
|
+
for format_type in image_formats:
|
|
520
|
+
register_extractor(format_type, xtxt_image_ocr_only, name="OCR-Ollama")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.4
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -308,6 +308,45 @@ image_response = requests.get("https://example.com/document.png")
|
|
|
308
308
|
text = xtxt(image_response.content)
|
|
309
309
|
```
|
|
310
310
|
|
|
311
|
+
### AI OCR Confidence Scoring (NEW)
|
|
312
|
+
|
|
313
|
+
⚠️ **IMPORTANT**: AI-powered OCR is experimental technology that may produce errors, hallucinations, or misinterpretations. Always validate results for critical applications.
|
|
314
|
+
|
|
315
|
+
The OCR-Ollama system includes confidence scoring to help identify unreliable results:
|
|
316
|
+
|
|
317
|
+
```python
|
|
318
|
+
from pyxtxt import xtxt_image_with_confidence, set_ollama_config
|
|
319
|
+
|
|
320
|
+
# Configure confidence threshold (0.0-1.0, default: 0.7)
|
|
321
|
+
set_ollama_config(confidence_threshold=0.8) # More restrictive
|
|
322
|
+
|
|
323
|
+
# Get text with confidence score
|
|
324
|
+
text, confidence = xtxt_image_with_confidence("document.png", mode="ocr")
|
|
325
|
+
print(f"Confidence: {confidence:.2f} ({confidence*100:.1f}%)")
|
|
326
|
+
print(f"Text: {text}")
|
|
327
|
+
|
|
328
|
+
# Check if reliable
|
|
329
|
+
if confidence < 0.7:
|
|
330
|
+
print("⚠️ Low confidence - result may be unreliable")
|
|
331
|
+
print("Consider using traditional OCR or manual verification")
|
|
332
|
+
else:
|
|
333
|
+
print("✅ Good confidence - result likely reliable")
|
|
334
|
+
```
|
|
335
|
+
|
|
336
|
+
#### Confidence Scoring Features
|
|
337
|
+
|
|
338
|
+
- **Hallucination Detection**: Penalizes historical/cultural references that often indicate AI misinterpretation
|
|
339
|
+
- **Pattern Recognition**: Rewards structured text (numbers, punctuation, proper formatting)
|
|
340
|
+
- **Uncertainty Detection**: Flags vague language ("appears to be", "seems like", etc.)
|
|
341
|
+
- **Quality Assessment**: Considers content length, repetition, and coherence
|
|
342
|
+
|
|
343
|
+
#### Common Hallucination Patterns (Automatically Detected)
|
|
344
|
+
- **Historical Content**: "ancient", "medieval", "Egyptian papyrus", "hieroglyphs"
|
|
345
|
+
- **Artistic Interpretations**: "painting", "artwork", "masterpiece", "Renaissance"
|
|
346
|
+
- **Fantasy Content**: "mystical", "magical", "dragon", "wizard"
|
|
347
|
+
- **Scientific Misinterpretation**: "fossil", "geological formation", "crystal structure"
|
|
348
|
+
- **Vague Language**: "unclear", "difficult to read", "appears to be"
|
|
349
|
+
|
|
311
350
|
### Command-Line OCR Example
|
|
312
351
|
|
|
313
352
|
A complete example script for command-line usage is available:
|
|
@@ -324,6 +363,11 @@ with open("ocr_example.py", "wb") as f:
|
|
|
324
363
|
# python ocr_example.py document.png
|
|
325
364
|
# python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
|
|
326
365
|
# python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
|
|
366
|
+
|
|
367
|
+
# NEW: Confidence scoring examples
|
|
368
|
+
# python ocr_example.py suspicious.png --show-confidence
|
|
369
|
+
# python ocr_example.py medical.png --confidence=0.9 --show-confidence
|
|
370
|
+
# python ocr_example.py diagram.png --confidence=0.5 --mode=describe --show-confidence
|
|
327
371
|
```
|
|
328
372
|
|
|
329
373
|
The script supports:
|
|
@@ -333,6 +377,8 @@ The script supports:
|
|
|
333
377
|
- **Style control**: descriptive, technical, simple, detailed
|
|
334
378
|
- **Length control**: short, medium, long captions
|
|
335
379
|
- **Temperature**: Adjust LLM creativity (0.0-1.0)
|
|
380
|
+
- **Confidence scoring**: Set threshold and display confidence scores
|
|
381
|
+
- **Quality filtering**: Automatically reject low-confidence results
|
|
336
382
|
|
|
337
383
|
### Show Available Formats
|
|
338
384
|
```python
|
|
@@ -376,6 +422,7 @@ text = xtxt(attachment_bytes)
|
|
|
376
422
|
|
|
377
423
|
## ⚠️ Known Limitations
|
|
378
424
|
|
|
425
|
+
### General Limitations
|
|
379
426
|
- **Legacy file detection**: When using raw streams without filenames, legacy files (.doc, .xls, .ppt) may not be correctly detected due to identical file signatures in libmagic
|
|
380
427
|
- **Filename hints recommended**: When available, providing original filenames improves detection accuracy
|
|
381
428
|
- **MSWrite .doc files**: Require `antiword` installation:
|
|
@@ -383,6 +430,41 @@ text = xtxt(attachment_bytes)
|
|
|
383
430
|
sudo apt-get update && sudo apt-get install antiword
|
|
384
431
|
```
|
|
385
432
|
|
|
433
|
+
### 🤖 AI-Powered Features - Important Warnings
|
|
434
|
+
|
|
435
|
+
**⚠️ EXPERIMENTAL TECHNOLOGY**: AI-powered features (OCR-Ollama, audio transcription) are based on machine learning models and may produce:
|
|
436
|
+
|
|
437
|
+
#### Potential Issues:
|
|
438
|
+
- **Hallucinations**: AI may "see" or "hear" content that isn't actually present
|
|
439
|
+
- **Misinterpretations**: Complex images may be incorrectly identified (e.g., X-ray images mistaken for historical artifacts)
|
|
440
|
+
- **Language Errors**: Transcription accuracy depends on audio quality, accents, and background noise
|
|
441
|
+
- **Context Confusion**: AI may apply inappropriate cultural/historical context to technical content
|
|
442
|
+
- **Model Dependence**: Results vary significantly between different AI models (gemma3, llava, whisper versions)
|
|
443
|
+
- **Bias and Inconsistency**: Models may exhibit cultural, linguistic, or domain-specific biases
|
|
444
|
+
|
|
445
|
+
#### Critical Applications Warning:
|
|
446
|
+
**🚨 DO NOT USE for critical applications** such as:
|
|
447
|
+
- Medical diagnosis or medical image interpretation
|
|
448
|
+
- Legal document analysis requiring perfect accuracy
|
|
449
|
+
- Financial data extraction where errors have monetary impact
|
|
450
|
+
- Security/safety systems where false positives/negatives are dangerous
|
|
451
|
+
- Academic research requiring citation-quality accuracy
|
|
452
|
+
|
|
453
|
+
#### Best Practices:
|
|
454
|
+
- **Always validate AI results** against source material when accuracy matters
|
|
455
|
+
- **Use confidence scoring** to identify potentially unreliable results
|
|
456
|
+
- **Cross-reference** with traditional OCR/transcription tools for important content
|
|
457
|
+
- **Human review** recommended for any production use case
|
|
458
|
+
- **Test thoroughly** with your specific content types and use cases
|
|
459
|
+
- **Fallback options**: Keep traditional OCR (EasyOCR) available as backup
|
|
460
|
+
|
|
461
|
+
#### Recommended Use Cases:
|
|
462
|
+
✅ Content discovery and initial text extraction
|
|
463
|
+
✅ Batch processing of low-stakes content
|
|
464
|
+
✅ Development and prototyping workflows
|
|
465
|
+
✅ Personal document organization
|
|
466
|
+
✅ Educational and learning projects
|
|
467
|
+
|
|
386
468
|
## 📖 Full Examples
|
|
387
469
|
|
|
388
470
|
### Accessing Examples After Installation
|
|
@@ -427,7 +509,13 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
427
509
|
- ✅ **NEW**: Caption length control (short/medium/long)
|
|
428
510
|
- ✅ **NEW**: Temperature and token limit configuration
|
|
429
511
|
- ✅ **NEW**: Command-line OCR example script with full parameter support
|
|
512
|
+
- ✅ **NEW**: **Confidence Scoring System** - Advanced AI hallucination detection
|
|
513
|
+
- ✅ **NEW**: `xtxt_image_with_confidence()` - Returns text + confidence score
|
|
514
|
+
- ✅ **NEW**: Automatic rejection of low-confidence results with configurable thresholds
|
|
515
|
+
- ✅ **NEW**: 60+ hallucination patterns detected (historical, artistic, fantasy, scientific)
|
|
516
|
+
- ✅ **NEW**: `--show-confidence` and `--confidence` CLI parameters
|
|
430
517
|
- ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
|
|
518
|
+
- ✅ **ENHANCED**: Comprehensive AI safety warnings and best practices documentation
|
|
431
519
|
- ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
|
|
432
520
|
|
|
433
521
|
### v0.2.4
|
|
@@ -1,202 +0,0 @@
|
|
|
1
|
-
# pyxtxt/extractors/image_ocr_ollama.py
|
|
2
|
-
from . import register_extractor
|
|
3
|
-
from io import BytesIO
|
|
4
|
-
import base64
|
|
5
|
-
|
|
6
|
-
try:
|
|
7
|
-
import ollama
|
|
8
|
-
from PIL import Image
|
|
9
|
-
except ImportError:
|
|
10
|
-
ollama = None
|
|
11
|
-
Image = None
|
|
12
|
-
|
|
13
|
-
# Global configuration for Ollama model and parameters
|
|
14
|
-
OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
|
|
15
|
-
OLLAMA_CONFIG = {
|
|
16
|
-
'language': 'auto', # Language hint for caption generation
|
|
17
|
-
'caption_length': 'medium', # short, medium, long
|
|
18
|
-
'style': 'descriptive', # descriptive, technical, simple, detailed
|
|
19
|
-
'temperature': 0.1, # Response creativity (0.0-1.0)
|
|
20
|
-
'max_tokens': 1500, # Maximum response length
|
|
21
|
-
'confidence_threshold': 0.7 # Minimum confidence for text extraction
|
|
22
|
-
}
|
|
23
|
-
|
|
24
|
-
def set_ollama_model(model_name: str):
|
|
25
|
-
"""
|
|
26
|
-
Set the Ollama model to use for OCR.
|
|
27
|
-
|
|
28
|
-
Recommended multimodal models:
|
|
29
|
-
- gemma3:4b (default, balanced speed/quality)
|
|
30
|
-
- gemma3:12b (higher quality, slower)
|
|
31
|
-
- gemma3:27b (best quality, very slow)
|
|
32
|
-
- llava:7b (alternative vision model)
|
|
33
|
-
- llava:13b (higher quality LLAVA)
|
|
34
|
-
"""
|
|
35
|
-
global OLLAMA_MODEL
|
|
36
|
-
OLLAMA_MODEL = model_name
|
|
37
|
-
print(f"✅ Ollama OCR model set to: {model_name}")
|
|
38
|
-
|
|
39
|
-
def get_ollama_model():
|
|
40
|
-
"""Get current Ollama model name"""
|
|
41
|
-
return OLLAMA_MODEL
|
|
42
|
-
|
|
43
|
-
def set_ollama_config(**kwargs):
|
|
44
|
-
"""
|
|
45
|
-
Configure Ollama LLM parameters for better caption generation.
|
|
46
|
-
|
|
47
|
-
Parameters:
|
|
48
|
-
- language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
|
|
49
|
-
- caption_length: Caption length ('short', 'medium', 'long')
|
|
50
|
-
- style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
|
|
51
|
-
- temperature: Response creativity 0.0-1.0 (default: 0.1)
|
|
52
|
-
- max_tokens: Maximum response length (default: 1500)
|
|
53
|
-
- confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
|
|
54
|
-
|
|
55
|
-
Examples:
|
|
56
|
-
set_ollama_config(language='italian', style='detailed')
|
|
57
|
-
set_ollama_config(caption_length='long', temperature=0.3)
|
|
58
|
-
"""
|
|
59
|
-
global OLLAMA_CONFIG
|
|
60
|
-
for key, value in kwargs.items():
|
|
61
|
-
if key in OLLAMA_CONFIG:
|
|
62
|
-
OLLAMA_CONFIG[key] = value
|
|
63
|
-
print(f"✅ Ollama config updated: {key} = {value}")
|
|
64
|
-
else:
|
|
65
|
-
print(f"⚠️ Unknown config parameter: {key}")
|
|
66
|
-
|
|
67
|
-
def get_ollama_config():
|
|
68
|
-
"""Get current Ollama configuration"""
|
|
69
|
-
return OLLAMA_CONFIG.copy()
|
|
70
|
-
|
|
71
|
-
def reset_ollama_config():
|
|
72
|
-
"""Reset Ollama configuration to defaults"""
|
|
73
|
-
global OLLAMA_CONFIG
|
|
74
|
-
OLLAMA_CONFIG = {
|
|
75
|
-
'language': 'auto',
|
|
76
|
-
'caption_length': 'medium',
|
|
77
|
-
'style': 'descriptive',
|
|
78
|
-
'temperature': 0.1,
|
|
79
|
-
'max_tokens': 1500,
|
|
80
|
-
'confidence_threshold': 0.7
|
|
81
|
-
}
|
|
82
|
-
print("✅ Ollama configuration reset to defaults")
|
|
83
|
-
|
|
84
|
-
if ollama and Image:
|
|
85
|
-
def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
|
|
86
|
-
"""
|
|
87
|
-
Extract text from images using Ollama with multimodal models.
|
|
88
|
-
|
|
89
|
-
Args:
|
|
90
|
-
file_buffer: Image file buffer
|
|
91
|
-
mode: "ocr" (text only) or "describe" (text + description)
|
|
92
|
-
model: Override default model (optional)
|
|
93
|
-
"""
|
|
94
|
-
try:
|
|
95
|
-
# Use specified model or global default
|
|
96
|
-
current_model = model or OLLAMA_MODEL
|
|
97
|
-
|
|
98
|
-
# Convert buffer to PIL Image
|
|
99
|
-
image = Image.open(BytesIO(file_buffer.read()))
|
|
100
|
-
|
|
101
|
-
# Convert to RGB if needed
|
|
102
|
-
if image.mode != 'RGB':
|
|
103
|
-
image = image.convert('RGB')
|
|
104
|
-
|
|
105
|
-
# Convert image to base64
|
|
106
|
-
buffered = BytesIO()
|
|
107
|
-
image.save(buffered, format="PNG")
|
|
108
|
-
img_base64 = base64.b64encode(buffered.getvalue()).decode()
|
|
109
|
-
|
|
110
|
-
# Get current configuration
|
|
111
|
-
config = OLLAMA_CONFIG
|
|
112
|
-
|
|
113
|
-
# Build language hint
|
|
114
|
-
lang_hint = ""
|
|
115
|
-
if config['language'] != 'auto':
|
|
116
|
-
lang_hint = f"Text language: {config['language']}. "
|
|
117
|
-
|
|
118
|
-
# Different prompts based on mode
|
|
119
|
-
if mode == "ocr":
|
|
120
|
-
prompt = f"""Extract ALL visible text from this image exactly as it appears.
|
|
121
|
-
Rules:
|
|
122
|
-
- Only return text that is actually written/printed in the image
|
|
123
|
-
- Preserve reading order (left to right, top to bottom)
|
|
124
|
-
- Maintain line breaks and formatting
|
|
125
|
-
- Include numbers, symbols, special characters
|
|
126
|
-
- Do NOT add descriptions, interpretations, or context
|
|
127
|
-
- {lang_hint}If no text is visible, return 'NO_TEXT_FOUND'
|
|
128
|
-
|
|
129
|
-
Extracted text:"""
|
|
130
|
-
|
|
131
|
-
else: # mode == "describe"
|
|
132
|
-
# Build style-specific prompts
|
|
133
|
-
style_prompts = {
|
|
134
|
-
'descriptive': "Provide a clear, descriptive explanation",
|
|
135
|
-
'technical': "Use technical terminology and precise descriptions",
|
|
136
|
-
'simple': "Use simple, easy-to-understand language",
|
|
137
|
-
'detailed': "Provide comprehensive details about all visual elements"
|
|
138
|
-
}
|
|
139
|
-
|
|
140
|
-
length_hints = {
|
|
141
|
-
'short': "Keep descriptions brief (1-2 sentences)",
|
|
142
|
-
'medium': "Provide moderate detail (2-4 sentences)",
|
|
143
|
-
'long': "Give comprehensive descriptions (4-8 sentences)"
|
|
144
|
-
}
|
|
145
|
-
|
|
146
|
-
style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
|
|
147
|
-
length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
|
|
148
|
-
|
|
149
|
-
prompt = f"""Analyze this image and provide:
|
|
150
|
-
1. All visible text exactly as written
|
|
151
|
-
2. Image description following these guidelines:
|
|
152
|
-
- {style_instruction}
|
|
153
|
-
- {length_instruction}
|
|
154
|
-
- {lang_hint}Focus on key visual elements, layout, and context
|
|
155
|
-
|
|
156
|
-
Format:
|
|
157
|
-
TEXT: [all visible text here, or NO_TEXT_FOUND if none]
|
|
158
|
-
DESCRIPTION: [image description following the guidelines above]"""
|
|
159
|
-
|
|
160
|
-
# Send request to Ollama with configured parameters
|
|
161
|
-
response = ollama.generate(
|
|
162
|
-
model=current_model,
|
|
163
|
-
prompt=prompt,
|
|
164
|
-
images=[img_base64],
|
|
165
|
-
options={
|
|
166
|
-
'temperature': config['temperature'],
|
|
167
|
-
'top_p': 0.9,
|
|
168
|
-
'num_predict': config['max_tokens']
|
|
169
|
-
}
|
|
170
|
-
)
|
|
171
|
-
|
|
172
|
-
# Extract and clean response
|
|
173
|
-
extracted_content = response.get('response', '').strip()
|
|
174
|
-
|
|
175
|
-
# Handle no-text case for OCR mode
|
|
176
|
-
if mode == "ocr" and ('NO_TEXT_FOUND' in extracted_content or len(extracted_content) < 3):
|
|
177
|
-
return ""
|
|
178
|
-
|
|
179
|
-
return extracted_content
|
|
180
|
-
|
|
181
|
-
except Exception as e:
|
|
182
|
-
print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
|
|
183
|
-
return ""
|
|
184
|
-
|
|
185
|
-
# Wrapper functions for each mode
|
|
186
|
-
def xtxt_image_ocr_only(file_buffer):
|
|
187
|
-
"""Traditional OCR: extract only visible text using Ollama"""
|
|
188
|
-
return xtxt_image_ocr_ollama(file_buffer, mode="ocr")
|
|
189
|
-
|
|
190
|
-
def xtxt_image_describe(file_buffer):
|
|
191
|
-
"""OCR + Description: text + image context using Ollama"""
|
|
192
|
-
return xtxt_image_ocr_ollama(file_buffer, mode="describe")
|
|
193
|
-
|
|
194
|
-
# Register OCR-only version as default
|
|
195
|
-
# Note: Will override traditional EasyOCR if both modules are loaded
|
|
196
|
-
image_formats = [
|
|
197
|
-
"image/jpeg", "image/jpg", "image/png",
|
|
198
|
-
"image/bmp", "image/tiff", "image/webp"
|
|
199
|
-
]
|
|
200
|
-
|
|
201
|
-
for format_type in image_formats:
|
|
202
|
-
register_extractor(format_type, xtxt_image_ocr_only, name="OCR-Ollama")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|