pyxtxt 0.3.1__tar.gz → 0.3.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. {pyxtxt-0.3.1/src/pyxtxt.egg-info → pyxtxt-0.3.3}/PKG-INFO +104 -21
  2. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/README.md +103 -20
  3. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/pyproject.toml +1 -1
  4. pyxtxt-0.3.3/src/pyxtxt/__init__.py +15 -0
  5. pyxtxt-0.3.3/src/pyxtxt/estrattori/ocr_ollama.py +245 -0
  6. {pyxtxt-0.3.1 → pyxtxt-0.3.3/src/pyxtxt.egg-info}/PKG-INFO +104 -21
  7. pyxtxt-0.3.1/src/pyxtxt/__init__.py +0 -8
  8. pyxtxt-0.3.1/src/pyxtxt/estrattori/ocr_ollama.py +0 -125
  9. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/LICENSE +0 -0
  10. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/MANIFEST.in +0 -0
  11. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/setup.cfg +0 -0
  12. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/core.py +0 -0
  13. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/__init__.py +0 -0
  14. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/audio.py +0 -0
  15. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/doc.py +0 -0
  16. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/docx.py +0 -0
  17. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/eml.py +0 -0
  18. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/epub.py +0 -0
  19. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/html.py +0 -0
  20. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/md.py +0 -0
  21. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/msg.py +0 -0
  22. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/ocr.py +0 -0
  23. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/odt.py +0 -0
  24. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/pdf.py +0 -0
  25. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/pptx.py +0 -0
  26. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/rtf.py +0 -0
  27. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/svg.py +0 -0
  28. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/tex.py +0 -0
  29. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/txt.py +0 -0
  30. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/xls.py +0 -0
  31. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/xlsx.py +0 -0
  32. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/xml.py +0 -0
  33. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/examples.py +0 -0
  34. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/pyxtxt.py +0 -0
  35. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
  36. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  37. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/requires.txt +0 -0
  38. {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.1
3
+ Version: 0.3.3
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -264,17 +264,43 @@ text = xtxt("invoice.tiff")
264
264
 
265
265
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
266
266
  # Requires: ollama server running + gemma3:4b model
267
- from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
267
+ from pyxtxt import (
268
+ xtxt, xtxt_image_describe,
269
+ set_ollama_model, set_ollama_config, get_ollama_config
270
+ )
268
271
 
269
272
  # Configure model (optional, default is gemma3:4b)
270
- set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
271
-
272
- # Extract only text (OCR mode)
273
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
274
+
275
+ # Configure LLM parameters for better captions
276
+ set_ollama_config(
277
+ language='italian', # Language hint for captions
278
+ caption_length='long', # short, medium, long
279
+ style='detailed', # descriptive, technical, simple, detailed
280
+ temperature=0.2, # Creativity level (0.0-1.0)
281
+ max_tokens=2000 # Maximum response length
282
+ )
283
+
284
+ # Extract only text (OCR mode)
273
285
  text = xtxt("complex_document.png")
286
+ print(f"Extracted text: {text}")
287
+
288
+ # Extract text + detailed caption
289
+ full_analysis = xtxt_image_describe("scientific_diagram.png")
290
+ print(full_analysis)
291
+ # Output example:
292
+ # TEXT: Figura 2.1: Struttura molecolare del DNA
293
+ # DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
294
+ # con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
295
+ # le basi azotate (adenina, timina, citosina, guanina).
296
+
297
+ # Check current configuration
298
+ config = get_ollama_config()
299
+ print(f"Current config: {config}")
274
300
 
275
- # Extract text + image description
276
- description = xtxt_image_describe(open("diagram.png", "rb"))
277
- # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
301
+ # Reset to defaults if needed
302
+ from pyxtxt import reset_ollama_config
303
+ reset_ollama_config()
278
304
 
279
305
  # From web images
280
306
  import requests
@@ -282,6 +308,32 @@ image_response = requests.get("https://example.com/document.png")
282
308
  text = xtxt(image_response.content)
283
309
  ```
284
310
 
311
+ ### Command-Line OCR Example
312
+
313
+ A complete example script for command-line usage is available:
314
+
315
+ ```python
316
+ # Download and run the example script
317
+ import requests
318
+
319
+ example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
320
+ with open("ocr_example.py", "wb") as f:
321
+ f.write(requests.get(example_url).content)
322
+
323
+ # Usage examples:
324
+ # python ocr_example.py document.png
325
+ # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
326
+ # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
327
+ ```
328
+
329
+ The script supports:
330
+ - **OCR mode**: Extract only text from images
331
+ - **Describe mode**: Extract text + generate detailed captions
332
+ - **Language hints**: Specify caption language (italian, english, etc.)
333
+ - **Style control**: descriptive, technical, simple, detailed
334
+ - **Length control**: short, medium, long captions
335
+ - **Temperature**: Adjust LLM creativity (0.0-1.0)
336
+
285
337
  ### Show Available Formats
286
338
  ```python
287
339
  from pyxtxt import extxt_available_formats
@@ -367,7 +419,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
367
419
 
368
420
  ## 📊 Changelog
369
421
 
370
- ### v0.2.4
422
+ ### v0.2.5 (Current Development)
423
+ - ✅ **NEW**: AI-powered OCR with Ollama LLM integration
424
+ - ✅ **NEW**: Advanced caption generation with configurable parameters
425
+ - ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
426
+ - ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
427
+ - ✅ **NEW**: Caption length control (short/medium/long)
428
+ - ✅ **NEW**: Temperature and token limit configuration
429
+ - ✅ **NEW**: Command-line OCR example script with full parameter support
430
+ - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
431
+ - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
432
+
433
+ ### v0.2.4
371
434
  - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
372
435
  - ✅ **ENHANCED**: Audio transcription now supports video files
373
436
  - ✅ Whisper automatically extracts audio track from videos
@@ -375,15 +438,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
375
438
 
376
439
  ### v0.2.3
377
440
  - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
378
- - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
379
- - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
380
- - ✅ Separate optional dependencies for heavy features (audio/OCR)
381
- - ✅ Performance optimizations with model caching
382
- - ✅ Improved multilingual OCR support (Italian/English)
383
-
384
- ### v0.1.24+
385
- - ✅ Added support for `bytes` objects
386
- - ✅ Added support for `requests.Response` objects
387
- - ✅ Added `xtxt_from_url()` helper function
388
- - ✅ Improved type hints and error handling
389
- - ✅ Enhanced web content processing capabilities
441
+ - ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
442
+ - ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
443
+ - ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
444
+ - ✅ Performance optimizations with model caching for heavy operations
445
+ - ✅ Improved multilingual OCR support with automatic language detection
446
+
447
+ ### v0.2.0-0.2.2
448
+ - ✅ **MAJOR**: Architectural improvements with automatic extractor registration
449
+ - ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
450
+ - ✅ **FIXED**: Critical memory management issues in MSG extractor
451
+ - ✅ **FIXED**: Documentation links and path references
452
+ - ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
453
+ - ✅ Comprehensive testing across all newly supported formats
454
+
455
+ ### v0.1.24
456
+ - ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
457
+ - ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
458
+ - ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
459
+ - ✅ **ENHANCED**: Web-ready architecture for modern applications
460
+ - ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
461
+ - ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
462
+ - ✅ **REMOVED**: Debug print statements from production code
463
+
464
+ ### v0.1.0-0.1.23
465
+ - ✅ **CORE**: Initial release with modular extractor architecture
466
+ - ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
467
+ - ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
468
+ - ✅ **CORE**: MIME type detection with python-magic
469
+ - ✅ **CORE**: BytesIO buffer support for memory-efficient processing
470
+ - ✅ **CORE**: Single dispatch pattern for type-based routing
471
+ - ✅ **CORE**: Automatic dependency management with optional installs
472
+ - ✅ **CORE**: Published to PyPI with proper package structure
@@ -160,17 +160,43 @@ text = xtxt("invoice.tiff")
160
160
 
161
161
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
162
162
  # Requires: ollama server running + gemma3:4b model
163
- from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
163
+ from pyxtxt import (
164
+ xtxt, xtxt_image_describe,
165
+ set_ollama_model, set_ollama_config, get_ollama_config
166
+ )
164
167
 
165
168
  # Configure model (optional, default is gemma3:4b)
166
- set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
167
-
168
- # Extract only text (OCR mode)
169
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
170
+
171
+ # Configure LLM parameters for better captions
172
+ set_ollama_config(
173
+ language='italian', # Language hint for captions
174
+ caption_length='long', # short, medium, long
175
+ style='detailed', # descriptive, technical, simple, detailed
176
+ temperature=0.2, # Creativity level (0.0-1.0)
177
+ max_tokens=2000 # Maximum response length
178
+ )
179
+
180
+ # Extract only text (OCR mode)
169
181
  text = xtxt("complex_document.png")
182
+ print(f"Extracted text: {text}")
183
+
184
+ # Extract text + detailed caption
185
+ full_analysis = xtxt_image_describe("scientific_diagram.png")
186
+ print(full_analysis)
187
+ # Output example:
188
+ # TEXT: Figura 2.1: Struttura molecolare del DNA
189
+ # DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
190
+ # con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
191
+ # le basi azotate (adenina, timina, citosina, guanina).
192
+
193
+ # Check current configuration
194
+ config = get_ollama_config()
195
+ print(f"Current config: {config}")
170
196
 
171
- # Extract text + image description
172
- description = xtxt_image_describe(open("diagram.png", "rb"))
173
- # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
197
+ # Reset to defaults if needed
198
+ from pyxtxt import reset_ollama_config
199
+ reset_ollama_config()
174
200
 
175
201
  # From web images
176
202
  import requests
@@ -178,6 +204,32 @@ image_response = requests.get("https://example.com/document.png")
178
204
  text = xtxt(image_response.content)
179
205
  ```
180
206
 
207
+ ### Command-Line OCR Example
208
+
209
+ A complete example script for command-line usage is available:
210
+
211
+ ```python
212
+ # Download and run the example script
213
+ import requests
214
+
215
+ example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
216
+ with open("ocr_example.py", "wb") as f:
217
+ f.write(requests.get(example_url).content)
218
+
219
+ # Usage examples:
220
+ # python ocr_example.py document.png
221
+ # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
222
+ # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
223
+ ```
224
+
225
+ The script supports:
226
+ - **OCR mode**: Extract only text from images
227
+ - **Describe mode**: Extract text + generate detailed captions
228
+ - **Language hints**: Specify caption language (italian, english, etc.)
229
+ - **Style control**: descriptive, technical, simple, detailed
230
+ - **Length control**: short, medium, long captions
231
+ - **Temperature**: Adjust LLM creativity (0.0-1.0)
232
+
181
233
  ### Show Available Formats
182
234
  ```python
183
235
  from pyxtxt import extxt_available_formats
@@ -263,7 +315,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
263
315
 
264
316
  ## 📊 Changelog
265
317
 
266
- ### v0.2.4
318
+ ### v0.2.5 (Current Development)
319
+ - ✅ **NEW**: AI-powered OCR with Ollama LLM integration
320
+ - ✅ **NEW**: Advanced caption generation with configurable parameters
321
+ - ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
322
+ - ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
323
+ - ✅ **NEW**: Caption length control (short/medium/long)
324
+ - ✅ **NEW**: Temperature and token limit configuration
325
+ - ✅ **NEW**: Command-line OCR example script with full parameter support
326
+ - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
327
+ - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
328
+
329
+ ### v0.2.4
267
330
  - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
268
331
  - ✅ **ENHANCED**: Audio transcription now supports video files
269
332
  - ✅ Whisper automatically extracts audio track from videos
@@ -271,15 +334,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
271
334
 
272
335
  ### v0.2.3
273
336
  - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
274
- - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
275
- - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
276
- - ✅ Separate optional dependencies for heavy features (audio/OCR)
277
- - ✅ Performance optimizations with model caching
278
- - ✅ Improved multilingual OCR support (Italian/English)
279
-
280
- ### v0.1.24+
281
- - ✅ Added support for `bytes` objects
282
- - ✅ Added support for `requests.Response` objects
283
- - ✅ Added `xtxt_from_url()` helper function
284
- - ✅ Improved type hints and error handling
285
- - ✅ Enhanced web content processing capabilities
337
+ - ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
338
+ - ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
339
+ - ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
340
+ - ✅ Performance optimizations with model caching for heavy operations
341
+ - ✅ Improved multilingual OCR support with automatic language detection
342
+
343
+ ### v0.2.0-0.2.2
344
+ - ✅ **MAJOR**: Architectural improvements with automatic extractor registration
345
+ - ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
346
+ - ✅ **FIXED**: Critical memory management issues in MSG extractor
347
+ - ✅ **FIXED**: Documentation links and path references
348
+ - ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
349
+ - ✅ Comprehensive testing across all newly supported formats
350
+
351
+ ### v0.1.24
352
+ - ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
353
+ - ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
354
+ - ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
355
+ - ✅ **ENHANCED**: Web-ready architecture for modern applications
356
+ - ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
357
+ - ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
358
+ - ✅ **REMOVED**: Debug print statements from production code
359
+
360
+ ### v0.1.0-0.1.23
361
+ - ✅ **CORE**: Initial release with modular extractor architecture
362
+ - ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
363
+ - ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
364
+ - ✅ **CORE**: MIME type detection with python-magic
365
+ - ✅ **CORE**: BytesIO buffer support for memory-efficient processing
366
+ - ✅ **CORE**: Single dispatch pattern for type-based routing
367
+ - ✅ **CORE**: Automatic dependency management with optional installs
368
+ - ✅ **CORE**: Published to PyPI with proper package structure
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.3.1"
3
+ version = "0.3.3"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -0,0 +1,15 @@
1
+ from .core import xtxt, extxt_available_formats, xtxt_from_url
2
+
3
+ # Import OCR-Ollama functions if available
4
+ try:
5
+ from .estrattori.ocr_ollama import (
6
+ set_ollama_model, get_ollama_model, xtxt_image_describe,
7
+ set_ollama_config, get_ollama_config, reset_ollama_config
8
+ )
9
+ __all__ = [
10
+ "xtxt", "extxt_available_formats", "xtxt_from_url",
11
+ "set_ollama_model", "get_ollama_model", "xtxt_image_describe",
12
+ "set_ollama_config", "get_ollama_config", "reset_ollama_config"
13
+ ]
14
+ except ImportError:
15
+ __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
@@ -0,0 +1,245 @@
1
+ # pyxtxt/extractors/image_ocr_ollama.py
2
+ from . import register_extractor
3
+ from io import BytesIO
4
+ import base64
5
+
6
+ try:
7
+ import ollama
8
+ from PIL import Image
9
+ except ImportError:
10
+ ollama = None
11
+ Image = None
12
+
13
+ # Global configuration for Ollama model and parameters
14
+ OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
15
+ OLLAMA_CONFIG = {
16
+ 'language': 'auto', # Language hint for caption generation
17
+ 'caption_length': 'medium', # short, medium, long
18
+ 'style': 'descriptive', # descriptive, technical, simple, detailed
19
+ 'temperature': 0.1, # Response creativity (0.0-1.0)
20
+ 'max_tokens': 1500, # Maximum response length
21
+ 'confidence_threshold': 0.7, # Minimum confidence for text extraction
22
+ 'context': 'general' # Context hint: general, cookbook, document, diagram, etc.
23
+ }
24
+
25
+ def set_ollama_model(model_name: str):
26
+ """
27
+ Set the Ollama model to use for OCR.
28
+
29
+ Recommended multimodal models:
30
+ - gemma3:4b (default, balanced speed/quality)
31
+ - gemma3:12b (higher quality, slower)
32
+ - gemma3:27b (best quality, very slow)
33
+ - llava:7b (alternative vision model)
34
+ - llava:13b (higher quality LLAVA)
35
+ """
36
+ global OLLAMA_MODEL
37
+ OLLAMA_MODEL = model_name
38
+ print(f"✅ Ollama OCR model set to: {model_name}")
39
+
40
+ def get_ollama_model():
41
+ """Get current Ollama model name"""
42
+ return OLLAMA_MODEL
43
+
44
+ def set_ollama_config(**kwargs):
45
+ """
46
+ Configure Ollama LLM parameters for better caption generation.
47
+
48
+ Parameters:
49
+ - language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
50
+ - caption_length: Caption length ('short', 'medium', 'long')
51
+ - style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
52
+ - context: Content context ('general', 'cookbook', 'document', 'handwriting', 'technical')
53
+ - temperature: Response creativity 0.0-1.0 (default: 0.1)
54
+ - max_tokens: Maximum response length (default: 1500)
55
+ - confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
56
+
57
+ Examples:
58
+ set_ollama_config(language='italian', style='detailed')
59
+ set_ollama_config(context='document', caption_length='long')
60
+ set_ollama_config(context='handwriting', temperature=0.2)
61
+ """
62
+ global OLLAMA_CONFIG
63
+ for key, value in kwargs.items():
64
+ if key in OLLAMA_CONFIG:
65
+ OLLAMA_CONFIG[key] = value
66
+ print(f"✅ Ollama config updated: {key} = {value}")
67
+ else:
68
+ print(f"⚠️ Unknown config parameter: {key}")
69
+
70
+ def get_ollama_config():
71
+ """Get current Ollama configuration"""
72
+ return OLLAMA_CONFIG.copy()
73
+
74
+ def reset_ollama_config():
75
+ """Reset Ollama configuration to defaults"""
76
+ global OLLAMA_CONFIG
77
+ OLLAMA_CONFIG = {
78
+ 'language': 'auto',
79
+ 'caption_length': 'medium',
80
+ 'style': 'descriptive',
81
+ 'temperature': 0.1,
82
+ 'max_tokens': 1500,
83
+ 'confidence_threshold': 0.7,
84
+ 'context': 'general'
85
+ }
86
+ print("✅ Ollama configuration reset to defaults")
87
+
88
+ if ollama and Image:
89
+ def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
90
+ """
91
+ Extract text from images using Ollama with multimodal models.
92
+
93
+ Args:
94
+ file_buffer: Image file buffer
95
+ mode: "ocr" (text only) or "describe" (text + description)
96
+ model: Override default model (optional)
97
+ """
98
+ try:
99
+ # Use specified model or global default
100
+ current_model = model or OLLAMA_MODEL
101
+
102
+ # Convert buffer to PIL Image
103
+ # Reset buffer position if it has read method
104
+ if hasattr(file_buffer, 'seek'):
105
+ file_buffer.seek(0)
106
+ image_data = file_buffer.read()
107
+ image = Image.open(BytesIO(image_data))
108
+
109
+ # Convert to RGB if needed
110
+ if image.mode != 'RGB':
111
+ image = image.convert('RGB')
112
+
113
+ # Convert image to base64
114
+ buffered = BytesIO()
115
+ image.save(buffered, format="PNG")
116
+ img_base64 = base64.b64encode(buffered.getvalue()).decode()
117
+
118
+ # Get current configuration
119
+ config = OLLAMA_CONFIG
120
+
121
+ # Build language hint
122
+ lang_hint = ""
123
+ if config['language'] != 'auto':
124
+ lang_hint = f"Text language: {config['language']}. "
125
+
126
+ # Different prompts based on mode
127
+ if mode == "ocr":
128
+ prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
129
+ - Titles, headings, and main text content
130
+ - Small print, captions, labels, and annotations
131
+ - Numbers, measurements, quantities, and symbols
132
+ - Menu items, ingredient lists, cooking instructions
133
+ - Any text in boxes, speech bubbles, or decorative elements
134
+
135
+ IMPORTANT:
136
+ - Read carefully and include even small or partially visible text
137
+ - Preserve the original formatting and line breaks where possible
138
+ - {lang_hint}Process the text from left to right, top to bottom
139
+ - If you cannot find any readable text at all, respond with 'NO_TEXT_FOUND'
140
+
141
+ Extracted text:"""
142
+
143
+ else: # mode == "describe"
144
+ # Build style-specific prompts
145
+ style_prompts = {
146
+ 'descriptive': "Provide a clear, descriptive explanation",
147
+ 'technical': "Use technical terminology and precise descriptions",
148
+ 'simple': "Use simple, easy-to-understand language",
149
+ 'detailed': "Provide comprehensive details about all visual elements"
150
+ }
151
+
152
+ length_hints = {
153
+ 'short': "Keep descriptions brief (1-2 sentences)",
154
+ 'medium': "Provide moderate detail (2-4 sentences)",
155
+ 'long': "Give comprehensive descriptions (4-8 sentences)"
156
+ }
157
+
158
+ style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
159
+ length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
160
+
161
+ # Context-specific hints (only when explicitly set)
162
+ context_hint = ""
163
+ context = config.get('context', 'general').lower()
164
+ if context == 'cookbook' or context == 'recipe':
165
+ context_hint = """
166
+ - If this appears to be a recipe/cookbook page, include: ingredients, cooking steps, quantities, cooking times
167
+ - Mention any photos of prepared dishes or cooking techniques shown
168
+ - Note any special formatting like ingredient lists, step numbers, or cooking tips"""
169
+ elif context == 'document':
170
+ context_hint = """
171
+ - Focus on document structure: headers, paragraphs, sections, page numbers
172
+ - Note any official formatting, letterheads, signatures, or stamps"""
173
+ elif context == 'handwriting' or context == 'notes':
174
+ context_hint = """
175
+ - Pay special attention to handwritten text which may be harder to read
176
+ - Note any sketches, diagrams, or informal formatting typical of personal notes"""
177
+ elif context == 'technical' or context == 'diagram':
178
+ context_hint = """
179
+ - Focus on technical elements: labels, measurements, specifications, diagrams
180
+ - Include any mathematical formulas, technical symbols, or engineering notations"""
181
+
182
+ prompt = f"""Analyze this image and provide:
183
+ 1. All visible text exactly as written (preserve formatting, line breaks, bullet points)
184
+ 2. Image description following these guidelines:
185
+ - {style_instruction}
186
+ - {length_instruction}
187
+ - {lang_hint}Focus on key visual elements, layout, and context{context_hint}
188
+
189
+ Format:
190
+ TEXT: [all visible text here, or NO_TEXT_FOUND if none]
191
+ DESCRIPTION: [image description following the guidelines above]"""
192
+
193
+ # Send request to Ollama with configured parameters
194
+ response = ollama.generate(
195
+ model=current_model,
196
+ prompt=prompt,
197
+ images=[img_base64],
198
+ options={
199
+ 'temperature': config['temperature'],
200
+ 'top_p': 0.9,
201
+ 'num_predict': config['max_tokens']
202
+ }
203
+ )
204
+
205
+ # Extract and clean response
206
+ extracted_content = response.get('response', '').strip()
207
+
208
+ # Handle no-text case for OCR mode
209
+ if mode == "ocr" and ('NO_TEXT_FOUND' in extracted_content or len(extracted_content) < 3):
210
+ return ""
211
+
212
+ return extracted_content
213
+
214
+ except Exception as e:
215
+ print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
216
+ return ""
217
+
218
+ # Wrapper functions for each mode
219
+ def xtxt_image_ocr_only(file_input):
220
+ """Traditional OCR: extract only visible text using Ollama"""
221
+ # Handle both file paths and buffers
222
+ if isinstance(file_input, str):
223
+ with open(file_input, 'rb') as f:
224
+ return xtxt_image_ocr_ollama(f, mode="ocr")
225
+ else:
226
+ return xtxt_image_ocr_ollama(file_input, mode="ocr")
227
+
228
+ def xtxt_image_describe(file_input):
229
+ """OCR + Description: text + image context using Ollama"""
230
+ # Handle both file paths and buffers
231
+ if isinstance(file_input, str):
232
+ with open(file_input, 'rb') as f:
233
+ return xtxt_image_ocr_ollama(f, mode="describe")
234
+ else:
235
+ return xtxt_image_ocr_ollama(file_input, mode="describe")
236
+
237
+ # Register OCR-only version as default
238
+ # Note: Will override traditional EasyOCR if both modules are loaded
239
+ image_formats = [
240
+ "image/jpeg", "image/jpg", "image/png",
241
+ "image/bmp", "image/tiff", "image/webp"
242
+ ]
243
+
244
+ for format_type in image_formats:
245
+ register_extractor(format_type, xtxt_image_ocr_only, name="OCR-Ollama")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.1
3
+ Version: 0.3.3
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -264,17 +264,43 @@ text = xtxt("invoice.tiff")
264
264
 
265
265
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
266
266
  # Requires: ollama server running + gemma3:4b model
267
- from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
267
+ from pyxtxt import (
268
+ xtxt, xtxt_image_describe,
269
+ set_ollama_model, set_ollama_config, get_ollama_config
270
+ )
268
271
 
269
272
  # Configure model (optional, default is gemma3:4b)
270
- set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
271
-
272
- # Extract only text (OCR mode)
273
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
274
+
275
+ # Configure LLM parameters for better captions
276
+ set_ollama_config(
277
+ language='italian', # Language hint for captions
278
+ caption_length='long', # short, medium, long
279
+ style='detailed', # descriptive, technical, simple, detailed
280
+ temperature=0.2, # Creativity level (0.0-1.0)
281
+ max_tokens=2000 # Maximum response length
282
+ )
283
+
284
+ # Extract only text (OCR mode)
273
285
  text = xtxt("complex_document.png")
286
+ print(f"Extracted text: {text}")
287
+
288
+ # Extract text + detailed caption
289
+ full_analysis = xtxt_image_describe("scientific_diagram.png")
290
+ print(full_analysis)
291
+ # Output example:
292
+ # TEXT: Figura 2.1: Struttura molecolare del DNA
293
+ # DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
294
+ # con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
295
+ # le basi azotate (adenina, timina, citosina, guanina).
296
+
297
+ # Check current configuration
298
+ config = get_ollama_config()
299
+ print(f"Current config: {config}")
274
300
 
275
- # Extract text + image description
276
- description = xtxt_image_describe(open("diagram.png", "rb"))
277
- # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
301
+ # Reset to defaults if needed
302
+ from pyxtxt import reset_ollama_config
303
+ reset_ollama_config()
278
304
 
279
305
  # From web images
280
306
  import requests
@@ -282,6 +308,32 @@ image_response = requests.get("https://example.com/document.png")
282
308
  text = xtxt(image_response.content)
283
309
  ```
284
310
 
311
+ ### Command-Line OCR Example
312
+
313
+ A complete example script for command-line usage is available:
314
+
315
+ ```python
316
+ # Download and run the example script
317
+ import requests
318
+
319
+ example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
320
+ with open("ocr_example.py", "wb") as f:
321
+ f.write(requests.get(example_url).content)
322
+
323
+ # Usage examples:
324
+ # python ocr_example.py document.png
325
+ # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
326
+ # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
327
+ ```
328
+
329
+ The script supports:
330
+ - **OCR mode**: Extract only text from images
331
+ - **Describe mode**: Extract text + generate detailed captions
332
+ - **Language hints**: Specify caption language (italian, english, etc.)
333
+ - **Style control**: descriptive, technical, simple, detailed
334
+ - **Length control**: short, medium, long captions
335
+ - **Temperature**: Adjust LLM creativity (0.0-1.0)
336
+
285
337
  ### Show Available Formats
286
338
  ```python
287
339
  from pyxtxt import extxt_available_formats
@@ -367,7 +419,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
367
419
 
368
420
  ## 📊 Changelog
369
421
 
370
- ### v0.2.4
422
+ ### v0.2.5 (Current Development)
423
+ - ✅ **NEW**: AI-powered OCR with Ollama LLM integration
424
+ - ✅ **NEW**: Advanced caption generation with configurable parameters
425
+ - ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
426
+ - ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
427
+ - ✅ **NEW**: Caption length control (short/medium/long)
428
+ - ✅ **NEW**: Temperature and token limit configuration
429
+ - ✅ **NEW**: Command-line OCR example script with full parameter support
430
+ - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
431
+ - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
432
+
433
+ ### v0.2.4
371
434
  - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
372
435
  - ✅ **ENHANCED**: Audio transcription now supports video files
373
436
  - ✅ Whisper automatically extracts audio track from videos
@@ -375,15 +438,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
375
438
 
376
439
  ### v0.2.3
377
440
  - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
378
- - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
379
- - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
380
- - ✅ Separate optional dependencies for heavy features (audio/OCR)
381
- - ✅ Performance optimizations with model caching
382
- - ✅ Improved multilingual OCR support (Italian/English)
383
-
384
- ### v0.1.24+
385
- - ✅ Added support for `bytes` objects
386
- - ✅ Added support for `requests.Response` objects
387
- - ✅ Added `xtxt_from_url()` helper function
388
- - ✅ Improved type hints and error handling
389
- - ✅ Enhanced web content processing capabilities
441
+ - ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
442
+ - ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
443
+ - ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
444
+ - ✅ Performance optimizations with model caching for heavy operations
445
+ - ✅ Improved multilingual OCR support with automatic language detection
446
+
447
+ ### v0.2.0-0.2.2
448
+ - ✅ **MAJOR**: Architectural improvements with automatic extractor registration
449
+ - ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
450
+ - ✅ **FIXED**: Critical memory management issues in MSG extractor
451
+ - ✅ **FIXED**: Documentation links and path references
452
+ - ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
453
+ - ✅ Comprehensive testing across all newly supported formats
454
+
455
+ ### v0.1.24
456
+ - ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
457
+ - ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
458
+ - ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
459
+ - ✅ **ENHANCED**: Web-ready architecture for modern applications
460
+ - ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
461
+ - ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
462
+ - ✅ **REMOVED**: Debug print statements from production code
463
+
464
+ ### v0.1.0-0.1.23
465
+ - ✅ **CORE**: Initial release with modular extractor architecture
466
+ - ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
467
+ - ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
468
+ - ✅ **CORE**: MIME type detection with python-magic
469
+ - ✅ **CORE**: BytesIO buffer support for memory-efficient processing
470
+ - ✅ **CORE**: Single dispatch pattern for type-based routing
471
+ - ✅ **CORE**: Automatic dependency management with optional installs
472
+ - ✅ **CORE**: Published to PyPI with proper package structure
@@ -1,8 +0,0 @@
1
- from .core import xtxt, extxt_available_formats, xtxt_from_url
2
-
3
- # Import OCR-Ollama functions if available
4
- try:
5
- from .estrattori.ocr_ollama import set_ollama_model, get_ollama_model, xtxt_image_describe
6
- __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url", "set_ollama_model", "get_ollama_model", "xtxt_image_describe"]
7
- except ImportError:
8
- __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
@@ -1,125 +0,0 @@
1
- # pyxtxt/extractors/image_ocr_ollama.py
2
- from . import register_extractor
3
- from io import BytesIO
4
- import base64
5
-
6
- try:
7
- import ollama
8
- from PIL import Image
9
- except ImportError:
10
- ollama = None
11
- Image = None
12
-
13
- # Global configuration for Ollama model
14
- OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
15
-
16
- def set_ollama_model(model_name: str):
17
- """
18
- Set the Ollama model to use for OCR.
19
-
20
- Recommended multimodal models:
21
- - gemma3:4b (default, balanced speed/quality)
22
- - gemma3:12b (higher quality, slower)
23
- - gemma3:27b (best quality, very slow)
24
- - llava:7b (alternative vision model)
25
- - llava:13b (higher quality LLAVA)
26
- """
27
- global OLLAMA_MODEL
28
- OLLAMA_MODEL = model_name
29
- print(f"✅ Ollama OCR model set to: {model_name}")
30
-
31
- def get_ollama_model():
32
- """Get current Ollama model name"""
33
- return OLLAMA_MODEL
34
-
35
- if ollama and Image:
36
- def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
37
- """
38
- Extract text from images using Ollama with multimodal models.
39
-
40
- Args:
41
- file_buffer: Image file buffer
42
- mode: "ocr" (text only) or "describe" (text + description)
43
- model: Override default model (optional)
44
- """
45
- try:
46
- # Use specified model or global default
47
- current_model = model or OLLAMA_MODEL
48
-
49
- # Convert buffer to PIL Image
50
- image = Image.open(BytesIO(file_buffer.read()))
51
-
52
- # Convert to RGB if needed
53
- if image.mode != 'RGB':
54
- image = image.convert('RGB')
55
-
56
- # Convert image to base64
57
- buffered = BytesIO()
58
- image.save(buffered, format="PNG")
59
- img_base64 = base64.b64encode(buffered.getvalue()).decode()
60
-
61
- # Different prompts based on mode
62
- if mode == "ocr":
63
- prompt = """Extract ALL visible text from this image exactly as it appears.
64
- Rules:
65
- - Only return text that is actually written/printed in the image
66
- - Preserve reading order (left to right, top to bottom)
67
- - Maintain line breaks and formatting
68
- - Include numbers, symbols, special characters
69
- - Do NOT add descriptions, interpretations, or context
70
- - If no text is visible, return 'NO_TEXT_FOUND'
71
-
72
- Extracted text:"""
73
-
74
- else: # mode == "describe"
75
- prompt = """Analyze this image and provide:
76
- 1. All visible text exactly as written
77
- 2. Brief description of the image content and context
78
-
79
- Format:
80
- TEXT: [all visible text here, or NO_TEXT_FOUND if none]
81
- DESCRIPTION: [brief image description and context]"""
82
-
83
- # Send request to Ollama
84
- response = ollama.generate(
85
- model=current_model,
86
- prompt=prompt,
87
- images=[img_base64],
88
- options={
89
- 'temperature': 0.1, # Low temperature for accuracy
90
- 'top_p': 0.9,
91
- 'num_predict': 1500
92
- }
93
- )
94
-
95
- # Extract and clean response
96
- extracted_content = response.get('response', '').strip()
97
-
98
- # Handle no-text case for OCR mode
99
- if mode == "ocr" and ('NO_TEXT_FOUND' in extracted_content or len(extracted_content) < 3):
100
- return ""
101
-
102
- return extracted_content
103
-
104
- except Exception as e:
105
- print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
106
- return ""
107
-
108
- # Wrapper functions for each mode
109
- def xtxt_image_ocr_only(file_buffer):
110
- """Traditional OCR: extract only visible text using Ollama"""
111
- return xtxt_image_ocr_ollama(file_buffer, mode="ocr")
112
-
113
- def xtxt_image_describe(file_buffer):
114
- """OCR + Description: text + image context using Ollama"""
115
- return xtxt_image_ocr_ollama(file_buffer, mode="describe")
116
-
117
- # Register OCR-only version as default
118
- # Note: Will override traditional EasyOCR if both modules are loaded
119
- image_formats = [
120
- "image/jpeg", "image/jpg", "image/png",
121
- "image/bmp", "image/tiff", "image/webp"
122
- ]
123
-
124
- for format_type in image_formats:
125
- register_extractor(format_type, xtxt_image_ocr_only, name="OCR-Ollama")
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes