pyxtxt 0.3.1__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {pyxtxt-0.3.1/src/pyxtxt.egg-info → pyxtxt-0.3.2}/PKG-INFO +104 -21
  2. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/README.md +103 -20
  3. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/pyproject.toml +1 -1
  4. pyxtxt-0.3.2/src/pyxtxt/__init__.py +15 -0
  5. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/ocr_ollama.py +86 -9
  6. {pyxtxt-0.3.1 → pyxtxt-0.3.2/src/pyxtxt.egg-info}/PKG-INFO +104 -21
  7. pyxtxt-0.3.1/src/pyxtxt/__init__.py +0 -8
  8. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/LICENSE +0 -0
  9. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/MANIFEST.in +0 -0
  10. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/setup.cfg +0 -0
  11. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/core.py +0 -0
  12. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/__init__.py +0 -0
  13. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/audio.py +0 -0
  14. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/doc.py +0 -0
  15. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/docx.py +0 -0
  16. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/eml.py +0 -0
  17. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/epub.py +0 -0
  18. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/html.py +0 -0
  19. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/md.py +0 -0
  20. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/msg.py +0 -0
  21. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/ocr.py +0 -0
  22. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/odt.py +0 -0
  23. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/pdf.py +0 -0
  24. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/pptx.py +0 -0
  25. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/rtf.py +0 -0
  26. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/svg.py +0 -0
  27. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/tex.py +0 -0
  28. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/txt.py +0 -0
  29. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/xls.py +0 -0
  30. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/xlsx.py +0 -0
  31. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/xml.py +0 -0
  32. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/examples.py +0 -0
  33. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt/pyxtxt.py +0 -0
  34. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
  35. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  36. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt.egg-info/requires.txt +0 -0
  37. {pyxtxt-0.3.1 → pyxtxt-0.3.2}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.1
3
+ Version: 0.3.2
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -264,17 +264,43 @@ text = xtxt("invoice.tiff")
264
264
 
265
265
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
266
266
  # Requires: ollama server running + gemma3:4b model
267
- from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
267
+ from pyxtxt import (
268
+ xtxt, xtxt_image_describe,
269
+ set_ollama_model, set_ollama_config, get_ollama_config
270
+ )
268
271
 
269
272
  # Configure model (optional, default is gemma3:4b)
270
- set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
271
-
272
- # Extract only text (OCR mode)
273
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
274
+
275
+ # Configure LLM parameters for better captions
276
+ set_ollama_config(
277
+ language='italian', # Language hint for captions
278
+ caption_length='long', # short, medium, long
279
+ style='detailed', # descriptive, technical, simple, detailed
280
+ temperature=0.2, # Creativity level (0.0-1.0)
281
+ max_tokens=2000 # Maximum response length
282
+ )
283
+
284
+ # Extract only text (OCR mode)
273
285
  text = xtxt("complex_document.png")
286
+ print(f"Extracted text: {text}")
287
+
288
+ # Extract text + detailed caption
289
+ full_analysis = xtxt_image_describe("scientific_diagram.png")
290
+ print(full_analysis)
291
+ # Output example:
292
+ # TEXT: Figura 2.1: Struttura molecolare del DNA
293
+ # DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
294
+ # con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
295
+ # le basi azotate (adenina, timina, citosina, guanina).
296
+
297
+ # Check current configuration
298
+ config = get_ollama_config()
299
+ print(f"Current config: {config}")
274
300
 
275
- # Extract text + image description
276
- description = xtxt_image_describe(open("diagram.png", "rb"))
277
- # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
301
+ # Reset to defaults if needed
302
+ from pyxtxt import reset_ollama_config
303
+ reset_ollama_config()
278
304
 
279
305
  # From web images
280
306
  import requests
@@ -282,6 +308,32 @@ image_response = requests.get("https://example.com/document.png")
282
308
  text = xtxt(image_response.content)
283
309
  ```
284
310
 
311
+ ### Command-Line OCR Example
312
+
313
+ A complete example script for command-line usage is available:
314
+
315
+ ```python
316
+ # Download and run the example script
317
+ import requests
318
+
319
+ example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
320
+ with open("ocr_example.py", "wb") as f:
321
+ f.write(requests.get(example_url).content)
322
+
323
+ # Usage examples:
324
+ # python ocr_example.py document.png
325
+ # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
326
+ # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
327
+ ```
328
+
329
+ The script supports:
330
+ - **OCR mode**: Extract only text from images
331
+ - **Describe mode**: Extract text + generate detailed captions
332
+ - **Language hints**: Specify caption language (italian, english, etc.)
333
+ - **Style control**: descriptive, technical, simple, detailed
334
+ - **Length control**: short, medium, long captions
335
+ - **Temperature**: Adjust LLM creativity (0.0-1.0)
336
+
285
337
  ### Show Available Formats
286
338
  ```python
287
339
  from pyxtxt import extxt_available_formats
@@ -367,7 +419,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
367
419
 
368
420
  ## 📊 Changelog
369
421
 
370
- ### v0.2.4
422
+ ### v0.2.5 (Current Development)
423
+ - ✅ **NEW**: AI-powered OCR with Ollama LLM integration
424
+ - ✅ **NEW**: Advanced caption generation with configurable parameters
425
+ - ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
426
+ - ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
427
+ - ✅ **NEW**: Caption length control (short/medium/long)
428
+ - ✅ **NEW**: Temperature and token limit configuration
429
+ - ✅ **NEW**: Command-line OCR example script with full parameter support
430
+ - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
431
+ - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
432
+
433
+ ### v0.2.4
371
434
  - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
372
435
  - ✅ **ENHANCED**: Audio transcription now supports video files
373
436
  - ✅ Whisper automatically extracts audio track from videos
@@ -375,15 +438,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
375
438
 
376
439
  ### v0.2.3
377
440
  - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
378
- - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
379
- - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
380
- - ✅ Separate optional dependencies for heavy features (audio/OCR)
381
- - ✅ Performance optimizations with model caching
382
- - ✅ Improved multilingual OCR support (Italian/English)
383
-
384
- ### v0.1.24+
385
- - ✅ Added support for `bytes` objects
386
- - ✅ Added support for `requests.Response` objects
387
- - ✅ Added `xtxt_from_url()` helper function
388
- - ✅ Improved type hints and error handling
389
- - ✅ Enhanced web content processing capabilities
441
+ - ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
442
+ - ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
443
+ - ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
444
+ - ✅ Performance optimizations with model caching for heavy operations
445
+ - ✅ Improved multilingual OCR support with automatic language detection
446
+
447
+ ### v0.2.0-0.2.2
448
+ - ✅ **MAJOR**: Architectural improvements with automatic extractor registration
449
+ - ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
450
+ - ✅ **FIXED**: Critical memory management issues in MSG extractor
451
+ - ✅ **FIXED**: Documentation links and path references
452
+ - ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
453
+ - ✅ Comprehensive testing across all newly supported formats
454
+
455
+ ### v0.1.24
456
+ - ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
457
+ - ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
458
+ - ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
459
+ - ✅ **ENHANCED**: Web-ready architecture for modern applications
460
+ - ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
461
+ - ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
462
+ - ✅ **REMOVED**: Debug print statements from production code
463
+
464
+ ### v0.1.0-0.1.23
465
+ - ✅ **CORE**: Initial release with modular extractor architecture
466
+ - ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
467
+ - ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
468
+ - ✅ **CORE**: MIME type detection with python-magic
469
+ - ✅ **CORE**: BytesIO buffer support for memory-efficient processing
470
+ - ✅ **CORE**: Single dispatch pattern for type-based routing
471
+ - ✅ **CORE**: Automatic dependency management with optional installs
472
+ - ✅ **CORE**: Published to PyPI with proper package structure
@@ -160,17 +160,43 @@ text = xtxt("invoice.tiff")
160
160
 
161
161
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
162
162
  # Requires: ollama server running + gemma3:4b model
163
- from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
163
+ from pyxtxt import (
164
+ xtxt, xtxt_image_describe,
165
+ set_ollama_model, set_ollama_config, get_ollama_config
166
+ )
164
167
 
165
168
  # Configure model (optional, default is gemma3:4b)
166
- set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
167
-
168
- # Extract only text (OCR mode)
169
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
170
+
171
+ # Configure LLM parameters for better captions
172
+ set_ollama_config(
173
+ language='italian', # Language hint for captions
174
+ caption_length='long', # short, medium, long
175
+ style='detailed', # descriptive, technical, simple, detailed
176
+ temperature=0.2, # Creativity level (0.0-1.0)
177
+ max_tokens=2000 # Maximum response length
178
+ )
179
+
180
+ # Extract only text (OCR mode)
169
181
  text = xtxt("complex_document.png")
182
+ print(f"Extracted text: {text}")
183
+
184
+ # Extract text + detailed caption
185
+ full_analysis = xtxt_image_describe("scientific_diagram.png")
186
+ print(full_analysis)
187
+ # Output example:
188
+ # TEXT: Figura 2.1: Struttura molecolare del DNA
189
+ # DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
190
+ # con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
191
+ # le basi azotate (adenina, timina, citosina, guanina).
192
+
193
+ # Check current configuration
194
+ config = get_ollama_config()
195
+ print(f"Current config: {config}")
170
196
 
171
- # Extract text + image description
172
- description = xtxt_image_describe(open("diagram.png", "rb"))
173
- # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
197
+ # Reset to defaults if needed
198
+ from pyxtxt import reset_ollama_config
199
+ reset_ollama_config()
174
200
 
175
201
  # From web images
176
202
  import requests
@@ -178,6 +204,32 @@ image_response = requests.get("https://example.com/document.png")
178
204
  text = xtxt(image_response.content)
179
205
  ```
180
206
 
207
+ ### Command-Line OCR Example
208
+
209
+ A complete example script for command-line usage is available:
210
+
211
+ ```python
212
+ # Download and run the example script
213
+ import requests
214
+
215
+ example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
216
+ with open("ocr_example.py", "wb") as f:
217
+ f.write(requests.get(example_url).content)
218
+
219
+ # Usage examples:
220
+ # python ocr_example.py document.png
221
+ # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
222
+ # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
223
+ ```
224
+
225
+ The script supports:
226
+ - **OCR mode**: Extract only text from images
227
+ - **Describe mode**: Extract text + generate detailed captions
228
+ - **Language hints**: Specify caption language (italian, english, etc.)
229
+ - **Style control**: descriptive, technical, simple, detailed
230
+ - **Length control**: short, medium, long captions
231
+ - **Temperature**: Adjust LLM creativity (0.0-1.0)
232
+
181
233
  ### Show Available Formats
182
234
  ```python
183
235
  from pyxtxt import extxt_available_formats
@@ -263,7 +315,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
263
315
 
264
316
  ## 📊 Changelog
265
317
 
266
- ### v0.2.4
318
+ ### v0.2.5 (Current Development)
319
+ - ✅ **NEW**: AI-powered OCR with Ollama LLM integration
320
+ - ✅ **NEW**: Advanced caption generation with configurable parameters
321
+ - ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
322
+ - ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
323
+ - ✅ **NEW**: Caption length control (short/medium/long)
324
+ - ✅ **NEW**: Temperature and token limit configuration
325
+ - ✅ **NEW**: Command-line OCR example script with full parameter support
326
+ - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
327
+ - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
328
+
329
+ ### v0.2.4
267
330
  - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
268
331
  - ✅ **ENHANCED**: Audio transcription now supports video files
269
332
  - ✅ Whisper automatically extracts audio track from videos
@@ -271,15 +334,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
271
334
 
272
335
  ### v0.2.3
273
336
  - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
274
- - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
275
- - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
276
- - ✅ Separate optional dependencies for heavy features (audio/OCR)
277
- - ✅ Performance optimizations with model caching
278
- - ✅ Improved multilingual OCR support (Italian/English)
279
-
280
- ### v0.1.24+
281
- - ✅ Added support for `bytes` objects
282
- - ✅ Added support for `requests.Response` objects
283
- - ✅ Added `xtxt_from_url()` helper function
284
- - ✅ Improved type hints and error handling
285
- - ✅ Enhanced web content processing capabilities
337
+ - ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
338
+ - ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
339
+ - ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
340
+ - ✅ Performance optimizations with model caching for heavy operations
341
+ - ✅ Improved multilingual OCR support with automatic language detection
342
+
343
+ ### v0.2.0-0.2.2
344
+ - ✅ **MAJOR**: Architectural improvements with automatic extractor registration
345
+ - ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
346
+ - ✅ **FIXED**: Critical memory management issues in MSG extractor
347
+ - ✅ **FIXED**: Documentation links and path references
348
+ - ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
349
+ - ✅ Comprehensive testing across all newly supported formats
350
+
351
+ ### v0.1.24
352
+ - ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
353
+ - ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
354
+ - ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
355
+ - ✅ **ENHANCED**: Web-ready architecture for modern applications
356
+ - ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
357
+ - ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
358
+ - ✅ **REMOVED**: Debug print statements from production code
359
+
360
+ ### v0.1.0-0.1.23
361
+ - ✅ **CORE**: Initial release with modular extractor architecture
362
+ - ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
363
+ - ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
364
+ - ✅ **CORE**: MIME type detection with python-magic
365
+ - ✅ **CORE**: BytesIO buffer support for memory-efficient processing
366
+ - ✅ **CORE**: Single dispatch pattern for type-based routing
367
+ - ✅ **CORE**: Automatic dependency management with optional installs
368
+ - ✅ **CORE**: Published to PyPI with proper package structure
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.3.1"
3
+ version = "0.3.2"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -0,0 +1,15 @@
1
+ from .core import xtxt, extxt_available_formats, xtxt_from_url
2
+
3
+ # Import OCR-Ollama functions if available
4
+ try:
5
+ from .estrattori.ocr_ollama import (
6
+ set_ollama_model, get_ollama_model, xtxt_image_describe,
7
+ set_ollama_config, get_ollama_config, reset_ollama_config
8
+ )
9
+ __all__ = [
10
+ "xtxt", "extxt_available_formats", "xtxt_from_url",
11
+ "set_ollama_model", "get_ollama_model", "xtxt_image_describe",
12
+ "set_ollama_config", "get_ollama_config", "reset_ollama_config"
13
+ ]
14
+ except ImportError:
15
+ __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
@@ -10,8 +10,16 @@ except ImportError:
10
10
  ollama = None
11
11
  Image = None
12
12
 
13
- # Global configuration for Ollama model
13
+ # Global configuration for Ollama model and parameters
14
14
  OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
15
+ OLLAMA_CONFIG = {
16
+ 'language': 'auto', # Language hint for caption generation
17
+ 'caption_length': 'medium', # short, medium, long
18
+ 'style': 'descriptive', # descriptive, technical, simple, detailed
19
+ 'temperature': 0.1, # Response creativity (0.0-1.0)
20
+ 'max_tokens': 1500, # Maximum response length
21
+ 'confidence_threshold': 0.7 # Minimum confidence for text extraction
22
+ }
15
23
 
16
24
  def set_ollama_model(model_name: str):
17
25
  """
@@ -32,6 +40,47 @@ def get_ollama_model():
32
40
  """Get current Ollama model name"""
33
41
  return OLLAMA_MODEL
34
42
 
43
+ def set_ollama_config(**kwargs):
44
+ """
45
+ Configure Ollama LLM parameters for better caption generation.
46
+
47
+ Parameters:
48
+ - language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
49
+ - caption_length: Caption length ('short', 'medium', 'long')
50
+ - style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
51
+ - temperature: Response creativity 0.0-1.0 (default: 0.1)
52
+ - max_tokens: Maximum response length (default: 1500)
53
+ - confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
54
+
55
+ Examples:
56
+ set_ollama_config(language='italian', style='detailed')
57
+ set_ollama_config(caption_length='long', temperature=0.3)
58
+ """
59
+ global OLLAMA_CONFIG
60
+ for key, value in kwargs.items():
61
+ if key in OLLAMA_CONFIG:
62
+ OLLAMA_CONFIG[key] = value
63
+ print(f"✅ Ollama config updated: {key} = {value}")
64
+ else:
65
+ print(f"⚠️ Unknown config parameter: {key}")
66
+
67
+ def get_ollama_config():
68
+ """Get current Ollama configuration"""
69
+ return OLLAMA_CONFIG.copy()
70
+
71
+ def reset_ollama_config():
72
+ """Reset Ollama configuration to defaults"""
73
+ global OLLAMA_CONFIG
74
+ OLLAMA_CONFIG = {
75
+ 'language': 'auto',
76
+ 'caption_length': 'medium',
77
+ 'style': 'descriptive',
78
+ 'temperature': 0.1,
79
+ 'max_tokens': 1500,
80
+ 'confidence_threshold': 0.7
81
+ }
82
+ print("✅ Ollama configuration reset to defaults")
83
+
35
84
  if ollama and Image:
36
85
  def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
37
86
  """
@@ -58,37 +107,65 @@ if ollama and Image:
58
107
  image.save(buffered, format="PNG")
59
108
  img_base64 = base64.b64encode(buffered.getvalue()).decode()
60
109
 
110
+ # Get current configuration
111
+ config = OLLAMA_CONFIG
112
+
113
+ # Build language hint
114
+ lang_hint = ""
115
+ if config['language'] != 'auto':
116
+ lang_hint = f"Text language: {config['language']}. "
117
+
61
118
  # Different prompts based on mode
62
119
  if mode == "ocr":
63
- prompt = """Extract ALL visible text from this image exactly as it appears.
120
+ prompt = f"""Extract ALL visible text from this image exactly as it appears.
64
121
  Rules:
65
122
  - Only return text that is actually written/printed in the image
66
123
  - Preserve reading order (left to right, top to bottom)
67
124
  - Maintain line breaks and formatting
68
125
  - Include numbers, symbols, special characters
69
126
  - Do NOT add descriptions, interpretations, or context
70
- - If no text is visible, return 'NO_TEXT_FOUND'
127
+ - {lang_hint}If no text is visible, return 'NO_TEXT_FOUND'
71
128
 
72
129
  Extracted text:"""
73
130
 
74
131
  else: # mode == "describe"
75
- prompt = """Analyze this image and provide:
132
+ # Build style-specific prompts
133
+ style_prompts = {
134
+ 'descriptive': "Provide a clear, descriptive explanation",
135
+ 'technical': "Use technical terminology and precise descriptions",
136
+ 'simple': "Use simple, easy-to-understand language",
137
+ 'detailed': "Provide comprehensive details about all visual elements"
138
+ }
139
+
140
+ length_hints = {
141
+ 'short': "Keep descriptions brief (1-2 sentences)",
142
+ 'medium': "Provide moderate detail (2-4 sentences)",
143
+ 'long': "Give comprehensive descriptions (4-8 sentences)"
144
+ }
145
+
146
+ style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
147
+ length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
148
+
149
+ prompt = f"""Analyze this image and provide:
76
150
  1. All visible text exactly as written
77
- 2. Brief description of the image content and context
151
+ 2. Image description following these guidelines:
152
+ - {style_instruction}
153
+ - {length_instruction}
154
+ - {lang_hint}Focus on key visual elements, layout, and context
78
155
 
79
156
  Format:
80
157
  TEXT: [all visible text here, or NO_TEXT_FOUND if none]
81
- DESCRIPTION: [brief image description and context]"""
158
+ DESCRIPTION: [image description following the guidelines above]"""
82
159
 
83
- # Send request to Ollama
160
+ # Send request to Ollama with configured parameters
84
161
  response = ollama.generate(
85
162
  model=current_model,
86
163
  prompt=prompt,
87
164
  images=[img_base64],
88
165
  options={
89
- 'temperature': 0.1, # Low temperature for accuracy
166
+ 'temperature': config['temperature'],
90
167
  'top_p': 0.9,
91
- 'num_predict': 1500
168
+ 'num_predict': config['max_tokens']
92
169
  }
93
170
  )
94
171
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.1
3
+ Version: 0.3.2
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -264,17 +264,43 @@ text = xtxt("invoice.tiff")
264
264
 
265
265
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
266
266
  # Requires: ollama server running + gemma3:4b model
267
- from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
267
+ from pyxtxt import (
268
+ xtxt, xtxt_image_describe,
269
+ set_ollama_model, set_ollama_config, get_ollama_config
270
+ )
268
271
 
269
272
  # Configure model (optional, default is gemma3:4b)
270
- set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
271
-
272
- # Extract only text (OCR mode)
273
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
274
+
275
+ # Configure LLM parameters for better captions
276
+ set_ollama_config(
277
+ language='italian', # Language hint for captions
278
+ caption_length='long', # short, medium, long
279
+ style='detailed', # descriptive, technical, simple, detailed
280
+ temperature=0.2, # Creativity level (0.0-1.0)
281
+ max_tokens=2000 # Maximum response length
282
+ )
283
+
284
+ # Extract only text (OCR mode)
273
285
  text = xtxt("complex_document.png")
286
+ print(f"Extracted text: {text}")
287
+
288
+ # Extract text + detailed caption
289
+ full_analysis = xtxt_image_describe("scientific_diagram.png")
290
+ print(full_analysis)
291
+ # Output example:
292
+ # TEXT: Figura 2.1: Struttura molecolare del DNA
293
+ # DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
294
+ # con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
295
+ # le basi azotate (adenina, timina, citosina, guanina).
296
+
297
+ # Check current configuration
298
+ config = get_ollama_config()
299
+ print(f"Current config: {config}")
274
300
 
275
- # Extract text + image description
276
- description = xtxt_image_describe(open("diagram.png", "rb"))
277
- # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
301
+ # Reset to defaults if needed
302
+ from pyxtxt import reset_ollama_config
303
+ reset_ollama_config()
278
304
 
279
305
  # From web images
280
306
  import requests
@@ -282,6 +308,32 @@ image_response = requests.get("https://example.com/document.png")
282
308
  text = xtxt(image_response.content)
283
309
  ```
284
310
 
311
+ ### Command-Line OCR Example
312
+
313
+ A complete example script for command-line usage is available:
314
+
315
+ ```python
316
+ # Download and run the example script
317
+ import requests
318
+
319
+ example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
320
+ with open("ocr_example.py", "wb") as f:
321
+ f.write(requests.get(example_url).content)
322
+
323
+ # Usage examples:
324
+ # python ocr_example.py document.png
325
+ # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
326
+ # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
327
+ ```
328
+
329
+ The script supports:
330
+ - **OCR mode**: Extract only text from images
331
+ - **Describe mode**: Extract text + generate detailed captions
332
+ - **Language hints**: Specify caption language (italian, english, etc.)
333
+ - **Style control**: descriptive, technical, simple, detailed
334
+ - **Length control**: short, medium, long captions
335
+ - **Temperature**: Adjust LLM creativity (0.0-1.0)
336
+
285
337
  ### Show Available Formats
286
338
  ```python
287
339
  from pyxtxt import extxt_available_formats
@@ -367,7 +419,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
367
419
 
368
420
  ## 📊 Changelog
369
421
 
370
- ### v0.2.4
422
+ ### v0.2.5 (Current Development)
423
+ - ✅ **NEW**: AI-powered OCR with Ollama LLM integration
424
+ - ✅ **NEW**: Advanced caption generation with configurable parameters
425
+ - ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
426
+ - ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
427
+ - ✅ **NEW**: Caption length control (short/medium/long)
428
+ - ✅ **NEW**: Temperature and token limit configuration
429
+ - ✅ **NEW**: Command-line OCR example script with full parameter support
430
+ - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
431
+ - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
432
+
433
+ ### v0.2.4
371
434
  - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
372
435
  - ✅ **ENHANCED**: Audio transcription now supports video files
373
436
  - ✅ Whisper automatically extracts audio track from videos
@@ -375,15 +438,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
375
438
 
376
439
  ### v0.2.3
377
440
  - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
378
- - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
379
- - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
380
- - ✅ Separate optional dependencies for heavy features (audio/OCR)
381
- - ✅ Performance optimizations with model caching
382
- - ✅ Improved multilingual OCR support (Italian/English)
383
-
384
- ### v0.1.24+
385
- - ✅ Added support for `bytes` objects
386
- - ✅ Added support for `requests.Response` objects
387
- - ✅ Added `xtxt_from_url()` helper function
388
- - ✅ Improved type hints and error handling
389
- - ✅ Enhanced web content processing capabilities
441
+ - ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
442
+ - ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
443
+ - ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
444
+ - ✅ Performance optimizations with model caching for heavy operations
445
+ - ✅ Improved multilingual OCR support with automatic language detection
446
+
447
+ ### v0.2.0-0.2.2
448
+ - ✅ **MAJOR**: Architectural improvements with automatic extractor registration
449
+ - ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
450
+ - ✅ **FIXED**: Critical memory management issues in MSG extractor
451
+ - ✅ **FIXED**: Documentation links and path references
452
+ - ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
453
+ - ✅ Comprehensive testing across all newly supported formats
454
+
455
+ ### v0.1.24
456
+ - ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
457
+ - ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
458
+ - ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
459
+ - ✅ **ENHANCED**: Web-ready architecture for modern applications
460
+ - ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
461
+ - ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
462
+ - ✅ **REMOVED**: Debug print statements from production code
463
+
464
+ ### v0.1.0-0.1.23
465
+ - ✅ **CORE**: Initial release with modular extractor architecture
466
+ - ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
467
+ - ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
468
+ - ✅ **CORE**: MIME type detection with python-magic
469
+ - ✅ **CORE**: BytesIO buffer support for memory-efficient processing
470
+ - ✅ **CORE**: Single dispatch pattern for type-based routing
471
+ - ✅ **CORE**: Automatic dependency management with optional installs
472
+ - ✅ **CORE**: Published to PyPI with proper package structure
@@ -1,8 +0,0 @@
1
- from .core import xtxt, extxt_available_formats, xtxt_from_url
2
-
3
- # Import OCR-Ollama functions if available
4
- try:
5
- from .estrattori.ocr_ollama import set_ollama_model, get_ollama_model, xtxt_image_describe
6
- __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url", "set_ollama_model", "get_ollama_model", "xtxt_image_describe"]
7
- except ImportError:
8
- __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes