pyxtxt 0.3__tar.gz → 0.3.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {pyxtxt-0.3/src/pyxtxt.egg-info → pyxtxt-0.3.2}/PKG-INFO +109 -30
  2. {pyxtxt-0.3 → pyxtxt-0.3.2}/README.md +108 -29
  3. {pyxtxt-0.3 → pyxtxt-0.3.2}/pyproject.toml +1 -1
  4. pyxtxt-0.3.2/src/pyxtxt/__init__.py +15 -0
  5. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/ocr_ollama.py +86 -9
  6. {pyxtxt-0.3 → pyxtxt-0.3.2/src/pyxtxt.egg-info}/PKG-INFO +109 -30
  7. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt.egg-info/SOURCES.txt +1 -1
  8. pyxtxt-0.3/src/pyxtxt/__init__.py +0 -8
  9. {pyxtxt-0.3 → pyxtxt-0.3.2}/LICENSE +0 -0
  10. {pyxtxt-0.3 → pyxtxt-0.3.2}/MANIFEST.in +0 -0
  11. {pyxtxt-0.3 → pyxtxt-0.3.2}/setup.cfg +0 -0
  12. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/core.py +0 -0
  13. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/__init__.py +0 -0
  14. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/audio.py +0 -0
  15. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/doc.py +0 -0
  16. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/docx.py +0 -0
  17. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/eml.py +0 -0
  18. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/epub.py +0 -0
  19. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/html.py +0 -0
  20. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/md.py +0 -0
  21. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/msg.py +0 -0
  22. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/ocr.py +0 -0
  23. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/odt.py +0 -0
  24. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/pdf.py +0 -0
  25. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/pptx.py +0 -0
  26. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/rtf.py +0 -0
  27. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/svg.py +0 -0
  28. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/tex.py +0 -0
  29. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/txt.py +0 -0
  30. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/xls.py +0 -0
  31. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/xlsx.py +0 -0
  32. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/estrattori/xml.py +0 -0
  33. {pyxtxt-0.3 → pyxtxt-0.3.2/src/pyxtxt}/examples.py +0 -0
  34. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt/pyxtxt.py +0 -0
  35. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  36. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt.egg-info/requires.txt +0 -0
  37. {pyxtxt-0.3 → pyxtxt-0.3.2}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3
3
+ Version: 0.3.2
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -264,17 +264,43 @@ text = xtxt("invoice.tiff")
264
264
 
265
265
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
266
266
  # Requires: ollama server running + gemma3:4b model
267
- from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
267
+ from pyxtxt import (
268
+ xtxt, xtxt_image_describe,
269
+ set_ollama_model, set_ollama_config, get_ollama_config
270
+ )
268
271
 
269
272
  # Configure model (optional, default is gemma3:4b)
270
- set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
271
-
272
- # Extract only text (OCR mode)
273
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
274
+
275
+ # Configure LLM parameters for better captions
276
+ set_ollama_config(
277
+ language='italian', # Language hint for captions
278
+ caption_length='long', # short, medium, long
279
+ style='detailed', # descriptive, technical, simple, detailed
280
+ temperature=0.2, # Creativity level (0.0-1.0)
281
+ max_tokens=2000 # Maximum response length
282
+ )
283
+
284
+ # Extract only text (OCR mode)
273
285
  text = xtxt("complex_document.png")
286
+ print(f"Extracted text: {text}")
287
+
288
+ # Extract text + detailed caption
289
+ full_analysis = xtxt_image_describe("scientific_diagram.png")
290
+ print(full_analysis)
291
+ # Output example:
292
+ # TEXT: Figura 2.1: Struttura molecolare del DNA
293
+ # DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
294
+ # con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
295
+ # le basi azotate (adenina, timina, citosina, guanina).
274
296
 
275
- # Extract text + image description
276
- description = xtxt_image_describe(open("diagram.png", "rb"))
277
- # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
297
+ # Check current configuration
298
+ config = get_ollama_config()
299
+ print(f"Current config: {config}")
300
+
301
+ # Reset to defaults if needed
302
+ from pyxtxt import reset_ollama_config
303
+ reset_ollama_config()
278
304
 
279
305
  # From web images
280
306
  import requests
@@ -282,6 +308,32 @@ image_response = requests.get("https://example.com/document.png")
282
308
  text = xtxt(image_response.content)
283
309
  ```
284
310
 
311
+ ### Command-Line OCR Example
312
+
313
+ A complete example script for command-line usage is available:
314
+
315
+ ```python
316
+ # Download and run the example script
317
+ import requests
318
+
319
+ example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
320
+ with open("ocr_example.py", "wb") as f:
321
+ f.write(requests.get(example_url).content)
322
+
323
+ # Usage examples:
324
+ # python ocr_example.py document.png
325
+ # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
326
+ # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
327
+ ```
328
+
329
+ The script supports:
330
+ - **OCR mode**: Extract only text from images
331
+ - **Describe mode**: Extract text + generate detailed captions
332
+ - **Language hints**: Specify caption language (italian, english, etc.)
333
+ - **Style control**: descriptive, technical, simple, detailed
334
+ - **Length control**: short, medium, long captions
335
+ - **Temperature**: Adjust LLM creativity (0.0-1.0)
336
+
285
337
  ### Show Available Formats
286
338
  ```python
287
339
  from pyxtxt import extxt_available_formats
@@ -333,15 +385,8 @@ text = xtxt(attachment_bytes)
333
385
 
334
386
  ## 📖 Full Examples
335
387
 
336
- See [examples.py](./examples.py) for comprehensive usage examples including:
337
- - Local file processing
338
- - Memory buffer handling
339
- - Web content extraction
340
- - Error handling patterns
341
- - All supported formats demonstration
342
-
343
388
  ### Accessing Examples After Installation
344
- After installing PyxTxt from PyPI, you can access the examples file:
389
+ After installing PyxTxt from PyPI, you can access comprehensive usage examples including local file processing, memory buffer handling, web content extraction, error handling patterns, and all supported formats demonstration:
345
390
 
346
391
  ```python
347
392
  import pkg_resources
@@ -350,7 +395,10 @@ import pkg_resources
350
395
  examples_path = pkg_resources.resource_filename('pyxtxt', 'examples.py')
351
396
  print(f"Examples file location: {examples_path}")
352
397
 
353
- # Or read the content directly
398
+ # Run the examples directly
399
+ exec(open(examples_path).read())
400
+
401
+ # Or read the content to view examples
354
402
  examples_content = pkg_resources.resource_string('pyxtxt', 'examples.py').decode('utf-8')
355
403
  print(examples_content)
356
404
  ```
@@ -371,7 +419,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
371
419
 
372
420
  ## 📊 Changelog
373
421
 
374
- ### v0.2.4
422
+ ### v0.2.5 (Current Development)
423
+ - ✅ **NEW**: AI-powered OCR with Ollama LLM integration
424
+ - ✅ **NEW**: Advanced caption generation with configurable parameters
425
+ - ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
426
+ - ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
427
+ - ✅ **NEW**: Caption length control (short/medium/long)
428
+ - ✅ **NEW**: Temperature and token limit configuration
429
+ - ✅ **NEW**: Command-line OCR example script with full parameter support
430
+ - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
431
+ - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
432
+
433
+ ### v0.2.4
375
434
  - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
376
435
  - ✅ **ENHANCED**: Audio transcription now supports video files
377
436
  - ✅ Whisper automatically extracts audio track from videos
@@ -379,15 +438,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
379
438
 
380
439
  ### v0.2.3
381
440
  - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
382
- - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
383
- - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
384
- - ✅ Separate optional dependencies for heavy features (audio/OCR)
385
- - ✅ Performance optimizations with model caching
386
- - ✅ Improved multilingual OCR support (Italian/English)
387
-
388
- ### v0.1.24+
389
- - ✅ Added support for `bytes` objects
390
- - ✅ Added support for `requests.Response` objects
391
- - ✅ Added `xtxt_from_url()` helper function
392
- - ✅ Improved type hints and error handling
393
- - ✅ Enhanced web content processing capabilities
441
+ - ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
442
+ - ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
443
+ - ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
444
+ - ✅ Performance optimizations with model caching for heavy operations
445
+ - ✅ Improved multilingual OCR support with automatic language detection
446
+
447
+ ### v0.2.0-0.2.2
448
+ - ✅ **MAJOR**: Architectural improvements with automatic extractor registration
449
+ - ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
450
+ - ✅ **FIXED**: Critical memory management issues in MSG extractor
451
+ - ✅ **FIXED**: Documentation links and path references
452
+ - ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
453
+ - ✅ Comprehensive testing across all newly supported formats
454
+
455
+ ### v0.1.24
456
+ - ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
457
+ - ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
458
+ - ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
459
+ - ✅ **ENHANCED**: Web-ready architecture for modern applications
460
+ - ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
461
+ - ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
462
+ - ✅ **REMOVED**: Debug print statements from production code
463
+
464
+ ### v0.1.0-0.1.23
465
+ - ✅ **CORE**: Initial release with modular extractor architecture
466
+ - ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
467
+ - ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
468
+ - ✅ **CORE**: MIME type detection with python-magic
469
+ - ✅ **CORE**: BytesIO buffer support for memory-efficient processing
470
+ - ✅ **CORE**: Single dispatch pattern for type-based routing
471
+ - ✅ **CORE**: Automatic dependency management with optional installs
472
+ - ✅ **CORE**: Published to PyPI with proper package structure
@@ -160,17 +160,43 @@ text = xtxt("invoice.tiff")
160
160
 
161
161
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
162
162
  # Requires: ollama server running + gemma3:4b model
163
- from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
163
+ from pyxtxt import (
164
+ xtxt, xtxt_image_describe,
165
+ set_ollama_model, set_ollama_config, get_ollama_config
166
+ )
164
167
 
165
168
  # Configure model (optional, default is gemma3:4b)
166
- set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
167
-
168
- # Extract only text (OCR mode)
169
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
170
+
171
+ # Configure LLM parameters for better captions
172
+ set_ollama_config(
173
+ language='italian', # Language hint for captions
174
+ caption_length='long', # short, medium, long
175
+ style='detailed', # descriptive, technical, simple, detailed
176
+ temperature=0.2, # Creativity level (0.0-1.0)
177
+ max_tokens=2000 # Maximum response length
178
+ )
179
+
180
+ # Extract only text (OCR mode)
169
181
  text = xtxt("complex_document.png")
182
+ print(f"Extracted text: {text}")
183
+
184
+ # Extract text + detailed caption
185
+ full_analysis = xtxt_image_describe("scientific_diagram.png")
186
+ print(full_analysis)
187
+ # Output example:
188
+ # TEXT: Figura 2.1: Struttura molecolare del DNA
189
+ # DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
190
+ # con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
191
+ # le basi azotate (adenina, timina, citosina, guanina).
170
192
 
171
- # Extract text + image description
172
- description = xtxt_image_describe(open("diagram.png", "rb"))
173
- # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
193
+ # Check current configuration
194
+ config = get_ollama_config()
195
+ print(f"Current config: {config}")
196
+
197
+ # Reset to defaults if needed
198
+ from pyxtxt import reset_ollama_config
199
+ reset_ollama_config()
174
200
 
175
201
  # From web images
176
202
  import requests
@@ -178,6 +204,32 @@ image_response = requests.get("https://example.com/document.png")
178
204
  text = xtxt(image_response.content)
179
205
  ```
180
206
 
207
+ ### Command-Line OCR Example
208
+
209
+ A complete example script for command-line usage is available:
210
+
211
+ ```python
212
+ # Download and run the example script
213
+ import requests
214
+
215
+ example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
216
+ with open("ocr_example.py", "wb") as f:
217
+ f.write(requests.get(example_url).content)
218
+
219
+ # Usage examples:
220
+ # python ocr_example.py document.png
221
+ # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
222
+ # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
223
+ ```
224
+
225
+ The script supports:
226
+ - **OCR mode**: Extract only text from images
227
+ - **Describe mode**: Extract text + generate detailed captions
228
+ - **Language hints**: Specify caption language (italian, english, etc.)
229
+ - **Style control**: descriptive, technical, simple, detailed
230
+ - **Length control**: short, medium, long captions
231
+ - **Temperature**: Adjust LLM creativity (0.0-1.0)
232
+
181
233
  ### Show Available Formats
182
234
  ```python
183
235
  from pyxtxt import extxt_available_formats
@@ -229,15 +281,8 @@ text = xtxt(attachment_bytes)
229
281
 
230
282
  ## 📖 Full Examples
231
283
 
232
- See [examples.py](./examples.py) for comprehensive usage examples including:
233
- - Local file processing
234
- - Memory buffer handling
235
- - Web content extraction
236
- - Error handling patterns
237
- - All supported formats demonstration
238
-
239
284
  ### Accessing Examples After Installation
240
- After installing PyxTxt from PyPI, you can access the examples file:
285
+ After installing PyxTxt from PyPI, you can access comprehensive usage examples including local file processing, memory buffer handling, web content extraction, error handling patterns, and all supported formats demonstration:
241
286
 
242
287
  ```python
243
288
  import pkg_resources
@@ -246,7 +291,10 @@ import pkg_resources
246
291
  examples_path = pkg_resources.resource_filename('pyxtxt', 'examples.py')
247
292
  print(f"Examples file location: {examples_path}")
248
293
 
249
- # Or read the content directly
294
+ # Run the examples directly
295
+ exec(open(examples_path).read())
296
+
297
+ # Or read the content to view examples
250
298
  examples_content = pkg_resources.resource_string('pyxtxt', 'examples.py').decode('utf-8')
251
299
  print(examples_content)
252
300
  ```
@@ -267,7 +315,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
267
315
 
268
316
  ## 📊 Changelog
269
317
 
270
- ### v0.2.4
318
+ ### v0.2.5 (Current Development)
319
+ - ✅ **NEW**: AI-powered OCR with Ollama LLM integration
320
+ - ✅ **NEW**: Advanced caption generation with configurable parameters
321
+ - ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
322
+ - ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
323
+ - ✅ **NEW**: Caption length control (short/medium/long)
324
+ - ✅ **NEW**: Temperature and token limit configuration
325
+ - ✅ **NEW**: Command-line OCR example script with full parameter support
326
+ - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
327
+ - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
328
+
329
+ ### v0.2.4
271
330
  - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
272
331
  - ✅ **ENHANCED**: Audio transcription now supports video files
273
332
  - ✅ Whisper automatically extracts audio track from videos
@@ -275,15 +334,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
275
334
 
276
335
  ### v0.2.3
277
336
  - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
278
- - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
279
- - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
280
- - ✅ Separate optional dependencies for heavy features (audio/OCR)
281
- - ✅ Performance optimizations with model caching
282
- - ✅ Improved multilingual OCR support (Italian/English)
283
-
284
- ### v0.1.24+
285
- - ✅ Added support for `bytes` objects
286
- - ✅ Added support for `requests.Response` objects
287
- - ✅ Added `xtxt_from_url()` helper function
288
- - ✅ Improved type hints and error handling
289
- - ✅ Enhanced web content processing capabilities
337
+ - ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
338
+ - ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
339
+ - ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
340
+ - ✅ Performance optimizations with model caching for heavy operations
341
+ - ✅ Improved multilingual OCR support with automatic language detection
342
+
343
+ ### v0.2.0-0.2.2
344
+ - ✅ **MAJOR**: Architectural improvements with automatic extractor registration
345
+ - ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
346
+ - ✅ **FIXED**: Critical memory management issues in MSG extractor
347
+ - ✅ **FIXED**: Documentation links and path references
348
+ - ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
349
+ - ✅ Comprehensive testing across all newly supported formats
350
+
351
+ ### v0.1.24
352
+ - ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
353
+ - ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
354
+ - ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
355
+ - ✅ **ENHANCED**: Web-ready architecture for modern applications
356
+ - ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
357
+ - ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
358
+ - ✅ **REMOVED**: Debug print statements from production code
359
+
360
+ ### v0.1.0-0.1.23
361
+ - ✅ **CORE**: Initial release with modular extractor architecture
362
+ - ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
363
+ - ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
364
+ - ✅ **CORE**: MIME type detection with python-magic
365
+ - ✅ **CORE**: BytesIO buffer support for memory-efficient processing
366
+ - ✅ **CORE**: Single dispatch pattern for type-based routing
367
+ - ✅ **CORE**: Automatic dependency management with optional installs
368
+ - ✅ **CORE**: Published to PyPI with proper package structure
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.3"
3
+ version = "0.3.2"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -0,0 +1,15 @@
1
+ from .core import xtxt, extxt_available_formats, xtxt_from_url
2
+
3
+ # Import OCR-Ollama functions if available
4
+ try:
5
+ from .estrattori.ocr_ollama import (
6
+ set_ollama_model, get_ollama_model, xtxt_image_describe,
7
+ set_ollama_config, get_ollama_config, reset_ollama_config
8
+ )
9
+ __all__ = [
10
+ "xtxt", "extxt_available_formats", "xtxt_from_url",
11
+ "set_ollama_model", "get_ollama_model", "xtxt_image_describe",
12
+ "set_ollama_config", "get_ollama_config", "reset_ollama_config"
13
+ ]
14
+ except ImportError:
15
+ __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
@@ -10,8 +10,16 @@ except ImportError:
10
10
  ollama = None
11
11
  Image = None
12
12
 
13
- # Global configuration for Ollama model
13
+ # Global configuration for Ollama model and parameters
14
14
  OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
15
+ OLLAMA_CONFIG = {
16
+ 'language': 'auto', # Language hint for caption generation
17
+ 'caption_length': 'medium', # short, medium, long
18
+ 'style': 'descriptive', # descriptive, technical, simple, detailed
19
+ 'temperature': 0.1, # Response creativity (0.0-1.0)
20
+ 'max_tokens': 1500, # Maximum response length
21
+ 'confidence_threshold': 0.7 # Minimum confidence for text extraction
22
+ }
15
23
 
16
24
  def set_ollama_model(model_name: str):
17
25
  """
@@ -32,6 +40,47 @@ def get_ollama_model():
32
40
  """Get current Ollama model name"""
33
41
  return OLLAMA_MODEL
34
42
 
43
+ def set_ollama_config(**kwargs):
44
+ """
45
+ Configure Ollama LLM parameters for better caption generation.
46
+
47
+ Parameters:
48
+ - language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
49
+ - caption_length: Caption length ('short', 'medium', 'long')
50
+ - style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
51
+ - temperature: Response creativity 0.0-1.0 (default: 0.1)
52
+ - max_tokens: Maximum response length (default: 1500)
53
+ - confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
54
+
55
+ Examples:
56
+ set_ollama_config(language='italian', style='detailed')
57
+ set_ollama_config(caption_length='long', temperature=0.3)
58
+ """
59
+ global OLLAMA_CONFIG
60
+ for key, value in kwargs.items():
61
+ if key in OLLAMA_CONFIG:
62
+ OLLAMA_CONFIG[key] = value
63
+ print(f"✅ Ollama config updated: {key} = {value}")
64
+ else:
65
+ print(f"⚠️ Unknown config parameter: {key}")
66
+
67
+ def get_ollama_config():
68
+ """Get current Ollama configuration"""
69
+ return OLLAMA_CONFIG.copy()
70
+
71
+ def reset_ollama_config():
72
+ """Reset Ollama configuration to defaults"""
73
+ global OLLAMA_CONFIG
74
+ OLLAMA_CONFIG = {
75
+ 'language': 'auto',
76
+ 'caption_length': 'medium',
77
+ 'style': 'descriptive',
78
+ 'temperature': 0.1,
79
+ 'max_tokens': 1500,
80
+ 'confidence_threshold': 0.7
81
+ }
82
+ print("✅ Ollama configuration reset to defaults")
83
+
35
84
  if ollama and Image:
36
85
  def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
37
86
  """
@@ -58,37 +107,65 @@ if ollama and Image:
58
107
  image.save(buffered, format="PNG")
59
108
  img_base64 = base64.b64encode(buffered.getvalue()).decode()
60
109
 
110
+ # Get current configuration
111
+ config = OLLAMA_CONFIG
112
+
113
+ # Build language hint
114
+ lang_hint = ""
115
+ if config['language'] != 'auto':
116
+ lang_hint = f"Text language: {config['language']}. "
117
+
61
118
  # Different prompts based on mode
62
119
  if mode == "ocr":
63
- prompt = """Extract ALL visible text from this image exactly as it appears.
120
+ prompt = f"""Extract ALL visible text from this image exactly as it appears.
64
121
  Rules:
65
122
  - Only return text that is actually written/printed in the image
66
123
  - Preserve reading order (left to right, top to bottom)
67
124
  - Maintain line breaks and formatting
68
125
  - Include numbers, symbols, special characters
69
126
  - Do NOT add descriptions, interpretations, or context
70
- - If no text is visible, return 'NO_TEXT_FOUND'
127
+ - {lang_hint}If no text is visible, return 'NO_TEXT_FOUND'
71
128
 
72
129
  Extracted text:"""
73
130
 
74
131
  else: # mode == "describe"
75
- prompt = """Analyze this image and provide:
132
+ # Build style-specific prompts
133
+ style_prompts = {
134
+ 'descriptive': "Provide a clear, descriptive explanation",
135
+ 'technical': "Use technical terminology and precise descriptions",
136
+ 'simple': "Use simple, easy-to-understand language",
137
+ 'detailed': "Provide comprehensive details about all visual elements"
138
+ }
139
+
140
+ length_hints = {
141
+ 'short': "Keep descriptions brief (1-2 sentences)",
142
+ 'medium': "Provide moderate detail (2-4 sentences)",
143
+ 'long': "Give comprehensive descriptions (4-8 sentences)"
144
+ }
145
+
146
+ style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
147
+ length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
148
+
149
+ prompt = f"""Analyze this image and provide:
76
150
  1. All visible text exactly as written
77
- 2. Brief description of the image content and context
151
+ 2. Image description following these guidelines:
152
+ - {style_instruction}
153
+ - {length_instruction}
154
+ - {lang_hint}Focus on key visual elements, layout, and context
78
155
 
79
156
  Format:
80
157
  TEXT: [all visible text here, or NO_TEXT_FOUND if none]
81
- DESCRIPTION: [brief image description and context]"""
158
+ DESCRIPTION: [image description following the guidelines above]"""
82
159
 
83
- # Send request to Ollama
160
+ # Send request to Ollama with configured parameters
84
161
  response = ollama.generate(
85
162
  model=current_model,
86
163
  prompt=prompt,
87
164
  images=[img_base64],
88
165
  options={
89
- 'temperature': 0.1, # Low temperature for accuracy
166
+ 'temperature': config['temperature'],
90
167
  'top_p': 0.9,
91
- 'num_predict': 1500
168
+ 'num_predict': config['max_tokens']
92
169
  }
93
170
  )
94
171
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3
3
+ Version: 0.3.2
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -264,17 +264,43 @@ text = xtxt("invoice.tiff")
264
264
 
265
265
  # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
266
266
  # Requires: ollama server running + gemma3:4b model
267
- from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
267
+ from pyxtxt import (
268
+ xtxt, xtxt_image_describe,
269
+ set_ollama_model, set_ollama_config, get_ollama_config
270
+ )
268
271
 
269
272
  # Configure model (optional, default is gemma3:4b)
270
- set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
271
-
272
- # Extract only text (OCR mode)
273
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
274
+
275
+ # Configure LLM parameters for better captions
276
+ set_ollama_config(
277
+ language='italian', # Language hint for captions
278
+ caption_length='long', # short, medium, long
279
+ style='detailed', # descriptive, technical, simple, detailed
280
+ temperature=0.2, # Creativity level (0.0-1.0)
281
+ max_tokens=2000 # Maximum response length
282
+ )
283
+
284
+ # Extract only text (OCR mode)
273
285
  text = xtxt("complex_document.png")
286
+ print(f"Extracted text: {text}")
287
+
288
+ # Extract text + detailed caption
289
+ full_analysis = xtxt_image_describe("scientific_diagram.png")
290
+ print(full_analysis)
291
+ # Output example:
292
+ # TEXT: Figura 2.1: Struttura molecolare del DNA
293
+ # DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
294
+ # con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
295
+ # le basi azotate (adenina, timina, citosina, guanina).
274
296
 
275
- # Extract text + image description
276
- description = xtxt_image_describe(open("diagram.png", "rb"))
277
- # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
297
+ # Check current configuration
298
+ config = get_ollama_config()
299
+ print(f"Current config: {config}")
300
+
301
+ # Reset to defaults if needed
302
+ from pyxtxt import reset_ollama_config
303
+ reset_ollama_config()
278
304
 
279
305
  # From web images
280
306
  import requests
@@ -282,6 +308,32 @@ image_response = requests.get("https://example.com/document.png")
282
308
  text = xtxt(image_response.content)
283
309
  ```
284
310
 
311
+ ### Command-Line OCR Example
312
+
313
+ A complete example script for command-line usage is available:
314
+
315
+ ```python
316
+ # Download and run the example script
317
+ import requests
318
+
319
+ example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
320
+ with open("ocr_example.py", "wb") as f:
321
+ f.write(requests.get(example_url).content)
322
+
323
+ # Usage examples:
324
+ # python ocr_example.py document.png
325
+ # python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
326
+ # python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
327
+ ```
328
+
329
+ The script supports:
330
+ - **OCR mode**: Extract only text from images
331
+ - **Describe mode**: Extract text + generate detailed captions
332
+ - **Language hints**: Specify caption language (italian, english, etc.)
333
+ - **Style control**: descriptive, technical, simple, detailed
334
+ - **Length control**: short, medium, long captions
335
+ - **Temperature**: Adjust LLM creativity (0.0-1.0)
336
+
285
337
  ### Show Available Formats
286
338
  ```python
287
339
  from pyxtxt import extxt_available_formats
@@ -333,15 +385,8 @@ text = xtxt(attachment_bytes)
333
385
 
334
386
  ## 📖 Full Examples
335
387
 
336
- See [examples.py](./examples.py) for comprehensive usage examples including:
337
- - Local file processing
338
- - Memory buffer handling
339
- - Web content extraction
340
- - Error handling patterns
341
- - All supported formats demonstration
342
-
343
388
  ### Accessing Examples After Installation
344
- After installing PyxTxt from PyPI, you can access the examples file:
389
+ After installing PyxTxt from PyPI, you can access comprehensive usage examples including local file processing, memory buffer handling, web content extraction, error handling patterns, and all supported formats demonstration:
345
390
 
346
391
  ```python
347
392
  import pkg_resources
@@ -350,7 +395,10 @@ import pkg_resources
350
395
  examples_path = pkg_resources.resource_filename('pyxtxt', 'examples.py')
351
396
  print(f"Examples file location: {examples_path}")
352
397
 
353
- # Or read the content directly
398
+ # Run the examples directly
399
+ exec(open(examples_path).read())
400
+
401
+ # Or read the content to view examples
354
402
  examples_content = pkg_resources.resource_string('pyxtxt', 'examples.py').decode('utf-8')
355
403
  print(examples_content)
356
404
  ```
@@ -371,7 +419,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
371
419
 
372
420
  ## 📊 Changelog
373
421
 
374
- ### v0.2.4
422
+ ### v0.2.5 (Current Development)
423
+ - ✅ **NEW**: AI-powered OCR with Ollama LLM integration
424
+ - ✅ **NEW**: Advanced caption generation with configurable parameters
425
+ - ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
426
+ - ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
427
+ - ✅ **NEW**: Caption length control (short/medium/long)
428
+ - ✅ **NEW**: Temperature and token limit configuration
429
+ - ✅ **NEW**: Command-line OCR example script with full parameter support
430
+ - ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
431
+ - ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
432
+
433
+ ### v0.2.4
375
434
  - ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
376
435
  - ✅ **ENHANCED**: Audio transcription now supports video files
377
436
  - ✅ Whisper automatically extracts audio track from videos
@@ -379,15 +438,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
379
438
 
380
439
  ### v0.2.3
381
440
  - ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
382
- - ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
383
- - ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX support
384
- - ✅ Separate optional dependencies for heavy features (audio/OCR)
385
- - ✅ Performance optimizations with model caching
386
- - ✅ Improved multilingual OCR support (Italian/English)
387
-
388
- ### v0.1.24+
389
- - ✅ Added support for `bytes` objects
390
- - ✅ Added support for `requests.Response` objects
391
- - ✅ Added `xtxt_from_url()` helper function
392
- - ✅ Improved type hints and error handling
393
- - ✅ Enhanced web content processing capabilities
441
+ - ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
442
+ - ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
443
+ - ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
444
+ - ✅ Performance optimizations with model caching for heavy operations
445
+ - ✅ Improved multilingual OCR support with automatic language detection
446
+
447
+ ### v0.2.0-0.2.2
448
+ - ✅ **MAJOR**: Architectural improvements with automatic extractor registration
449
+ - ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
450
+ - ✅ **FIXED**: Critical memory management issues in MSG extractor
451
+ - ✅ **FIXED**: Documentation links and path references
452
+ - ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
453
+ - ✅ Comprehensive testing across all newly supported formats
454
+
455
+ ### v0.1.24
456
+ - ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
457
+ - ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
458
+ - ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
459
+ - ✅ **ENHANCED**: Web-ready architecture for modern applications
460
+ - ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
461
+ - ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
462
+ - ✅ **REMOVED**: Debug print statements from production code
463
+
464
+ ### v0.1.0-0.1.23
465
+ - ✅ **CORE**: Initial release with modular extractor architecture
466
+ - ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
467
+ - ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
468
+ - ✅ **CORE**: MIME type detection with python-magic
469
+ - ✅ **CORE**: BytesIO buffer support for memory-efficient processing
470
+ - ✅ **CORE**: Single dispatch pattern for type-based routing
471
+ - ✅ **CORE**: Automatic dependency management with optional installs
472
+ - ✅ **CORE**: Published to PyPI with proper package structure
@@ -1,10 +1,10 @@
1
1
  LICENSE
2
2
  MANIFEST.in
3
3
  README.md
4
- examples.py
5
4
  pyproject.toml
6
5
  src/pyxtxt/__init__.py
7
6
  src/pyxtxt/core.py
7
+ src/pyxtxt/examples.py
8
8
  src/pyxtxt/pyxtxt.py
9
9
  src/pyxtxt.egg-info/PKG-INFO
10
10
  src/pyxtxt.egg-info/SOURCES.txt
@@ -1,8 +0,0 @@
1
- from .core import xtxt, extxt_available_formats, xtxt_from_url
2
-
3
- # Import OCR-Ollama functions if available
4
- try:
5
- from .estrattori.ocr_ollama import set_ollama_model, get_ollama_model, xtxt_image_describe
6
- __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url", "set_ollama_model", "get_ollama_model", "xtxt_image_describe"]
7
- except ImportError:
8
- __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes