pyxtxt 0.3.1__tar.gz → 0.3.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyxtxt-0.3.1/src/pyxtxt.egg-info → pyxtxt-0.3.3}/PKG-INFO +104 -21
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/README.md +103 -20
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/pyproject.toml +1 -1
- pyxtxt-0.3.3/src/pyxtxt/__init__.py +15 -0
- pyxtxt-0.3.3/src/pyxtxt/estrattori/ocr_ollama.py +245 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3/src/pyxtxt.egg-info}/PKG-INFO +104 -21
- pyxtxt-0.3.1/src/pyxtxt/__init__.py +0 -8
- pyxtxt-0.3.1/src/pyxtxt/estrattori/ocr_ollama.py +0 -125
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/LICENSE +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/MANIFEST.in +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/setup.cfg +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/core.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/__init__.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/audio.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/doc.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/docx.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/eml.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/epub.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/html.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/md.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/msg.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/ocr.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/odt.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/pdf.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/pptx.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/rtf.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/svg.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/tex.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/txt.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/xls.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/xlsx.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/xml.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/examples.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt/pyxtxt.py +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/requires.txt +0 -0
- {pyxtxt-0.3.1 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.3
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -264,17 +264,43 @@ text = xtxt("invoice.tiff")
|
|
|
264
264
|
|
|
265
265
|
# AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
|
|
266
266
|
# Requires: ollama server running + gemma3:4b model
|
|
267
|
-
from pyxtxt
|
|
267
|
+
from pyxtxt import (
|
|
268
|
+
xtxt, xtxt_image_describe,
|
|
269
|
+
set_ollama_model, set_ollama_config, get_ollama_config
|
|
270
|
+
)
|
|
268
271
|
|
|
269
272
|
# Configure model (optional, default is gemma3:4b)
|
|
270
|
-
set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
|
|
271
|
-
|
|
272
|
-
#
|
|
273
|
+
set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
|
|
274
|
+
|
|
275
|
+
# Configure LLM parameters for better captions
|
|
276
|
+
set_ollama_config(
|
|
277
|
+
language='italian', # Language hint for captions
|
|
278
|
+
caption_length='long', # short, medium, long
|
|
279
|
+
style='detailed', # descriptive, technical, simple, detailed
|
|
280
|
+
temperature=0.2, # Creativity level (0.0-1.0)
|
|
281
|
+
max_tokens=2000 # Maximum response length
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
# Extract only text (OCR mode)
|
|
273
285
|
text = xtxt("complex_document.png")
|
|
286
|
+
print(f"Extracted text: {text}")
|
|
287
|
+
|
|
288
|
+
# Extract text + detailed caption
|
|
289
|
+
full_analysis = xtxt_image_describe("scientific_diagram.png")
|
|
290
|
+
print(full_analysis)
|
|
291
|
+
# Output example:
|
|
292
|
+
# TEXT: Figura 2.1: Struttura molecolare del DNA
|
|
293
|
+
# DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
|
|
294
|
+
# con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
|
|
295
|
+
# le basi azotate (adenina, timina, citosina, guanina).
|
|
296
|
+
|
|
297
|
+
# Check current configuration
|
|
298
|
+
config = get_ollama_config()
|
|
299
|
+
print(f"Current config: {config}")
|
|
274
300
|
|
|
275
|
-
#
|
|
276
|
-
|
|
277
|
-
|
|
301
|
+
# Reset to defaults if needed
|
|
302
|
+
from pyxtxt import reset_ollama_config
|
|
303
|
+
reset_ollama_config()
|
|
278
304
|
|
|
279
305
|
# From web images
|
|
280
306
|
import requests
|
|
@@ -282,6 +308,32 @@ image_response = requests.get("https://example.com/document.png")
|
|
|
282
308
|
text = xtxt(image_response.content)
|
|
283
309
|
```
|
|
284
310
|
|
|
311
|
+
### Command-Line OCR Example
|
|
312
|
+
|
|
313
|
+
A complete example script for command-line usage is available:
|
|
314
|
+
|
|
315
|
+
```python
|
|
316
|
+
# Download and run the example script
|
|
317
|
+
import requests
|
|
318
|
+
|
|
319
|
+
example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
|
|
320
|
+
with open("ocr_example.py", "wb") as f:
|
|
321
|
+
f.write(requests.get(example_url).content)
|
|
322
|
+
|
|
323
|
+
# Usage examples:
|
|
324
|
+
# python ocr_example.py document.png
|
|
325
|
+
# python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
|
|
326
|
+
# python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
The script supports:
|
|
330
|
+
- **OCR mode**: Extract only text from images
|
|
331
|
+
- **Describe mode**: Extract text + generate detailed captions
|
|
332
|
+
- **Language hints**: Specify caption language (italian, english, etc.)
|
|
333
|
+
- **Style control**: descriptive, technical, simple, detailed
|
|
334
|
+
- **Length control**: short, medium, long captions
|
|
335
|
+
- **Temperature**: Adjust LLM creativity (0.0-1.0)
|
|
336
|
+
|
|
285
337
|
### Show Available Formats
|
|
286
338
|
```python
|
|
287
339
|
from pyxtxt import extxt_available_formats
|
|
@@ -367,7 +419,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
367
419
|
|
|
368
420
|
## 📊 Changelog
|
|
369
421
|
|
|
370
|
-
### v0.2.
|
|
422
|
+
### v0.2.5 (Current Development)
|
|
423
|
+
- ✅ **NEW**: AI-powered OCR with Ollama LLM integration
|
|
424
|
+
- ✅ **NEW**: Advanced caption generation with configurable parameters
|
|
425
|
+
- ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
|
|
426
|
+
- ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
|
|
427
|
+
- ✅ **NEW**: Caption length control (short/medium/long)
|
|
428
|
+
- ✅ **NEW**: Temperature and token limit configuration
|
|
429
|
+
- ✅ **NEW**: Command-line OCR example script with full parameter support
|
|
430
|
+
- ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
|
|
431
|
+
- ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
|
|
432
|
+
|
|
433
|
+
### v0.2.4
|
|
371
434
|
- ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
|
|
372
435
|
- ✅ **ENHANCED**: Audio transcription now supports video files
|
|
373
436
|
- ✅ Whisper automatically extracts audio track from videos
|
|
@@ -375,15 +438,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
375
438
|
|
|
376
439
|
### v0.2.3
|
|
377
440
|
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
378
|
-
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
379
|
-
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX
|
|
380
|
-
- ✅
|
|
381
|
-
- ✅ Performance optimizations with model caching
|
|
382
|
-
- ✅ Improved multilingual OCR support
|
|
383
|
-
|
|
384
|
-
### v0.
|
|
385
|
-
- ✅
|
|
386
|
-
- ✅
|
|
387
|
-
- ✅
|
|
388
|
-
- ✅
|
|
389
|
-
- ✅
|
|
441
|
+
- ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
|
|
442
|
+
- ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
|
|
443
|
+
- ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
|
|
444
|
+
- ✅ Performance optimizations with model caching for heavy operations
|
|
445
|
+
- ✅ Improved multilingual OCR support with automatic language detection
|
|
446
|
+
|
|
447
|
+
### v0.2.0-0.2.2
|
|
448
|
+
- ✅ **MAJOR**: Architectural improvements with automatic extractor registration
|
|
449
|
+
- ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
|
|
450
|
+
- ✅ **FIXED**: Critical memory management issues in MSG extractor
|
|
451
|
+
- ✅ **FIXED**: Documentation links and path references
|
|
452
|
+
- ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
|
|
453
|
+
- ✅ Comprehensive testing across all newly supported formats
|
|
454
|
+
|
|
455
|
+
### v0.1.24
|
|
456
|
+
- ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
|
|
457
|
+
- ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
|
|
458
|
+
- ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
|
|
459
|
+
- ✅ **ENHANCED**: Web-ready architecture for modern applications
|
|
460
|
+
- ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
|
|
461
|
+
- ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
|
|
462
|
+
- ✅ **REMOVED**: Debug print statements from production code
|
|
463
|
+
|
|
464
|
+
### v0.1.0-0.1.23
|
|
465
|
+
- ✅ **CORE**: Initial release with modular extractor architecture
|
|
466
|
+
- ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
|
|
467
|
+
- ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
|
|
468
|
+
- ✅ **CORE**: MIME type detection with python-magic
|
|
469
|
+
- ✅ **CORE**: BytesIO buffer support for memory-efficient processing
|
|
470
|
+
- ✅ **CORE**: Single dispatch pattern for type-based routing
|
|
471
|
+
- ✅ **CORE**: Automatic dependency management with optional installs
|
|
472
|
+
- ✅ **CORE**: Published to PyPI with proper package structure
|
|
@@ -160,17 +160,43 @@ text = xtxt("invoice.tiff")
|
|
|
160
160
|
|
|
161
161
|
# AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
|
|
162
162
|
# Requires: ollama server running + gemma3:4b model
|
|
163
|
-
from pyxtxt
|
|
163
|
+
from pyxtxt import (
|
|
164
|
+
xtxt, xtxt_image_describe,
|
|
165
|
+
set_ollama_model, set_ollama_config, get_ollama_config
|
|
166
|
+
)
|
|
164
167
|
|
|
165
168
|
# Configure model (optional, default is gemma3:4b)
|
|
166
|
-
set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
|
|
167
|
-
|
|
168
|
-
#
|
|
169
|
+
set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
|
|
170
|
+
|
|
171
|
+
# Configure LLM parameters for better captions
|
|
172
|
+
set_ollama_config(
|
|
173
|
+
language='italian', # Language hint for captions
|
|
174
|
+
caption_length='long', # short, medium, long
|
|
175
|
+
style='detailed', # descriptive, technical, simple, detailed
|
|
176
|
+
temperature=0.2, # Creativity level (0.0-1.0)
|
|
177
|
+
max_tokens=2000 # Maximum response length
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
# Extract only text (OCR mode)
|
|
169
181
|
text = xtxt("complex_document.png")
|
|
182
|
+
print(f"Extracted text: {text}")
|
|
183
|
+
|
|
184
|
+
# Extract text + detailed caption
|
|
185
|
+
full_analysis = xtxt_image_describe("scientific_diagram.png")
|
|
186
|
+
print(full_analysis)
|
|
187
|
+
# Output example:
|
|
188
|
+
# TEXT: Figura 2.1: Struttura molecolare del DNA
|
|
189
|
+
# DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
|
|
190
|
+
# con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
|
|
191
|
+
# le basi azotate (adenina, timina, citosina, guanina).
|
|
192
|
+
|
|
193
|
+
# Check current configuration
|
|
194
|
+
config = get_ollama_config()
|
|
195
|
+
print(f"Current config: {config}")
|
|
170
196
|
|
|
171
|
-
#
|
|
172
|
-
|
|
173
|
-
|
|
197
|
+
# Reset to defaults if needed
|
|
198
|
+
from pyxtxt import reset_ollama_config
|
|
199
|
+
reset_ollama_config()
|
|
174
200
|
|
|
175
201
|
# From web images
|
|
176
202
|
import requests
|
|
@@ -178,6 +204,32 @@ image_response = requests.get("https://example.com/document.png")
|
|
|
178
204
|
text = xtxt(image_response.content)
|
|
179
205
|
```
|
|
180
206
|
|
|
207
|
+
### Command-Line OCR Example
|
|
208
|
+
|
|
209
|
+
A complete example script for command-line usage is available:
|
|
210
|
+
|
|
211
|
+
```python
|
|
212
|
+
# Download and run the example script
|
|
213
|
+
import requests
|
|
214
|
+
|
|
215
|
+
example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
|
|
216
|
+
with open("ocr_example.py", "wb") as f:
|
|
217
|
+
f.write(requests.get(example_url).content)
|
|
218
|
+
|
|
219
|
+
# Usage examples:
|
|
220
|
+
# python ocr_example.py document.png
|
|
221
|
+
# python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
|
|
222
|
+
# python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
The script supports:
|
|
226
|
+
- **OCR mode**: Extract only text from images
|
|
227
|
+
- **Describe mode**: Extract text + generate detailed captions
|
|
228
|
+
- **Language hints**: Specify caption language (italian, english, etc.)
|
|
229
|
+
- **Style control**: descriptive, technical, simple, detailed
|
|
230
|
+
- **Length control**: short, medium, long captions
|
|
231
|
+
- **Temperature**: Adjust LLM creativity (0.0-1.0)
|
|
232
|
+
|
|
181
233
|
### Show Available Formats
|
|
182
234
|
```python
|
|
183
235
|
from pyxtxt import extxt_available_formats
|
|
@@ -263,7 +315,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
263
315
|
|
|
264
316
|
## 📊 Changelog
|
|
265
317
|
|
|
266
|
-
### v0.2.
|
|
318
|
+
### v0.2.5 (Current Development)
|
|
319
|
+
- ✅ **NEW**: AI-powered OCR with Ollama LLM integration
|
|
320
|
+
- ✅ **NEW**: Advanced caption generation with configurable parameters
|
|
321
|
+
- ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
|
|
322
|
+
- ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
|
|
323
|
+
- ✅ **NEW**: Caption length control (short/medium/long)
|
|
324
|
+
- ✅ **NEW**: Temperature and token limit configuration
|
|
325
|
+
- ✅ **NEW**: Command-line OCR example script with full parameter support
|
|
326
|
+
- ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
|
|
327
|
+
- ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
|
|
328
|
+
|
|
329
|
+
### v0.2.4
|
|
267
330
|
- ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
|
|
268
331
|
- ✅ **ENHANCED**: Audio transcription now supports video files
|
|
269
332
|
- ✅ Whisper automatically extracts audio track from videos
|
|
@@ -271,15 +334,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
271
334
|
|
|
272
335
|
### v0.2.3
|
|
273
336
|
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
274
|
-
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
275
|
-
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX
|
|
276
|
-
- ✅
|
|
277
|
-
- ✅ Performance optimizations with model caching
|
|
278
|
-
- ✅ Improved multilingual OCR support
|
|
279
|
-
|
|
280
|
-
### v0.
|
|
281
|
-
- ✅
|
|
282
|
-
- ✅
|
|
283
|
-
- ✅
|
|
284
|
-
- ✅
|
|
285
|
-
- ✅
|
|
337
|
+
- ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
|
|
338
|
+
- ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
|
|
339
|
+
- ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
|
|
340
|
+
- ✅ Performance optimizations with model caching for heavy operations
|
|
341
|
+
- ✅ Improved multilingual OCR support with automatic language detection
|
|
342
|
+
|
|
343
|
+
### v0.2.0-0.2.2
|
|
344
|
+
- ✅ **MAJOR**: Architectural improvements with automatic extractor registration
|
|
345
|
+
- ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
|
|
346
|
+
- ✅ **FIXED**: Critical memory management issues in MSG extractor
|
|
347
|
+
- ✅ **FIXED**: Documentation links and path references
|
|
348
|
+
- ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
|
|
349
|
+
- ✅ Comprehensive testing across all newly supported formats
|
|
350
|
+
|
|
351
|
+
### v0.1.24
|
|
352
|
+
- ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
|
|
353
|
+
- ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
|
|
354
|
+
- ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
|
|
355
|
+
- ✅ **ENHANCED**: Web-ready architecture for modern applications
|
|
356
|
+
- ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
|
|
357
|
+
- ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
|
|
358
|
+
- ✅ **REMOVED**: Debug print statements from production code
|
|
359
|
+
|
|
360
|
+
### v0.1.0-0.1.23
|
|
361
|
+
- ✅ **CORE**: Initial release with modular extractor architecture
|
|
362
|
+
- ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
|
|
363
|
+
- ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
|
|
364
|
+
- ✅ **CORE**: MIME type detection with python-magic
|
|
365
|
+
- ✅ **CORE**: BytesIO buffer support for memory-efficient processing
|
|
366
|
+
- ✅ **CORE**: Single dispatch pattern for type-based routing
|
|
367
|
+
- ✅ **CORE**: Automatic dependency management with optional installs
|
|
368
|
+
- ✅ **CORE**: Published to PyPI with proper package structure
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
from .core import xtxt, extxt_available_formats, xtxt_from_url
|
|
2
|
+
|
|
3
|
+
# Import OCR-Ollama functions if available
|
|
4
|
+
try:
|
|
5
|
+
from .estrattori.ocr_ollama import (
|
|
6
|
+
set_ollama_model, get_ollama_model, xtxt_image_describe,
|
|
7
|
+
set_ollama_config, get_ollama_config, reset_ollama_config
|
|
8
|
+
)
|
|
9
|
+
__all__ = [
|
|
10
|
+
"xtxt", "extxt_available_formats", "xtxt_from_url",
|
|
11
|
+
"set_ollama_model", "get_ollama_model", "xtxt_image_describe",
|
|
12
|
+
"set_ollama_config", "get_ollama_config", "reset_ollama_config"
|
|
13
|
+
]
|
|
14
|
+
except ImportError:
|
|
15
|
+
__all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
|
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
# pyxtxt/extractors/image_ocr_ollama.py
|
|
2
|
+
from . import register_extractor
|
|
3
|
+
from io import BytesIO
|
|
4
|
+
import base64
|
|
5
|
+
|
|
6
|
+
try:
|
|
7
|
+
import ollama
|
|
8
|
+
from PIL import Image
|
|
9
|
+
except ImportError:
|
|
10
|
+
ollama = None
|
|
11
|
+
Image = None
|
|
12
|
+
|
|
13
|
+
# Global configuration for Ollama model and parameters
|
|
14
|
+
OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
|
|
15
|
+
OLLAMA_CONFIG = {
|
|
16
|
+
'language': 'auto', # Language hint for caption generation
|
|
17
|
+
'caption_length': 'medium', # short, medium, long
|
|
18
|
+
'style': 'descriptive', # descriptive, technical, simple, detailed
|
|
19
|
+
'temperature': 0.1, # Response creativity (0.0-1.0)
|
|
20
|
+
'max_tokens': 1500, # Maximum response length
|
|
21
|
+
'confidence_threshold': 0.7, # Minimum confidence for text extraction
|
|
22
|
+
'context': 'general' # Context hint: general, cookbook, document, diagram, etc.
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
def set_ollama_model(model_name: str):
|
|
26
|
+
"""
|
|
27
|
+
Set the Ollama model to use for OCR.
|
|
28
|
+
|
|
29
|
+
Recommended multimodal models:
|
|
30
|
+
- gemma3:4b (default, balanced speed/quality)
|
|
31
|
+
- gemma3:12b (higher quality, slower)
|
|
32
|
+
- gemma3:27b (best quality, very slow)
|
|
33
|
+
- llava:7b (alternative vision model)
|
|
34
|
+
- llava:13b (higher quality LLAVA)
|
|
35
|
+
"""
|
|
36
|
+
global OLLAMA_MODEL
|
|
37
|
+
OLLAMA_MODEL = model_name
|
|
38
|
+
print(f"✅ Ollama OCR model set to: {model_name}")
|
|
39
|
+
|
|
40
|
+
def get_ollama_model():
|
|
41
|
+
"""Get current Ollama model name"""
|
|
42
|
+
return OLLAMA_MODEL
|
|
43
|
+
|
|
44
|
+
def set_ollama_config(**kwargs):
|
|
45
|
+
"""
|
|
46
|
+
Configure Ollama LLM parameters for better caption generation.
|
|
47
|
+
|
|
48
|
+
Parameters:
|
|
49
|
+
- language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
|
|
50
|
+
- caption_length: Caption length ('short', 'medium', 'long')
|
|
51
|
+
- style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
|
|
52
|
+
- context: Content context ('general', 'cookbook', 'document', 'handwriting', 'technical')
|
|
53
|
+
- temperature: Response creativity 0.0-1.0 (default: 0.1)
|
|
54
|
+
- max_tokens: Maximum response length (default: 1500)
|
|
55
|
+
- confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
|
|
56
|
+
|
|
57
|
+
Examples:
|
|
58
|
+
set_ollama_config(language='italian', style='detailed')
|
|
59
|
+
set_ollama_config(context='document', caption_length='long')
|
|
60
|
+
set_ollama_config(context='handwriting', temperature=0.2)
|
|
61
|
+
"""
|
|
62
|
+
global OLLAMA_CONFIG
|
|
63
|
+
for key, value in kwargs.items():
|
|
64
|
+
if key in OLLAMA_CONFIG:
|
|
65
|
+
OLLAMA_CONFIG[key] = value
|
|
66
|
+
print(f"✅ Ollama config updated: {key} = {value}")
|
|
67
|
+
else:
|
|
68
|
+
print(f"⚠️ Unknown config parameter: {key}")
|
|
69
|
+
|
|
70
|
+
def get_ollama_config():
|
|
71
|
+
"""Get current Ollama configuration"""
|
|
72
|
+
return OLLAMA_CONFIG.copy()
|
|
73
|
+
|
|
74
|
+
def reset_ollama_config():
|
|
75
|
+
"""Reset Ollama configuration to defaults"""
|
|
76
|
+
global OLLAMA_CONFIG
|
|
77
|
+
OLLAMA_CONFIG = {
|
|
78
|
+
'language': 'auto',
|
|
79
|
+
'caption_length': 'medium',
|
|
80
|
+
'style': 'descriptive',
|
|
81
|
+
'temperature': 0.1,
|
|
82
|
+
'max_tokens': 1500,
|
|
83
|
+
'confidence_threshold': 0.7,
|
|
84
|
+
'context': 'general'
|
|
85
|
+
}
|
|
86
|
+
print("✅ Ollama configuration reset to defaults")
|
|
87
|
+
|
|
88
|
+
if ollama and Image:
|
|
89
|
+
def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
|
|
90
|
+
"""
|
|
91
|
+
Extract text from images using Ollama with multimodal models.
|
|
92
|
+
|
|
93
|
+
Args:
|
|
94
|
+
file_buffer: Image file buffer
|
|
95
|
+
mode: "ocr" (text only) or "describe" (text + description)
|
|
96
|
+
model: Override default model (optional)
|
|
97
|
+
"""
|
|
98
|
+
try:
|
|
99
|
+
# Use specified model or global default
|
|
100
|
+
current_model = model or OLLAMA_MODEL
|
|
101
|
+
|
|
102
|
+
# Convert buffer to PIL Image
|
|
103
|
+
# Reset buffer position if it has read method
|
|
104
|
+
if hasattr(file_buffer, 'seek'):
|
|
105
|
+
file_buffer.seek(0)
|
|
106
|
+
image_data = file_buffer.read()
|
|
107
|
+
image = Image.open(BytesIO(image_data))
|
|
108
|
+
|
|
109
|
+
# Convert to RGB if needed
|
|
110
|
+
if image.mode != 'RGB':
|
|
111
|
+
image = image.convert('RGB')
|
|
112
|
+
|
|
113
|
+
# Convert image to base64
|
|
114
|
+
buffered = BytesIO()
|
|
115
|
+
image.save(buffered, format="PNG")
|
|
116
|
+
img_base64 = base64.b64encode(buffered.getvalue()).decode()
|
|
117
|
+
|
|
118
|
+
# Get current configuration
|
|
119
|
+
config = OLLAMA_CONFIG
|
|
120
|
+
|
|
121
|
+
# Build language hint
|
|
122
|
+
lang_hint = ""
|
|
123
|
+
if config['language'] != 'auto':
|
|
124
|
+
lang_hint = f"Text language: {config['language']}. "
|
|
125
|
+
|
|
126
|
+
# Different prompts based on mode
|
|
127
|
+
if mode == "ocr":
|
|
128
|
+
prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
|
|
129
|
+
- Titles, headings, and main text content
|
|
130
|
+
- Small print, captions, labels, and annotations
|
|
131
|
+
- Numbers, measurements, quantities, and symbols
|
|
132
|
+
- Menu items, ingredient lists, cooking instructions
|
|
133
|
+
- Any text in boxes, speech bubbles, or decorative elements
|
|
134
|
+
|
|
135
|
+
IMPORTANT:
|
|
136
|
+
- Read carefully and include even small or partially visible text
|
|
137
|
+
- Preserve the original formatting and line breaks where possible
|
|
138
|
+
- {lang_hint}Process the text from left to right, top to bottom
|
|
139
|
+
- If you cannot find any readable text at all, respond with 'NO_TEXT_FOUND'
|
|
140
|
+
|
|
141
|
+
Extracted text:"""
|
|
142
|
+
|
|
143
|
+
else: # mode == "describe"
|
|
144
|
+
# Build style-specific prompts
|
|
145
|
+
style_prompts = {
|
|
146
|
+
'descriptive': "Provide a clear, descriptive explanation",
|
|
147
|
+
'technical': "Use technical terminology and precise descriptions",
|
|
148
|
+
'simple': "Use simple, easy-to-understand language",
|
|
149
|
+
'detailed': "Provide comprehensive details about all visual elements"
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
length_hints = {
|
|
153
|
+
'short': "Keep descriptions brief (1-2 sentences)",
|
|
154
|
+
'medium': "Provide moderate detail (2-4 sentences)",
|
|
155
|
+
'long': "Give comprehensive descriptions (4-8 sentences)"
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
|
|
159
|
+
length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
|
|
160
|
+
|
|
161
|
+
# Context-specific hints (only when explicitly set)
|
|
162
|
+
context_hint = ""
|
|
163
|
+
context = config.get('context', 'general').lower()
|
|
164
|
+
if context == 'cookbook' or context == 'recipe':
|
|
165
|
+
context_hint = """
|
|
166
|
+
- If this appears to be a recipe/cookbook page, include: ingredients, cooking steps, quantities, cooking times
|
|
167
|
+
- Mention any photos of prepared dishes or cooking techniques shown
|
|
168
|
+
- Note any special formatting like ingredient lists, step numbers, or cooking tips"""
|
|
169
|
+
elif context == 'document':
|
|
170
|
+
context_hint = """
|
|
171
|
+
- Focus on document structure: headers, paragraphs, sections, page numbers
|
|
172
|
+
- Note any official formatting, letterheads, signatures, or stamps"""
|
|
173
|
+
elif context == 'handwriting' or context == 'notes':
|
|
174
|
+
context_hint = """
|
|
175
|
+
- Pay special attention to handwritten text which may be harder to read
|
|
176
|
+
- Note any sketches, diagrams, or informal formatting typical of personal notes"""
|
|
177
|
+
elif context == 'technical' or context == 'diagram':
|
|
178
|
+
context_hint = """
|
|
179
|
+
- Focus on technical elements: labels, measurements, specifications, diagrams
|
|
180
|
+
- Include any mathematical formulas, technical symbols, or engineering notations"""
|
|
181
|
+
|
|
182
|
+
prompt = f"""Analyze this image and provide:
|
|
183
|
+
1. All visible text exactly as written (preserve formatting, line breaks, bullet points)
|
|
184
|
+
2. Image description following these guidelines:
|
|
185
|
+
- {style_instruction}
|
|
186
|
+
- {length_instruction}
|
|
187
|
+
- {lang_hint}Focus on key visual elements, layout, and context{context_hint}
|
|
188
|
+
|
|
189
|
+
Format:
|
|
190
|
+
TEXT: [all visible text here, or NO_TEXT_FOUND if none]
|
|
191
|
+
DESCRIPTION: [image description following the guidelines above]"""
|
|
192
|
+
|
|
193
|
+
# Send request to Ollama with configured parameters
|
|
194
|
+
response = ollama.generate(
|
|
195
|
+
model=current_model,
|
|
196
|
+
prompt=prompt,
|
|
197
|
+
images=[img_base64],
|
|
198
|
+
options={
|
|
199
|
+
'temperature': config['temperature'],
|
|
200
|
+
'top_p': 0.9,
|
|
201
|
+
'num_predict': config['max_tokens']
|
|
202
|
+
}
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
# Extract and clean response
|
|
206
|
+
extracted_content = response.get('response', '').strip()
|
|
207
|
+
|
|
208
|
+
# Handle no-text case for OCR mode
|
|
209
|
+
if mode == "ocr" and ('NO_TEXT_FOUND' in extracted_content or len(extracted_content) < 3):
|
|
210
|
+
return ""
|
|
211
|
+
|
|
212
|
+
return extracted_content
|
|
213
|
+
|
|
214
|
+
except Exception as e:
|
|
215
|
+
print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
|
|
216
|
+
return ""
|
|
217
|
+
|
|
218
|
+
# Wrapper functions for each mode
|
|
219
|
+
def xtxt_image_ocr_only(file_input):
|
|
220
|
+
"""Traditional OCR: extract only visible text using Ollama"""
|
|
221
|
+
# Handle both file paths and buffers
|
|
222
|
+
if isinstance(file_input, str):
|
|
223
|
+
with open(file_input, 'rb') as f:
|
|
224
|
+
return xtxt_image_ocr_ollama(f, mode="ocr")
|
|
225
|
+
else:
|
|
226
|
+
return xtxt_image_ocr_ollama(file_input, mode="ocr")
|
|
227
|
+
|
|
228
|
+
def xtxt_image_describe(file_input):
|
|
229
|
+
"""OCR + Description: text + image context using Ollama"""
|
|
230
|
+
# Handle both file paths and buffers
|
|
231
|
+
if isinstance(file_input, str):
|
|
232
|
+
with open(file_input, 'rb') as f:
|
|
233
|
+
return xtxt_image_ocr_ollama(f, mode="describe")
|
|
234
|
+
else:
|
|
235
|
+
return xtxt_image_ocr_ollama(file_input, mode="describe")
|
|
236
|
+
|
|
237
|
+
# Register OCR-only version as default
|
|
238
|
+
# Note: Will override traditional EasyOCR if both modules are loaded
|
|
239
|
+
image_formats = [
|
|
240
|
+
"image/jpeg", "image/jpg", "image/png",
|
|
241
|
+
"image/bmp", "image/tiff", "image/webp"
|
|
242
|
+
]
|
|
243
|
+
|
|
244
|
+
for format_type in image_formats:
|
|
245
|
+
register_extractor(format_type, xtxt_image_ocr_only, name="OCR-Ollama")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyxtxt
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.3
|
|
4
4
|
Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
|
|
5
5
|
Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -264,17 +264,43 @@ text = xtxt("invoice.tiff")
|
|
|
264
264
|
|
|
265
265
|
# AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
|
|
266
266
|
# Requires: ollama server running + gemma3:4b model
|
|
267
|
-
from pyxtxt
|
|
267
|
+
from pyxtxt import (
|
|
268
|
+
xtxt, xtxt_image_describe,
|
|
269
|
+
set_ollama_model, set_ollama_config, get_ollama_config
|
|
270
|
+
)
|
|
268
271
|
|
|
269
272
|
# Configure model (optional, default is gemma3:4b)
|
|
270
|
-
set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
|
|
271
|
-
|
|
272
|
-
#
|
|
273
|
+
set_ollama_model("gemma3:12b") # or llava:7b, llava:13b, gemma3:27b
|
|
274
|
+
|
|
275
|
+
# Configure LLM parameters for better captions
|
|
276
|
+
set_ollama_config(
|
|
277
|
+
language='italian', # Language hint for captions
|
|
278
|
+
caption_length='long', # short, medium, long
|
|
279
|
+
style='detailed', # descriptive, technical, simple, detailed
|
|
280
|
+
temperature=0.2, # Creativity level (0.0-1.0)
|
|
281
|
+
max_tokens=2000 # Maximum response length
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
# Extract only text (OCR mode)
|
|
273
285
|
text = xtxt("complex_document.png")
|
|
286
|
+
print(f"Extracted text: {text}")
|
|
287
|
+
|
|
288
|
+
# Extract text + detailed caption
|
|
289
|
+
full_analysis = xtxt_image_describe("scientific_diagram.png")
|
|
290
|
+
print(full_analysis)
|
|
291
|
+
# Output example:
|
|
292
|
+
# TEXT: Figura 2.1: Struttura molecolare del DNA
|
|
293
|
+
# DESCRIPTION: Diagramma scientifico dettagliato che mostra la doppia elica del DNA
|
|
294
|
+
# con nucleotidi colorati, legami idrogeno evidenziati e etichette in italiano per
|
|
295
|
+
# le basi azotate (adenina, timina, citosina, guanina).
|
|
296
|
+
|
|
297
|
+
# Check current configuration
|
|
298
|
+
config = get_ollama_config()
|
|
299
|
+
print(f"Current config: {config}")
|
|
274
300
|
|
|
275
|
-
#
|
|
276
|
-
|
|
277
|
-
|
|
301
|
+
# Reset to defaults if needed
|
|
302
|
+
from pyxtxt import reset_ollama_config
|
|
303
|
+
reset_ollama_config()
|
|
278
304
|
|
|
279
305
|
# From web images
|
|
280
306
|
import requests
|
|
@@ -282,6 +308,32 @@ image_response = requests.get("https://example.com/document.png")
|
|
|
282
308
|
text = xtxt(image_response.content)
|
|
283
309
|
```
|
|
284
310
|
|
|
311
|
+
### Command-Line OCR Example
|
|
312
|
+
|
|
313
|
+
A complete example script for command-line usage is available:
|
|
314
|
+
|
|
315
|
+
```python
|
|
316
|
+
# Download and run the example script
|
|
317
|
+
import requests
|
|
318
|
+
|
|
319
|
+
example_url = "https://raw.githubusercontent.com/yourusername/pyxtxt/main/ocr_example.py"
|
|
320
|
+
with open("ocr_example.py", "wb") as f:
|
|
321
|
+
f.write(requests.get(example_url).content)
|
|
322
|
+
|
|
323
|
+
# Usage examples:
|
|
324
|
+
# python ocr_example.py document.png
|
|
325
|
+
# python ocr_example.py chart.jpg --mode=describe --lang=italian --style=detailed
|
|
326
|
+
# python ocr_example.py diagram.png --mode=describe --length=long --temp=0.3
|
|
327
|
+
```
|
|
328
|
+
|
|
329
|
+
The script supports:
|
|
330
|
+
- **OCR mode**: Extract only text from images
|
|
331
|
+
- **Describe mode**: Extract text + generate detailed captions
|
|
332
|
+
- **Language hints**: Specify caption language (italian, english, etc.)
|
|
333
|
+
- **Style control**: descriptive, technical, simple, detailed
|
|
334
|
+
- **Length control**: short, medium, long captions
|
|
335
|
+
- **Temperature**: Adjust LLM creativity (0.0-1.0)
|
|
336
|
+
|
|
285
337
|
### Show Available Formats
|
|
286
338
|
```python
|
|
287
339
|
from pyxtxt import extxt_available_formats
|
|
@@ -367,7 +419,18 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
367
419
|
|
|
368
420
|
## 📊 Changelog
|
|
369
421
|
|
|
370
|
-
### v0.2.
|
|
422
|
+
### v0.2.5 (Current Development)
|
|
423
|
+
- ✅ **NEW**: AI-powered OCR with Ollama LLM integration
|
|
424
|
+
- ✅ **NEW**: Advanced caption generation with configurable parameters
|
|
425
|
+
- ✅ **NEW**: `set_ollama_config()` for fine-tuning LLM behavior
|
|
426
|
+
- ✅ **NEW**: Language hints, style control (descriptive/technical/simple/detailed)
|
|
427
|
+
- ✅ **NEW**: Caption length control (short/medium/long)
|
|
428
|
+
- ✅ **NEW**: Temperature and token limit configuration
|
|
429
|
+
- ✅ **NEW**: Command-line OCR example script with full parameter support
|
|
430
|
+
- ✅ **ENHANCED**: OCR-Ollama mode with both text extraction and image description
|
|
431
|
+
- ✅ Support for gemma3:4b, gemma3:12b, gemma3:27b, llava:7b, llava:13b models
|
|
432
|
+
|
|
433
|
+
### v0.2.4
|
|
371
434
|
- ✅ **NEW**: Video transcription support (MP4, MOV, AVI, WebM, MKV)
|
|
372
435
|
- ✅ **ENHANCED**: Audio transcription now supports video files
|
|
373
436
|
- ✅ Whisper automatically extracts audio track from videos
|
|
@@ -375,15 +438,35 @@ Pull requests, issues, and feedback are warmly welcome! 🚀
|
|
|
375
438
|
|
|
376
439
|
### v0.2.3
|
|
377
440
|
- ✅ **NEW**: Audio transcription support (MP3, WAV, M4A, FLAC, etc.)
|
|
378
|
-
- ✅ **NEW**: OCR from images (JPEG, PNG, TIFF, BMP, WebP)
|
|
379
|
-
- ✅ **NEW**: Markdown, EPUB, RTF, EML, MSG, LaTeX
|
|
380
|
-
- ✅
|
|
381
|
-
- ✅ Performance optimizations with model caching
|
|
382
|
-
- ✅ Improved multilingual OCR support
|
|
383
|
-
|
|
384
|
-
### v0.
|
|
385
|
-
- ✅
|
|
386
|
-
- ✅
|
|
387
|
-
- ✅
|
|
388
|
-
- ✅
|
|
389
|
-
- ✅
|
|
441
|
+
- ✅ **NEW**: Traditional OCR from images (JPEG, PNG, TIFF, BMP, WebP) via EasyOCR
|
|
442
|
+
- ✅ **NEW**: 6 additional format extractors: Markdown, EPUB, RTF, EML, MSG, LaTeX
|
|
443
|
+
- ✅ **NEW**: Modular dependencies with `[audio]`, `[ocr]`, `[all]` installation groups
|
|
444
|
+
- ✅ Performance optimizations with model caching for heavy operations
|
|
445
|
+
- ✅ Improved multilingual OCR support with automatic language detection
|
|
446
|
+
|
|
447
|
+
### v0.2.0-0.2.2
|
|
448
|
+
- ✅ **MAJOR**: Architectural improvements with automatic extractor registration
|
|
449
|
+
- ✅ **NEW**: 6 format extractors added in single session (md, epub, rtf, eml, msg, tex)
|
|
450
|
+
- ✅ **FIXED**: Critical memory management issues in MSG extractor
|
|
451
|
+
- ✅ **FIXED**: Documentation links and path references
|
|
452
|
+
- ✅ **ENHANCED**: Error handling with graceful degradation for missing dependencies
|
|
453
|
+
- ✅ Comprehensive testing across all newly supported formats
|
|
454
|
+
|
|
455
|
+
### v0.1.24
|
|
456
|
+
- ✅ **NEW**: Support for raw `bytes` objects (web downloads, API responses)
|
|
457
|
+
- ✅ **NEW**: Support for `requests.Response` objects (direct HTTP processing)
|
|
458
|
+
- ✅ **NEW**: `xtxt_from_url()` helper function for direct URL processing
|
|
459
|
+
- ✅ **ENHANCED**: Web-ready architecture for modern applications
|
|
460
|
+
- ✅ **FIXED**: Type hints and Optional[str] return types throughout codebase
|
|
461
|
+
- ✅ **FIXED**: Critical bug in xlsx.py:46 (indentation error)
|
|
462
|
+
- ✅ **REMOVED**: Debug print statements from production code
|
|
463
|
+
|
|
464
|
+
### v0.1.0-0.1.23
|
|
465
|
+
- ✅ **CORE**: Initial release with modular extractor architecture
|
|
466
|
+
- ✅ **CORE**: Support for PDF, DOCX, PPTX, XLSX, ODT, HTML, XML, TXT formats
|
|
467
|
+
- ✅ **CORE**: Legacy Office support (.doc, .xls, .ppt) with graceful handling
|
|
468
|
+
- ✅ **CORE**: MIME type detection with python-magic
|
|
469
|
+
- ✅ **CORE**: BytesIO buffer support for memory-efficient processing
|
|
470
|
+
- ✅ **CORE**: Single dispatch pattern for type-based routing
|
|
471
|
+
- ✅ **CORE**: Automatic dependency management with optional installs
|
|
472
|
+
- ✅ **CORE**: Published to PyPI with proper package structure
|
|
@@ -1,8 +0,0 @@
|
|
|
1
|
-
from .core import xtxt, extxt_available_formats, xtxt_from_url
|
|
2
|
-
|
|
3
|
-
# Import OCR-Ollama functions if available
|
|
4
|
-
try:
|
|
5
|
-
from .estrattori.ocr_ollama import set_ollama_model, get_ollama_model, xtxt_image_describe
|
|
6
|
-
__all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url", "set_ollama_model", "get_ollama_model", "xtxt_image_describe"]
|
|
7
|
-
except ImportError:
|
|
8
|
-
__all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
|
|
@@ -1,125 +0,0 @@
|
|
|
1
|
-
# pyxtxt/extractors/image_ocr_ollama.py
|
|
2
|
-
from . import register_extractor
|
|
3
|
-
from io import BytesIO
|
|
4
|
-
import base64
|
|
5
|
-
|
|
6
|
-
try:
|
|
7
|
-
import ollama
|
|
8
|
-
from PIL import Image
|
|
9
|
-
except ImportError:
|
|
10
|
-
ollama = None
|
|
11
|
-
Image = None
|
|
12
|
-
|
|
13
|
-
# Global configuration for Ollama model
|
|
14
|
-
OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
|
|
15
|
-
|
|
16
|
-
def set_ollama_model(model_name: str):
|
|
17
|
-
"""
|
|
18
|
-
Set the Ollama model to use for OCR.
|
|
19
|
-
|
|
20
|
-
Recommended multimodal models:
|
|
21
|
-
- gemma3:4b (default, balanced speed/quality)
|
|
22
|
-
- gemma3:12b (higher quality, slower)
|
|
23
|
-
- gemma3:27b (best quality, very slow)
|
|
24
|
-
- llava:7b (alternative vision model)
|
|
25
|
-
- llava:13b (higher quality LLAVA)
|
|
26
|
-
"""
|
|
27
|
-
global OLLAMA_MODEL
|
|
28
|
-
OLLAMA_MODEL = model_name
|
|
29
|
-
print(f"✅ Ollama OCR model set to: {model_name}")
|
|
30
|
-
|
|
31
|
-
def get_ollama_model():
|
|
32
|
-
"""Get current Ollama model name"""
|
|
33
|
-
return OLLAMA_MODEL
|
|
34
|
-
|
|
35
|
-
if ollama and Image:
|
|
36
|
-
def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
|
|
37
|
-
"""
|
|
38
|
-
Extract text from images using Ollama with multimodal models.
|
|
39
|
-
|
|
40
|
-
Args:
|
|
41
|
-
file_buffer: Image file buffer
|
|
42
|
-
mode: "ocr" (text only) or "describe" (text + description)
|
|
43
|
-
model: Override default model (optional)
|
|
44
|
-
"""
|
|
45
|
-
try:
|
|
46
|
-
# Use specified model or global default
|
|
47
|
-
current_model = model or OLLAMA_MODEL
|
|
48
|
-
|
|
49
|
-
# Convert buffer to PIL Image
|
|
50
|
-
image = Image.open(BytesIO(file_buffer.read()))
|
|
51
|
-
|
|
52
|
-
# Convert to RGB if needed
|
|
53
|
-
if image.mode != 'RGB':
|
|
54
|
-
image = image.convert('RGB')
|
|
55
|
-
|
|
56
|
-
# Convert image to base64
|
|
57
|
-
buffered = BytesIO()
|
|
58
|
-
image.save(buffered, format="PNG")
|
|
59
|
-
img_base64 = base64.b64encode(buffered.getvalue()).decode()
|
|
60
|
-
|
|
61
|
-
# Different prompts based on mode
|
|
62
|
-
if mode == "ocr":
|
|
63
|
-
prompt = """Extract ALL visible text from this image exactly as it appears.
|
|
64
|
-
Rules:
|
|
65
|
-
- Only return text that is actually written/printed in the image
|
|
66
|
-
- Preserve reading order (left to right, top to bottom)
|
|
67
|
-
- Maintain line breaks and formatting
|
|
68
|
-
- Include numbers, symbols, special characters
|
|
69
|
-
- Do NOT add descriptions, interpretations, or context
|
|
70
|
-
- If no text is visible, return 'NO_TEXT_FOUND'
|
|
71
|
-
|
|
72
|
-
Extracted text:"""
|
|
73
|
-
|
|
74
|
-
else: # mode == "describe"
|
|
75
|
-
prompt = """Analyze this image and provide:
|
|
76
|
-
1. All visible text exactly as written
|
|
77
|
-
2. Brief description of the image content and context
|
|
78
|
-
|
|
79
|
-
Format:
|
|
80
|
-
TEXT: [all visible text here, or NO_TEXT_FOUND if none]
|
|
81
|
-
DESCRIPTION: [brief image description and context]"""
|
|
82
|
-
|
|
83
|
-
# Send request to Ollama
|
|
84
|
-
response = ollama.generate(
|
|
85
|
-
model=current_model,
|
|
86
|
-
prompt=prompt,
|
|
87
|
-
images=[img_base64],
|
|
88
|
-
options={
|
|
89
|
-
'temperature': 0.1, # Low temperature for accuracy
|
|
90
|
-
'top_p': 0.9,
|
|
91
|
-
'num_predict': 1500
|
|
92
|
-
}
|
|
93
|
-
)
|
|
94
|
-
|
|
95
|
-
# Extract and clean response
|
|
96
|
-
extracted_content = response.get('response', '').strip()
|
|
97
|
-
|
|
98
|
-
# Handle no-text case for OCR mode
|
|
99
|
-
if mode == "ocr" and ('NO_TEXT_FOUND' in extracted_content or len(extracted_content) < 3):
|
|
100
|
-
return ""
|
|
101
|
-
|
|
102
|
-
return extracted_content
|
|
103
|
-
|
|
104
|
-
except Exception as e:
|
|
105
|
-
print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
|
|
106
|
-
return ""
|
|
107
|
-
|
|
108
|
-
# Wrapper functions for each mode
|
|
109
|
-
def xtxt_image_ocr_only(file_buffer):
|
|
110
|
-
"""Traditional OCR: extract only visible text using Ollama"""
|
|
111
|
-
return xtxt_image_ocr_ollama(file_buffer, mode="ocr")
|
|
112
|
-
|
|
113
|
-
def xtxt_image_describe(file_buffer):
|
|
114
|
-
"""OCR + Description: text + image context using Ollama"""
|
|
115
|
-
return xtxt_image_ocr_ollama(file_buffer, mode="describe")
|
|
116
|
-
|
|
117
|
-
# Register OCR-only version as default
|
|
118
|
-
# Note: Will override traditional EasyOCR if both modules are loaded
|
|
119
|
-
image_formats = [
|
|
120
|
-
"image/jpeg", "image/jpg", "image/png",
|
|
121
|
-
"image/bmp", "image/tiff", "image/webp"
|
|
122
|
-
]
|
|
123
|
-
|
|
124
|
-
for format_type in image_formats:
|
|
125
|
-
register_extractor(format_type, xtxt_image_ocr_only, name="OCR-Ollama")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|