pyxtxt 0.2.4__tar.gz → 0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. {pyxtxt-0.2.4/src/pyxtxt.egg-info → pyxtxt-0.3}/PKG-INFO +26 -4
  2. {pyxtxt-0.2.4 → pyxtxt-0.3}/README.md +21 -3
  3. {pyxtxt-0.2.4 → pyxtxt-0.3}/pyproject.toml +6 -1
  4. pyxtxt-0.3/src/pyxtxt/__init__.py +8 -0
  5. pyxtxt-0.3/src/pyxtxt/estrattori/ocr_ollama.py +125 -0
  6. {pyxtxt-0.2.4 → pyxtxt-0.3/src/pyxtxt.egg-info}/PKG-INFO +26 -4
  7. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt.egg-info/SOURCES.txt +1 -0
  8. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt.egg-info/requires.txt +5 -0
  9. pyxtxt-0.2.4/src/pyxtxt/__init__.py +0 -1
  10. {pyxtxt-0.2.4 → pyxtxt-0.3}/LICENSE +0 -0
  11. {pyxtxt-0.2.4 → pyxtxt-0.3}/MANIFEST.in +0 -0
  12. {pyxtxt-0.2.4 → pyxtxt-0.3}/examples.py +0 -0
  13. {pyxtxt-0.2.4 → pyxtxt-0.3}/setup.cfg +0 -0
  14. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/core.py +0 -0
  15. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/__init__.py +0 -0
  16. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/audio.py +0 -0
  17. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/doc.py +0 -0
  18. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/docx.py +0 -0
  19. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/eml.py +0 -0
  20. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/epub.py +0 -0
  21. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/html.py +0 -0
  22. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/md.py +0 -0
  23. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/msg.py +0 -0
  24. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/ocr.py +0 -0
  25. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/odt.py +0 -0
  26. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/pdf.py +0 -0
  27. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/pptx.py +0 -0
  28. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/rtf.py +0 -0
  29. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/svg.py +0 -0
  30. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/tex.py +0 -0
  31. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/txt.py +0 -0
  32. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/xls.py +0 -0
  33. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/xlsx.py +0 -0
  34. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/estrattori/xml.py +0 -0
  35. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt/pyxtxt.py +0 -0
  36. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  37. {pyxtxt-0.2.4 → pyxtxt-0.3}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.2.4
3
+ Version: 0.3
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -78,6 +78,9 @@ Requires-Dist: openai-whisper; extra == "audio"
78
78
  Provides-Extra: ocr
79
79
  Requires-Dist: easyocr; extra == "ocr"
80
80
  Requires-Dist: pillow; extra == "ocr"
81
+ Provides-Extra: ocr-ollama
82
+ Requires-Dist: ollama; extra == "ocr-ollama"
83
+ Requires-Dist: pillow; extra == "ocr-ollama"
81
84
  Provides-Extra: all
82
85
  Requires-Dist: textract; extra == "all"
83
86
  Requires-Dist: PyMuPDF; extra == "all"
@@ -96,6 +99,7 @@ Requires-Dist: pylatexenc; extra == "all"
96
99
  Requires-Dist: openai-whisper; extra == "all"
97
100
  Requires-Dist: easyocr; extra == "all"
98
101
  Requires-Dist: pillow; extra == "all"
102
+ Requires-Dist: ollama; extra == "all"
99
103
  Dynamic: license-file
100
104
 
101
105
  # PyxTxt
@@ -141,10 +145,13 @@ pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
141
145
  # Audio transcription (~2GB download for Whisper models)
142
146
  pip install pyxtxt[audio]
143
147
 
144
- # OCR from images (~1GB download for EasyOCR models)
148
+ # Traditional OCR from images (~1GB download for EasyOCR models)
145
149
  pip install pyxtxt[ocr]
146
150
 
147
- # Both audio and OCR
151
+ # AI-powered OCR via Ollama (requires local Ollama + gemma3:4b model)
152
+ pip install pyxtxt[ocr-ollama]
153
+
154
+ # Both audio and traditional OCR
148
155
  pip install pyxtxt[audio,ocr]
149
156
  ```
150
157
 
@@ -185,6 +192,7 @@ Use python-magic-bin instead of python-magic for easier installation.
185
192
  - **LaTeX**: pylatexenc
186
193
  - **Audio**: openai-whisper (heavy ~2GB models)
187
194
  - **OCR**: easyocr, pillow (heavy ~1GB models)
195
+ - **OCR-Ollama**: ollama, pillow (requires local Ollama server)
188
196
 
189
197
  Dependencies are automatically installed based on selected optional groups.
190
198
 
@@ -249,11 +257,25 @@ text = xtxt(video_response.content)
249
257
  ```python
250
258
  from pyxtxt import xtxt
251
259
 
252
- # Extract text from images
260
+ # Traditional OCR with EasyOCR (install with: pip install pyxtxt[ocr])
253
261
  text = xtxt("scanned_document.png")
254
262
  text = xtxt("screenshot.jpg")
255
263
  text = xtxt("invoice.tiff")
256
264
 
265
+ # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
266
+ # Requires: ollama server running + gemma3:4b model
267
+ from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
268
+
269
+ # Configure model (optional, default is gemma3:4b)
270
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
271
+
272
+ # Extract only text (OCR mode)
273
+ text = xtxt("complex_document.png")
274
+
275
+ # Extract text + image description
276
+ description = xtxt_image_describe(open("diagram.png", "rb"))
277
+ # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
278
+
257
279
  # From web images
258
280
  import requests
259
281
  image_response = requests.get("https://example.com/document.png")
@@ -41,10 +41,13 @@ pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
41
41
  # Audio transcription (~2GB download for Whisper models)
42
42
  pip install pyxtxt[audio]
43
43
 
44
- # OCR from images (~1GB download for EasyOCR models)
44
+ # Traditional OCR from images (~1GB download for EasyOCR models)
45
45
  pip install pyxtxt[ocr]
46
46
 
47
- # Both audio and OCR
47
+ # AI-powered OCR via Ollama (requires local Ollama + gemma3:4b model)
48
+ pip install pyxtxt[ocr-ollama]
49
+
50
+ # Both audio and traditional OCR
48
51
  pip install pyxtxt[audio,ocr]
49
52
  ```
50
53
 
@@ -85,6 +88,7 @@ Use python-magic-bin instead of python-magic for easier installation.
85
88
  - **LaTeX**: pylatexenc
86
89
  - **Audio**: openai-whisper (heavy ~2GB models)
87
90
  - **OCR**: easyocr, pillow (heavy ~1GB models)
91
+ - **OCR-Ollama**: ollama, pillow (requires local Ollama server)
88
92
 
89
93
  Dependencies are automatically installed based on selected optional groups.
90
94
 
@@ -149,11 +153,25 @@ text = xtxt(video_response.content)
149
153
  ```python
150
154
  from pyxtxt import xtxt
151
155
 
152
- # Extract text from images
156
+ # Traditional OCR with EasyOCR (install with: pip install pyxtxt[ocr])
153
157
  text = xtxt("scanned_document.png")
154
158
  text = xtxt("screenshot.jpg")
155
159
  text = xtxt("invoice.tiff")
156
160
 
161
+ # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
162
+ # Requires: ollama server running + gemma3:4b model
163
+ from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
164
+
165
+ # Configure model (optional, default is gemma3:4b)
166
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
167
+
168
+ # Extract only text (OCR mode)
169
+ text = xtxt("complex_document.png")
170
+
171
+ # Extract text + image description
172
+ description = xtxt_image_describe(open("diagram.png", "rb"))
173
+ # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
174
+
157
175
  # From web images
158
176
  import requests
159
177
  image_response = requests.get("https://example.com/document.png")
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.2.4"
3
+ version = "0.3"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -80,6 +80,10 @@ ocr = [
80
80
  "easyocr",
81
81
  "pillow",
82
82
  ]
83
+ ocr-ollama = [
84
+ "ollama",
85
+ "pillow",
86
+ ]
83
87
  all = [
84
88
  "textract",
85
89
  "PyMuPDF",
@@ -98,6 +102,7 @@ all = [
98
102
  "openai-whisper",
99
103
  "easyocr",
100
104
  "pillow",
105
+ "ollama",
101
106
  ]
102
107
 
103
108
  [build-system]
@@ -0,0 +1,8 @@
1
+ from .core import xtxt, extxt_available_formats, xtxt_from_url
2
+
3
+ # Import OCR-Ollama functions if available
4
+ try:
5
+ from .estrattori.ocr_ollama import set_ollama_model, get_ollama_model, xtxt_image_describe
6
+ __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url", "set_ollama_model", "get_ollama_model", "xtxt_image_describe"]
7
+ except ImportError:
8
+ __all__ = ["xtxt", "extxt_available_formats", "xtxt_from_url"]
@@ -0,0 +1,125 @@
1
+ # pyxtxt/extractors/image_ocr_ollama.py
2
+ from . import register_extractor
3
+ from io import BytesIO
4
+ import base64
5
+
6
+ try:
7
+ import ollama
8
+ from PIL import Image
9
+ except ImportError:
10
+ ollama = None
11
+ Image = None
12
+
13
+ # Global configuration for Ollama model
14
+ OLLAMA_MODEL = "gemma3:4b" # Default multimodal model
15
+
16
+ def set_ollama_model(model_name: str):
17
+ """
18
+ Set the Ollama model to use for OCR.
19
+
20
+ Recommended multimodal models:
21
+ - gemma3:4b (default, balanced speed/quality)
22
+ - gemma3:12b (higher quality, slower)
23
+ - gemma3:27b (best quality, very slow)
24
+ - llava:7b (alternative vision model)
25
+ - llava:13b (higher quality LLAVA)
26
+ """
27
+ global OLLAMA_MODEL
28
+ OLLAMA_MODEL = model_name
29
+ print(f"✅ Ollama OCR model set to: {model_name}")
30
+
31
+ def get_ollama_model():
32
+ """Get current Ollama model name"""
33
+ return OLLAMA_MODEL
34
+
35
+ if ollama and Image:
36
+ def xtxt_image_ocr_ollama(file_buffer, mode="ocr", model=None):
37
+ """
38
+ Extract text from images using Ollama with multimodal models.
39
+
40
+ Args:
41
+ file_buffer: Image file buffer
42
+ mode: "ocr" (text only) or "describe" (text + description)
43
+ model: Override default model (optional)
44
+ """
45
+ try:
46
+ # Use specified model or global default
47
+ current_model = model or OLLAMA_MODEL
48
+
49
+ # Convert buffer to PIL Image
50
+ image = Image.open(BytesIO(file_buffer.read()))
51
+
52
+ # Convert to RGB if needed
53
+ if image.mode != 'RGB':
54
+ image = image.convert('RGB')
55
+
56
+ # Convert image to base64
57
+ buffered = BytesIO()
58
+ image.save(buffered, format="PNG")
59
+ img_base64 = base64.b64encode(buffered.getvalue()).decode()
60
+
61
+ # Different prompts based on mode
62
+ if mode == "ocr":
63
+ prompt = """Extract ALL visible text from this image exactly as it appears.
64
+ Rules:
65
+ - Only return text that is actually written/printed in the image
66
+ - Preserve reading order (left to right, top to bottom)
67
+ - Maintain line breaks and formatting
68
+ - Include numbers, symbols, special characters
69
+ - Do NOT add descriptions, interpretations, or context
70
+ - If no text is visible, return 'NO_TEXT_FOUND'
71
+
72
+ Extracted text:"""
73
+
74
+ else: # mode == "describe"
75
+ prompt = """Analyze this image and provide:
76
+ 1. All visible text exactly as written
77
+ 2. Brief description of the image content and context
78
+
79
+ Format:
80
+ TEXT: [all visible text here, or NO_TEXT_FOUND if none]
81
+ DESCRIPTION: [brief image description and context]"""
82
+
83
+ # Send request to Ollama
84
+ response = ollama.generate(
85
+ model=current_model,
86
+ prompt=prompt,
87
+ images=[img_base64],
88
+ options={
89
+ 'temperature': 0.1, # Low temperature for accuracy
90
+ 'top_p': 0.9,
91
+ 'num_predict': 1500
92
+ }
93
+ )
94
+
95
+ # Extract and clean response
96
+ extracted_content = response.get('response', '').strip()
97
+
98
+ # Handle no-text case for OCR mode
99
+ if mode == "ocr" and ('NO_TEXT_FOUND' in extracted_content or len(extracted_content) < 3):
100
+ return ""
101
+
102
+ return extracted_content
103
+
104
+ except Exception as e:
105
+ print(f"⚠️ Error extracting from image with Ollama {current_model}: {e}")
106
+ return ""
107
+
108
+ # Wrapper functions for each mode
109
+ def xtxt_image_ocr_only(file_buffer):
110
+ """Traditional OCR: extract only visible text using Ollama"""
111
+ return xtxt_image_ocr_ollama(file_buffer, mode="ocr")
112
+
113
+ def xtxt_image_describe(file_buffer):
114
+ """OCR + Description: text + image context using Ollama"""
115
+ return xtxt_image_ocr_ollama(file_buffer, mode="describe")
116
+
117
+ # Register OCR-only version as default
118
+ # Note: Will override traditional EasyOCR if both modules are loaded
119
+ image_formats = [
120
+ "image/jpeg", "image/jpg", "image/png",
121
+ "image/bmp", "image/tiff", "image/webp"
122
+ ]
123
+
124
+ for format_type in image_formats:
125
+ register_extractor(format_type, xtxt_image_ocr_only, name="OCR-Ollama")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.2.4
3
+ Version: 0.3
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -78,6 +78,9 @@ Requires-Dist: openai-whisper; extra == "audio"
78
78
  Provides-Extra: ocr
79
79
  Requires-Dist: easyocr; extra == "ocr"
80
80
  Requires-Dist: pillow; extra == "ocr"
81
+ Provides-Extra: ocr-ollama
82
+ Requires-Dist: ollama; extra == "ocr-ollama"
83
+ Requires-Dist: pillow; extra == "ocr-ollama"
81
84
  Provides-Extra: all
82
85
  Requires-Dist: textract; extra == "all"
83
86
  Requires-Dist: PyMuPDF; extra == "all"
@@ -96,6 +99,7 @@ Requires-Dist: pylatexenc; extra == "all"
96
99
  Requires-Dist: openai-whisper; extra == "all"
97
100
  Requires-Dist: easyocr; extra == "all"
98
101
  Requires-Dist: pillow; extra == "all"
102
+ Requires-Dist: ollama; extra == "all"
99
103
  Dynamic: license-file
100
104
 
101
105
  # PyxTxt
@@ -141,10 +145,13 @@ pip install pyxtxt[pdf,docx,presentation,spreadsheet,html,markdown,epub,email]
141
145
  # Audio transcription (~2GB download for Whisper models)
142
146
  pip install pyxtxt[audio]
143
147
 
144
- # OCR from images (~1GB download for EasyOCR models)
148
+ # Traditional OCR from images (~1GB download for EasyOCR models)
145
149
  pip install pyxtxt[ocr]
146
150
 
147
- # Both audio and OCR
151
+ # AI-powered OCR via Ollama (requires local Ollama + gemma3:4b model)
152
+ pip install pyxtxt[ocr-ollama]
153
+
154
+ # Both audio and traditional OCR
148
155
  pip install pyxtxt[audio,ocr]
149
156
  ```
150
157
 
@@ -185,6 +192,7 @@ Use python-magic-bin instead of python-magic for easier installation.
185
192
  - **LaTeX**: pylatexenc
186
193
  - **Audio**: openai-whisper (heavy ~2GB models)
187
194
  - **OCR**: easyocr, pillow (heavy ~1GB models)
195
+ - **OCR-Ollama**: ollama, pillow (requires local Ollama server)
188
196
 
189
197
  Dependencies are automatically installed based on selected optional groups.
190
198
 
@@ -249,11 +257,25 @@ text = xtxt(video_response.content)
249
257
  ```python
250
258
  from pyxtxt import xtxt
251
259
 
252
- # Extract text from images
260
+ # Traditional OCR with EasyOCR (install with: pip install pyxtxt[ocr])
253
261
  text = xtxt("scanned_document.png")
254
262
  text = xtxt("screenshot.jpg")
255
263
  text = xtxt("invoice.tiff")
256
264
 
265
+ # AI-powered OCR with Ollama (install with: pip install pyxtxt[ocr-ollama])
266
+ # Requires: ollama server running + gemma3:4b model
267
+ from pyxtxt.estrattori.ocr_ollama import set_ollama_model, xtxt_image_describe
268
+
269
+ # Configure model (optional, default is gemma3:4b)
270
+ set_ollama_model("gemma3:12b") # or llava:7b, llava:13b
271
+
272
+ # Extract only text (OCR mode)
273
+ text = xtxt("complex_document.png")
274
+
275
+ # Extract text + image description
276
+ description = xtxt_image_describe(open("diagram.png", "rb"))
277
+ # Output: "TEXT: Chart Title: Sales Report 2024\nDESCRIPTION: Bar chart showing quarterly sales data with blue bars"
278
+
257
279
  # From web images
258
280
  import requests
259
281
  image_response = requests.get("https://example.com/document.png")
@@ -21,6 +21,7 @@ src/pyxtxt/estrattori/html.py
21
21
  src/pyxtxt/estrattori/md.py
22
22
  src/pyxtxt/estrattori/msg.py
23
23
  src/pyxtxt/estrattori/ocr.py
24
+ src/pyxtxt/estrattori/ocr_ollama.py
24
25
  src/pyxtxt/estrattori/odt.py
25
26
  src/pyxtxt/estrattori/pdf.py
26
27
  src/pyxtxt/estrattori/pptx.py
@@ -23,6 +23,7 @@ pylatexenc
23
23
  openai-whisper
24
24
  easyocr
25
25
  pillow
26
+ ollama
26
27
 
27
28
  [audio]
28
29
  openai-whisper
@@ -55,6 +56,10 @@ beautifulsoup4
55
56
  easyocr
56
57
  pillow
57
58
 
59
+ [ocr-ollama]
60
+ ollama
61
+ pillow
62
+
58
63
  [odf]
59
64
  odfpy
60
65
 
@@ -1 +0,0 @@
1
- from .core import xtxt, extxt_available_formats, xtxt_from_url
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes