pyxtxt 0.3.2__tar.gz → 0.3.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. {pyxtxt-0.3.2/src/pyxtxt.egg-info → pyxtxt-0.3.3}/PKG-INFO +1 -1
  2. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/pyproject.toml +1 -1
  3. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/ocr_ollama.py +61 -18
  4. {pyxtxt-0.3.2 → pyxtxt-0.3.3/src/pyxtxt.egg-info}/PKG-INFO +1 -1
  5. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/LICENSE +0 -0
  6. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/MANIFEST.in +0 -0
  7. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/README.md +0 -0
  8. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/setup.cfg +0 -0
  9. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/__init__.py +0 -0
  10. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/core.py +0 -0
  11. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/__init__.py +0 -0
  12. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/audio.py +0 -0
  13. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/doc.py +0 -0
  14. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/docx.py +0 -0
  15. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/eml.py +0 -0
  16. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/epub.py +0 -0
  17. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/html.py +0 -0
  18. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/md.py +0 -0
  19. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/msg.py +0 -0
  20. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/ocr.py +0 -0
  21. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/odt.py +0 -0
  22. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/pdf.py +0 -0
  23. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/pptx.py +0 -0
  24. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/rtf.py +0 -0
  25. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/svg.py +0 -0
  26. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/tex.py +0 -0
  27. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/txt.py +0 -0
  28. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/xls.py +0 -0
  29. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/xlsx.py +0 -0
  30. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/estrattori/xml.py +0 -0
  31. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/examples.py +0 -0
  32. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt/pyxtxt.py +0 -0
  33. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/SOURCES.txt +0 -0
  34. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/dependency_links.txt +0 -0
  35. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/requires.txt +0 -0
  36. {pyxtxt-0.3.2 → pyxtxt-0.3.3}/src/pyxtxt.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.2
3
+ Version: 0.3.3
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "pyxtxt"
3
- version = "0.3.2"
3
+ version = "0.3.3"
4
4
  description = "A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.)."
5
5
  classifiers = [
6
6
  "Development Status :: 4 - Beta",
@@ -18,7 +18,8 @@ OLLAMA_CONFIG = {
18
18
  'style': 'descriptive', # descriptive, technical, simple, detailed
19
19
  'temperature': 0.1, # Response creativity (0.0-1.0)
20
20
  'max_tokens': 1500, # Maximum response length
21
- 'confidence_threshold': 0.7 # Minimum confidence for text extraction
21
+ 'confidence_threshold': 0.7, # Minimum confidence for text extraction
22
+ 'context': 'general' # Context hint: general, cookbook, document, diagram, etc.
22
23
  }
23
24
 
24
25
  def set_ollama_model(model_name: str):
@@ -48,13 +49,15 @@ def set_ollama_config(**kwargs):
48
49
  - language: Language hint ('auto', 'italian', 'english', 'spanish', etc.)
49
50
  - caption_length: Caption length ('short', 'medium', 'long')
50
51
  - style: Caption style ('descriptive', 'technical', 'simple', 'detailed')
52
+ - context: Content context ('general', 'cookbook', 'document', 'handwriting', 'technical')
51
53
  - temperature: Response creativity 0.0-1.0 (default: 0.1)
52
54
  - max_tokens: Maximum response length (default: 1500)
53
55
  - confidence_threshold: Text extraction confidence 0.0-1.0 (default: 0.7)
54
56
 
55
57
  Examples:
56
58
  set_ollama_config(language='italian', style='detailed')
57
- set_ollama_config(caption_length='long', temperature=0.3)
59
+ set_ollama_config(context='document', caption_length='long')
60
+ set_ollama_config(context='handwriting', temperature=0.2)
58
61
  """
59
62
  global OLLAMA_CONFIG
60
63
  for key, value in kwargs.items():
@@ -77,7 +80,8 @@ def reset_ollama_config():
77
80
  'style': 'descriptive',
78
81
  'temperature': 0.1,
79
82
  'max_tokens': 1500,
80
- 'confidence_threshold': 0.7
83
+ 'confidence_threshold': 0.7,
84
+ 'context': 'general'
81
85
  }
82
86
  print("✅ Ollama configuration reset to defaults")
83
87
 
@@ -96,7 +100,11 @@ if ollama and Image:
96
100
  current_model = model or OLLAMA_MODEL
97
101
 
98
102
  # Convert buffer to PIL Image
99
- image = Image.open(BytesIO(file_buffer.read()))
103
+ # Reset buffer position if it has read method
104
+ if hasattr(file_buffer, 'seek'):
105
+ file_buffer.seek(0)
106
+ image_data = file_buffer.read()
107
+ image = Image.open(BytesIO(image_data))
100
108
 
101
109
  # Convert to RGB if needed
102
110
  if image.mode != 'RGB':
@@ -117,14 +125,18 @@ if ollama and Image:
117
125
 
118
126
  # Different prompts based on mode
119
127
  if mode == "ocr":
120
- prompt = f"""Extract ALL visible text from this image exactly as it appears.
121
- Rules:
122
- - Only return text that is actually written/printed in the image
123
- - Preserve reading order (left to right, top to bottom)
124
- - Maintain line breaks and formatting
125
- - Include numbers, symbols, special characters
126
- - Do NOT add descriptions, interpretations, or context
127
- - {lang_hint}If no text is visible, return 'NO_TEXT_FOUND'
128
+ prompt = f"""Look carefully at this image and extract ALL text that you can see, including:
129
+ - Titles, headings, and main text content
130
+ - Small print, captions, labels, and annotations
131
+ - Numbers, measurements, quantities, and symbols
132
+ - Menu items, ingredient lists, cooking instructions
133
+ - Any text in boxes, speech bubbles, or decorative elements
134
+
135
+ IMPORTANT:
136
+ - Read carefully and include even small or partially visible text
137
+ - Preserve the original formatting and line breaks where possible
138
+ - {lang_hint}Process the text from left to right, top to bottom
139
+ - If you cannot find any readable text at all, respond with 'NO_TEXT_FOUND'
128
140
 
129
141
  Extracted text:"""
130
142
 
@@ -146,12 +158,33 @@ Extracted text:"""
146
158
  style_instruction = style_prompts.get(config['style'], style_prompts['descriptive'])
147
159
  length_instruction = length_hints.get(config['caption_length'], length_hints['medium'])
148
160
 
161
+ # Context-specific hints (only when explicitly set)
162
+ context_hint = ""
163
+ context = config.get('context', 'general').lower()
164
+ if context == 'cookbook' or context == 'recipe':
165
+ context_hint = """
166
+ - If this appears to be a recipe/cookbook page, include: ingredients, cooking steps, quantities, cooking times
167
+ - Mention any photos of prepared dishes or cooking techniques shown
168
+ - Note any special formatting like ingredient lists, step numbers, or cooking tips"""
169
+ elif context == 'document':
170
+ context_hint = """
171
+ - Focus on document structure: headers, paragraphs, sections, page numbers
172
+ - Note any official formatting, letterheads, signatures, or stamps"""
173
+ elif context == 'handwriting' or context == 'notes':
174
+ context_hint = """
175
+ - Pay special attention to handwritten text which may be harder to read
176
+ - Note any sketches, diagrams, or informal formatting typical of personal notes"""
177
+ elif context == 'technical' or context == 'diagram':
178
+ context_hint = """
179
+ - Focus on technical elements: labels, measurements, specifications, diagrams
180
+ - Include any mathematical formulas, technical symbols, or engineering notations"""
181
+
149
182
  prompt = f"""Analyze this image and provide:
150
- 1. All visible text exactly as written
183
+ 1. All visible text exactly as written (preserve formatting, line breaks, bullet points)
151
184
  2. Image description following these guidelines:
152
185
  - {style_instruction}
153
186
  - {length_instruction}
154
- - {lang_hint}Focus on key visual elements, layout, and context
187
+ - {lang_hint}Focus on key visual elements, layout, and context{context_hint}
155
188
 
156
189
  Format:
157
190
  TEXT: [all visible text here, or NO_TEXT_FOUND if none]
@@ -183,13 +216,23 @@ DESCRIPTION: [image description following the guidelines above]"""
183
216
  return ""
184
217
 
185
218
  # Wrapper functions for each mode
186
- def xtxt_image_ocr_only(file_buffer):
219
+ def xtxt_image_ocr_only(file_input):
187
220
  """Traditional OCR: extract only visible text using Ollama"""
188
- return xtxt_image_ocr_ollama(file_buffer, mode="ocr")
221
+ # Handle both file paths and buffers
222
+ if isinstance(file_input, str):
223
+ with open(file_input, 'rb') as f:
224
+ return xtxt_image_ocr_ollama(f, mode="ocr")
225
+ else:
226
+ return xtxt_image_ocr_ollama(file_input, mode="ocr")
189
227
 
190
- def xtxt_image_describe(file_buffer):
228
+ def xtxt_image_describe(file_input):
191
229
  """OCR + Description: text + image context using Ollama"""
192
- return xtxt_image_ocr_ollama(file_buffer, mode="describe")
230
+ # Handle both file paths and buffers
231
+ if isinstance(file_input, str):
232
+ with open(file_input, 'rb') as f:
233
+ return xtxt_image_ocr_ollama(f, mode="describe")
234
+ else:
235
+ return xtxt_image_ocr_ollama(file_input, mode="describe")
193
236
 
194
237
  # Register OCR-only version as default
195
238
  # Note: Will override traditional EasyOCR if both modules are loaded
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyxtxt
3
- Version: 0.3.2
3
+ Version: 0.3.3
4
4
  Summary: A Python library for extracting text from different types of files (PDF, DOCX, PPTX, XLSX, ODT, ecc.).
5
5
  Author-email: Giuseppe Levi <giuseppe.levi@gmail.com>
6
6
  License: MIT License
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes