HowdenParser 2.0.2__tar.gz → 3.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {howdenparser-2.0.2 → howdenparser-3.0.0}/HowdenParser/parser.py +116 -41
- {howdenparser-2.0.2 → howdenparser-3.0.0}/PKG-INFO +1 -1
- {howdenparser-2.0.2 → howdenparser-3.0.0}/pyproject.toml +1 -1
- {howdenparser-2.0.2 → howdenparser-3.0.0}/HowdenParser/__init__.py +0 -0
- {howdenparser-2.0.2 → howdenparser-3.0.0}/HowdenParser/parameter/__init__.py +0 -0
- {howdenparser-2.0.2 → howdenparser-3.0.0}/HowdenParser/parameter/huggingface.py +0 -0
- {howdenparser-2.0.2 → howdenparser-3.0.0}/HowdenParser/parameter/llamaparser.py +0 -0
- {howdenparser-2.0.2 → howdenparser-3.0.0}/HowdenParser/parameter/mistralocr.py +0 -0
- {howdenparser-2.0.2 → howdenparser-3.0.0}/README.md +0 -0
|
@@ -4,8 +4,8 @@ import logging
|
|
|
4
4
|
from pathlib import Path
|
|
5
5
|
from PyPDF2 import PdfReader
|
|
6
6
|
import dotenv
|
|
7
|
-
import
|
|
8
|
-
from inspect import signature
|
|
7
|
+
from typing import overload, Literal
|
|
8
|
+
from inspect import signature, Signature
|
|
9
9
|
|
|
10
10
|
dotenv.load_dotenv()
|
|
11
11
|
logger = logging.getLogger(__name__)
|
|
@@ -29,6 +29,13 @@ class BaseParser(ABC):
|
|
|
29
29
|
class Parser(BaseParser):
|
|
30
30
|
"""Factory + registry interface for all parsers."""
|
|
31
31
|
|
|
32
|
+
def __new__(cls, config_or_dict: "Parameter | dict | str" = None, **kwargs) -> BaseParser:
|
|
33
|
+
"""
|
|
34
|
+
Factory entrypoint.
|
|
35
|
+
Allows calling Parser(...) directly to create the correct subclass.
|
|
36
|
+
"""
|
|
37
|
+
return cls._create(config_or_dict, **kwargs)
|
|
38
|
+
|
|
32
39
|
@classmethod
|
|
33
40
|
def available_parsers(cls) -> None:
|
|
34
41
|
"""Print registered parsers and their init arguments."""
|
|
@@ -37,53 +44,128 @@ class Parser(BaseParser):
|
|
|
37
44
|
for name, parser_cls in BaseParser._registry.items():
|
|
38
45
|
sig = inspect.signature(parser_cls.__init__)
|
|
39
46
|
result[name] = [p for p in sig.parameters if p != "self"]
|
|
40
|
-
result.pop(
|
|
47
|
+
result.pop("", None)
|
|
41
48
|
for key, values in result.items():
|
|
42
49
|
print(f"{key} with parameters: {values}")
|
|
43
50
|
|
|
51
|
+
# ---- Overloads for IDE autocomplete ----
|
|
52
|
+
@overload
|
|
53
|
+
@classmethod
|
|
54
|
+
def _create(
|
|
55
|
+
cls,
|
|
56
|
+
*,
|
|
57
|
+
provider_and_model: Literal["llamaparser:"],
|
|
58
|
+
result_type: str,
|
|
59
|
+
mode: bool,
|
|
60
|
+
) -> "LlamaParser": ...
|
|
61
|
+
|
|
62
|
+
@overload
|
|
63
|
+
@classmethod
|
|
64
|
+
def _create(
|
|
65
|
+
cls,
|
|
66
|
+
*,
|
|
67
|
+
provider_and_model: Literal["mistralocr:ocr-large", "mistralocr:ocr-small"],
|
|
68
|
+
) -> "MistralOCRParser": ...
|
|
69
|
+
|
|
70
|
+
@overload
|
|
71
|
+
@classmethod
|
|
72
|
+
def _create(
|
|
73
|
+
cls,
|
|
74
|
+
*,
|
|
75
|
+
provider_and_model: Literal["langchain:gpt-3.5-turbo", "langchain:gpt-4"],
|
|
76
|
+
) -> "LangChainParser": ...
|
|
77
|
+
|
|
78
|
+
@overload
|
|
79
|
+
@classmethod
|
|
80
|
+
def _create(
|
|
81
|
+
cls,
|
|
82
|
+
*,
|
|
83
|
+
provider_and_model: Literal[
|
|
84
|
+
"huggingface:microsoft/trocr-base-handwritten",
|
|
85
|
+
"huggingface:microsoft/trocr-large-printed",
|
|
86
|
+
],
|
|
87
|
+
) -> "HuggingFaceParser": ...
|
|
88
|
+
|
|
89
|
+
# ---- Implementation ----
|
|
44
90
|
@classmethod
|
|
45
|
-
def
|
|
91
|
+
def _create(cls, config_or_dict: "Parameter | dict | str" = None, **kwargs) -> BaseParser:
|
|
46
92
|
"""
|
|
47
|
-
Dynamically create parser instances
|
|
48
|
-
|
|
93
|
+
Dynamically create parser instances.
|
|
94
|
+
Supports: Parameter, dict, str (provider_and_model), or kwargs.
|
|
49
95
|
"""
|
|
50
|
-
|
|
51
|
-
if
|
|
96
|
+
|
|
97
|
+
if config_or_dict is None:
|
|
98
|
+
config_dict = {}
|
|
99
|
+
elif hasattr(config_or_dict, "model_dump"):
|
|
52
100
|
config_dict = config_or_dict.model_dump()
|
|
53
101
|
elif isinstance(config_or_dict, dict):
|
|
54
102
|
config_dict = config_or_dict
|
|
103
|
+
elif isinstance(config_or_dict, str): # shorthand
|
|
104
|
+
config_dict = {"provider_and_model": config_or_dict}
|
|
55
105
|
else:
|
|
56
|
-
raise TypeError("Expected Parameter instance or
|
|
57
|
-
|
|
58
|
-
if "provider_and_model" not in kwargs and "provider_and_model" not in config_dict:
|
|
59
|
-
raise ValueError("provider_and_model must be specified")
|
|
106
|
+
raise TypeError("Expected Parameter instance, dict, str, or None for config_or_dict")
|
|
60
107
|
|
|
61
|
-
# Merge dict + kwargs (kwargs take precedence)
|
|
62
108
|
merged_args = {**config_dict, **kwargs}
|
|
63
109
|
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
110
|
+
if "provider_and_model" not in merged_args:
|
|
111
|
+
raise ValueError("provider_and_model must be specified, e.g. 'llamaparser:'")
|
|
112
|
+
|
|
113
|
+
provider_and_model = str(merged_args.get("provider_and_model")).strip()
|
|
114
|
+
|
|
115
|
+
if ":" not in provider_and_model:
|
|
116
|
+
raise ValueError("provider_and_model must include a colon, e.g. 'llamaparser:'")
|
|
117
|
+
|
|
118
|
+
provider, model = provider_and_model.split(":", 1)
|
|
119
|
+
provider = (provider or "").strip().lower()
|
|
120
|
+
model = (model or "").strip().lower()
|
|
121
|
+
|
|
122
|
+
if not provider:
|
|
123
|
+
raise ValueError(f"Invalid provider_and_model '{provider_and_model}': provider part is empty.")
|
|
69
124
|
|
|
70
125
|
if provider not in BaseParser._registry:
|
|
71
126
|
raise ValueError(f"Unknown parser '{provider}'. Available: {list(BaseParser._registry)}")
|
|
72
127
|
|
|
73
128
|
parser_cls = BaseParser._registry[provider]
|
|
74
129
|
|
|
75
|
-
#
|
|
130
|
+
# --- Filter valid constructor args ---
|
|
76
131
|
sig = signature(parser_cls.__init__)
|
|
77
132
|
valid_args = {k: v for k, v in merged_args.items() if k in sig.parameters and k != "self"}
|
|
78
133
|
|
|
79
|
-
# Auto-
|
|
134
|
+
# Auto-fill common args
|
|
80
135
|
if "model" in sig.parameters and "model" not in valid_args:
|
|
81
136
|
valid_args["model"] = model
|
|
82
|
-
|
|
83
|
-
# Auto-insert 'provider_and_model' if required
|
|
84
137
|
if "provider_and_model" in sig.parameters and "provider_and_model" not in valid_args:
|
|
85
138
|
valid_args["provider_and_model"] = provider_and_model
|
|
86
139
|
|
|
140
|
+
# --- Check required args ---
|
|
141
|
+
required_params = [
|
|
142
|
+
p.name
|
|
143
|
+
for p in sig.parameters.values()
|
|
144
|
+
if p.name != "self"
|
|
145
|
+
and p.default is Signature.empty
|
|
146
|
+
and p.kind in (p.POSITIONAL_OR_KEYWORD, p.KEYWORD_ONLY)
|
|
147
|
+
]
|
|
148
|
+
missing = [name for name in required_params if name not in valid_args]
|
|
149
|
+
|
|
150
|
+
# Special rule: for llamaparser we allow empty `model`
|
|
151
|
+
if provider != "llamaparser" and not model:
|
|
152
|
+
missing.append("model")
|
|
153
|
+
|
|
154
|
+
if missing:
|
|
155
|
+
example = ""
|
|
156
|
+
if provider == "llamaparser":
|
|
157
|
+
example = "Example: Parser('llamaparser:', result_type='md', mode=True)"
|
|
158
|
+
elif provider == "mistralocr":
|
|
159
|
+
example = "Example: Parser('mistralocr:ocr-large')"
|
|
160
|
+
elif provider == "huggingface":
|
|
161
|
+
example = "Example: Parser('huggingface:microsoft/trocr-base-handwritten')"
|
|
162
|
+
elif provider == "langchain":
|
|
163
|
+
example = "Example: Parser('langchain:gpt-3.5-turbo')"
|
|
164
|
+
|
|
165
|
+
raise TypeError(
|
|
166
|
+
f"{parser_cls.__name__} cannot be created because it is missing required argument(s): {', '.join(missing)}. {example}"
|
|
167
|
+
)
|
|
168
|
+
|
|
87
169
|
return parser_cls(**valid_args)
|
|
88
170
|
|
|
89
171
|
@abstractmethod
|
|
@@ -93,24 +175,21 @@ class Parser(BaseParser):
|
|
|
93
175
|
|
|
94
176
|
# --- Parsers ---
|
|
95
177
|
class MistralOCRParser(BaseParser, name="mistralocr"):
|
|
96
|
-
def __init__(self,provider_and_model:str) -> None:
|
|
178
|
+
def __init__(self, provider_and_model: str) -> None:
|
|
97
179
|
from mistralai import Mistral
|
|
98
|
-
|
|
99
180
|
self.model = provider_and_model.split(":")[1]
|
|
100
181
|
self.current_cost: float = 0.0
|
|
101
182
|
self.total_cost_euro: float = 0.0
|
|
102
|
-
|
|
103
183
|
api_key = os.getenv("MISTRAL-OCR-API-TOKEN")
|
|
104
184
|
if not api_key:
|
|
105
185
|
raise EnvironmentError("Missing MISTRAL-OCR-API-TOKEN in .env file.")
|
|
106
|
-
|
|
107
186
|
self.client = Mistral(api_key=api_key)
|
|
108
187
|
|
|
109
188
|
def parse(self, file_path: Path) -> str:
|
|
110
189
|
def upload_pdf(filename):
|
|
111
190
|
uploaded_pdf = self.client.files.upload(
|
|
112
191
|
file={"file_name": filename, "content": open(filename, "rb")},
|
|
113
|
-
purpose="ocr"
|
|
192
|
+
purpose="ocr",
|
|
114
193
|
)
|
|
115
194
|
signed_url = self.client.files.get_signed_url(file_id=uploaded_pdf.id)
|
|
116
195
|
return signed_url.url
|
|
@@ -120,12 +199,13 @@ class MistralOCRParser(BaseParser, name="mistralocr"):
|
|
|
120
199
|
document={"type": "document_url", "document_url": upload_pdf(file_path)},
|
|
121
200
|
include_image_base64=True,
|
|
122
201
|
)
|
|
123
|
-
|
|
124
202
|
self.current_cost = 1 / 1000 * self._count_pages(file_path)
|
|
125
203
|
self.total_cost_euro += self.current_cost
|
|
126
|
-
|
|
127
204
|
return "\n".join(doc.markdown for doc in ocr_response.pages)
|
|
128
205
|
|
|
206
|
+
def __call__(self, file_path: Path) -> str:
|
|
207
|
+
return self.parse(file_path)
|
|
208
|
+
|
|
129
209
|
@staticmethod
|
|
130
210
|
def _count_pages(file_path: Path) -> int:
|
|
131
211
|
reader = PdfReader(str(file_path))
|
|
@@ -142,45 +222,43 @@ class LangChainParser(BaseParser, name="langchain"):
|
|
|
142
222
|
response = self.model(text)
|
|
143
223
|
return {"source": "LangChain", "output": response}
|
|
144
224
|
|
|
225
|
+
def __call__(self, file_path: Path) -> str:
|
|
226
|
+
return self.parse(file_path)
|
|
227
|
+
|
|
145
228
|
|
|
146
229
|
class LlamaParser(BaseParser, name="llamaparser"):
|
|
147
230
|
def __init__(self, result_type: str, mode: bool, provider_and_model) -> None:
|
|
148
231
|
logging.info("Initializing LlamaParser...")
|
|
149
|
-
|
|
150
232
|
self.result_type = result_type
|
|
151
233
|
self.mode = mode
|
|
152
234
|
self.provider_and_model = provider_and_model
|
|
153
235
|
|
|
154
236
|
from llama_parse import LlamaParse, ResultType
|
|
155
|
-
|
|
156
237
|
if result_type.lower() in ("md", "markdown"):
|
|
157
238
|
result_type = ResultType.MD
|
|
158
|
-
|
|
159
239
|
api_key = os.getenv("LLAMA-PARSER-API-TOKEN")
|
|
160
240
|
if not api_key:
|
|
161
241
|
raise EnvironmentError("Missing LLAMA-PARSER-API-TOKEN in .env file.")
|
|
162
|
-
|
|
163
242
|
self._parser = LlamaParse(api_key=api_key, result_type=result_type, premium_mode=self.mode)
|
|
164
243
|
|
|
165
244
|
def parse(self, file_path: Path) -> str:
|
|
166
245
|
documents = self._parser.load_data(str(file_path))
|
|
167
246
|
return "\n".join(doc.text for doc in documents)
|
|
168
247
|
|
|
248
|
+
def __call__(self, file_path: Path) -> str:
|
|
249
|
+
return self.parse(file_path)
|
|
250
|
+
|
|
169
251
|
|
|
170
252
|
class HuggingFaceParser(BaseParser, name="huggingface"):
|
|
171
253
|
def __init__(self, provider_and_model: str) -> None:
|
|
172
254
|
from transformers import TrOCRProcessor, VisionEncoderDecoderModel
|
|
173
|
-
|
|
174
255
|
api_key = os.getenv("HF-API-TOKEN")
|
|
175
256
|
if not api_key:
|
|
176
257
|
raise EnvironmentError("Missing HF-API-TOKEN in .env file.")
|
|
177
|
-
|
|
178
258
|
model_name = provider_and_model.split(":")[1]
|
|
179
259
|
logger.info(f"Loading Hugging Face OCR model: {model_name}")
|
|
180
|
-
|
|
181
260
|
self.processor = TrOCRProcessor.from_pretrained(model_name, token=api_key)
|
|
182
261
|
self.model = VisionEncoderDecoderModel.from_pretrained(model_name, token=api_key)
|
|
183
|
-
|
|
184
262
|
logger.info("Model and processor loaded successfully.")
|
|
185
263
|
|
|
186
264
|
def parse(self, file_path: Path) -> str:
|
|
@@ -196,9 +274,6 @@ class HuggingFaceParser(BaseParser, name="huggingface"):
|
|
|
196
274
|
generated_ids = self.model.generate(pixel_values)
|
|
197
275
|
text = self.processor.batch_decode(generated_ids, skip_special_tokens=True)[0]
|
|
198
276
|
all_text += text + "\n"
|
|
199
|
-
logger.debug(f"OCR text (page {i}): {text[:100]}...")
|
|
200
|
-
|
|
277
|
+
logger.debug(f"OCR text (page {i}): {text[:100]}...")
|
|
201
278
|
logger.info("OCR completed for all pages.")
|
|
202
279
|
return all_text
|
|
203
|
-
|
|
204
|
-
|
|
@@ -14,7 +14,7 @@ build-backend = "poetry.core.masonry.api"
|
|
|
14
14
|
|
|
15
15
|
[tool.poetry]
|
|
16
16
|
name = "HowdenParser"
|
|
17
|
-
version = "
|
|
17
|
+
version = "3.0.0"
|
|
18
18
|
description = "A simple configuration manager with Pydantic and JSON export."
|
|
19
19
|
authors = [ "JesperThoftIllemannJ <jesper.jaeger@howdendanmark.dk>",]
|
|
20
20
|
readme = "README.md"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|