HowdenParser 0.1.13__tar.gz → 0.1.15__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- howdenparser-0.1.15/HowdenParser/__init__.py +3 -0
- howdenparser-0.1.15/HowdenParser/parameter/__init__.py +3 -0
- howdenparser-0.1.15/HowdenParser/parser.py +173 -0
- {howdenparser-0.1.13 → howdenparser-0.1.15}/PKG-INFO +1 -1
- {howdenparser-0.1.13 → howdenparser-0.1.15}/pyproject.toml +1 -1
- howdenparser-0.1.13/HowdenParser/__init__.py +0 -3
- howdenparser-0.1.13/HowdenParser/parameter/__init__.py +0 -15
- howdenparser-0.1.13/HowdenParser/parser.py +0 -159
- {howdenparser-0.1.13 → howdenparser-0.1.15}/HowdenParser/parameter/huggingface.py +0 -0
- {howdenparser-0.1.13 → howdenparser-0.1.15}/HowdenParser/parameter/llamaparser.py +0 -0
- {howdenparser-0.1.13 → howdenparser-0.1.15}/HowdenParser/parameter/mistralocr.py +0 -0
- {howdenparser-0.1.13 → howdenparser-0.1.15}/README.md +0 -0
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
from abc import ABC, abstractmethod
|
|
2
|
+
import os
|
|
3
|
+
import logging
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from PyPDF2 import PdfReader
|
|
6
|
+
import dotenv
|
|
7
|
+
|
|
8
|
+
dotenv.load_dotenv()
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class BaseParser(ABC):
|
|
12
|
+
_registry: dict[str, type["BaseParser"]] = {}
|
|
13
|
+
|
|
14
|
+
def __init_subclass__(cls, name: str | None = None, **kwargs):
|
|
15
|
+
"""Automatically register subclasses under a key."""
|
|
16
|
+
super().__init_subclass__(**kwargs)
|
|
17
|
+
key = name or cls.__name__.lower().replace("parser", "")
|
|
18
|
+
BaseParser._registry[key] = cls
|
|
19
|
+
BaseParser._registry.pop("", None)
|
|
20
|
+
|
|
21
|
+
@abstractmethod
|
|
22
|
+
def parse(self, text: str):
|
|
23
|
+
pass
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class Parser(BaseParser):
|
|
27
|
+
"""Factory + registry interface for all parsers."""
|
|
28
|
+
|
|
29
|
+
@classmethod
|
|
30
|
+
def available_parsers(cls) -> dict[str, list[str]]:
|
|
31
|
+
"""Return registered parsers and their init arguments."""
|
|
32
|
+
import inspect
|
|
33
|
+
result = {}
|
|
34
|
+
for name, parser_cls in BaseParser._registry.items():
|
|
35
|
+
sig = inspect.signature(parser_cls.__init__)
|
|
36
|
+
result[name] = [p for p in sig.parameters if p != "self"]
|
|
37
|
+
result.pop('', None)
|
|
38
|
+
return result
|
|
39
|
+
|
|
40
|
+
@classmethod
|
|
41
|
+
def create(cls, config: dict | None = None, **kwargs) -> BaseParser:
|
|
42
|
+
provider = kwargs["provider_and_model"].split(":")[0].lower()
|
|
43
|
+
model = kwargs["provider_and_model"].split(":")[1].lower()
|
|
44
|
+
if provider not in BaseParser._registry:
|
|
45
|
+
raise ValueError(f"Unknown parser '{provider}'. "
|
|
46
|
+
f"Available: {cls.available_parsers()}")
|
|
47
|
+
|
|
48
|
+
parser_cls = BaseParser._registry[provider]
|
|
49
|
+
import inspect
|
|
50
|
+
merged_args = {**(config or {}), **kwargs}
|
|
51
|
+
if "model" in inspect.signature(parser_cls.__init__).parameters:
|
|
52
|
+
merged_args["model"] = model
|
|
53
|
+
|
|
54
|
+
# Remove keys not in constructor
|
|
55
|
+
sig = inspect.signature(parser_cls.__init__)
|
|
56
|
+
valid_args = {k: v for k, v in merged_args.items() if k in sig.parameters and k != "self"}
|
|
57
|
+
|
|
58
|
+
return parser_cls(**valid_args)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@abstractmethod
|
|
64
|
+
def parse(self, text: str):
|
|
65
|
+
pass
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
# --- Parsers ---
|
|
69
|
+
class MistralOCRParser(BaseParser, name="mistralocr"):
|
|
70
|
+
def __init__(self,provider_and_model:str) -> None:
|
|
71
|
+
from mistralai import Mistral
|
|
72
|
+
|
|
73
|
+
self.model = provider_and_model.split(":")[1]
|
|
74
|
+
self.current_cost: float = 0.0
|
|
75
|
+
self.total_cost_euro: float = 0.0
|
|
76
|
+
|
|
77
|
+
api_key = os.getenv("MISTRAL-OCR-API-TOKEN")
|
|
78
|
+
if not api_key:
|
|
79
|
+
raise EnvironmentError("Missing MISTRAL-OCR-API-TOKEN in .env file.")
|
|
80
|
+
|
|
81
|
+
self.client = Mistral(api_key=api_key)
|
|
82
|
+
|
|
83
|
+
def parse(self, file_path: Path) -> str:
|
|
84
|
+
def upload_pdf(filename):
|
|
85
|
+
uploaded_pdf = self.client.files.upload(
|
|
86
|
+
file={"file_name": filename, "content": open(filename, "rb")},
|
|
87
|
+
purpose="ocr"
|
|
88
|
+
)
|
|
89
|
+
signed_url = self.client.files.get_signed_url(file_id=uploaded_pdf.id)
|
|
90
|
+
return signed_url.url
|
|
91
|
+
|
|
92
|
+
ocr_response = self.client.ocr.process(
|
|
93
|
+
model=self.model,
|
|
94
|
+
document={"type": "document_url", "document_url": upload_pdf(file_path)},
|
|
95
|
+
include_image_base64=True,
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
self.current_cost = 1 / 1000 * self._count_pages(file_path)
|
|
99
|
+
self.total_cost_euro += self.current_cost
|
|
100
|
+
|
|
101
|
+
return "\n".join(doc.markdown for doc in ocr_response.pages)
|
|
102
|
+
|
|
103
|
+
@staticmethod
|
|
104
|
+
def _count_pages(file_path: Path) -> int:
|
|
105
|
+
reader = PdfReader(str(file_path))
|
|
106
|
+
return len(reader.pages)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class LangChainParser(BaseParser, name="langchain"):
|
|
110
|
+
def __init__(self, model: str):
|
|
111
|
+
from langchain.llms import OpenAI
|
|
112
|
+
self.model_name = model
|
|
113
|
+
self.model = OpenAI(model_name=model)
|
|
114
|
+
|
|
115
|
+
def parse(self, text: str) -> dict:
|
|
116
|
+
response = self.model(text)
|
|
117
|
+
return {"source": "LangChain", "output": response}
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
class LlamaParser(BaseParser, name="llamaparser"):
|
|
121
|
+
def __init__(self, result_type: str, mode: bool) -> None:
|
|
122
|
+
logging.info("Initializing LlamaParser...")
|
|
123
|
+
|
|
124
|
+
from llama_parse import LlamaParse, ResultType
|
|
125
|
+
|
|
126
|
+
if result_type.lower() in ("md", "markdown"):
|
|
127
|
+
self.result_type = ResultType.MD
|
|
128
|
+
|
|
129
|
+
api_key = os.getenv("LLAMA-PARSER-API-TOKEN")
|
|
130
|
+
if not api_key:
|
|
131
|
+
raise EnvironmentError("Missing LLAMA-PARSER-API-TOKEN in .env file.")
|
|
132
|
+
|
|
133
|
+
self.parser = LlamaParse(api_key=api_key, result_type=self.result_type, premium_mode=mode)
|
|
134
|
+
|
|
135
|
+
def parse(self, file_path: Path) -> str:
|
|
136
|
+
documents = self.parser.load_data(str(file_path))
|
|
137
|
+
return "\n".join(doc.text for doc in documents)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
class HuggingFaceParser(BaseParser, name="huggingface"):
|
|
141
|
+
def __init__(self, model: str, result_type: str) -> None:
|
|
142
|
+
from transformers import pipeline
|
|
143
|
+
|
|
144
|
+
if result_type.lower() in ("md", "markdown"):
|
|
145
|
+
self.result_type = "markdown"
|
|
146
|
+
else:
|
|
147
|
+
self.result_type = "text"
|
|
148
|
+
|
|
149
|
+
api_key = os.getenv("HF-API-TOKEN")
|
|
150
|
+
if not api_key:
|
|
151
|
+
raise EnvironmentError("Missing HF-API-TOKEN in .env file.")
|
|
152
|
+
|
|
153
|
+
self.parser = pipeline(
|
|
154
|
+
task="document-question-answering",
|
|
155
|
+
model=model,
|
|
156
|
+
use_auth_token=api_key
|
|
157
|
+
)
|
|
158
|
+
|
|
159
|
+
def parse(self, file_path: Path) -> str:
|
|
160
|
+
import fitz
|
|
161
|
+
pdf_doc = fitz.open(file_path)
|
|
162
|
+
output_parts = []
|
|
163
|
+
|
|
164
|
+
for page in pdf_doc:
|
|
165
|
+
pix = page.get_pixmap(dpi=200)
|
|
166
|
+
img_bytes = pix.tobytes("png")
|
|
167
|
+
response = self.parser(img_bytes, question="Extract all text")
|
|
168
|
+
if response and "answer" in response[0]:
|
|
169
|
+
output_parts.append(response[0]["answer"])
|
|
170
|
+
|
|
171
|
+
return "\n\n".join(output_parts) if self.result_type == "markdown" else " ".join(output_parts)
|
|
172
|
+
|
|
173
|
+
|
|
@@ -14,7 +14,7 @@ build-backend = "poetry.core.masonry.api"
|
|
|
14
14
|
|
|
15
15
|
[tool.poetry]
|
|
16
16
|
name = "HowdenParser"
|
|
17
|
-
version = "0.1.
|
|
17
|
+
version = "0.1.15"
|
|
18
18
|
description = "A simple configuration manager with Pydantic and JSON export."
|
|
19
19
|
authors = [ "JesperThoftIllemannJ <jesper.jaeger@howdendanmark.dk>",]
|
|
20
20
|
readme = "README.md"
|
|
@@ -1,15 +0,0 @@
|
|
|
1
|
-
import os
|
|
2
|
-
import importlib
|
|
3
|
-
|
|
4
|
-
# Get all .py files in this folder except __init__.py
|
|
5
|
-
current_dir = os.path.dirname(__file__)
|
|
6
|
-
for filename in os.listdir(current_dir):
|
|
7
|
-
if filename.endswith(".py") and filename != "__init__.py":
|
|
8
|
-
module_name = filename[:-3] # remove .py
|
|
9
|
-
module = importlib.import_module(f".{module_name}", package=__name__)
|
|
10
|
-
|
|
11
|
-
# Add all classes from the module to the package namespace
|
|
12
|
-
for attr_name in dir(module):
|
|
13
|
-
attr = getattr(module, attr_name)
|
|
14
|
-
if isinstance(attr, type): # only classes
|
|
15
|
-
globals()[attr_name] = attr
|
|
@@ -1,159 +0,0 @@
|
|
|
1
|
-
from abc import ABC, abstractmethod
|
|
2
|
-
from dotenv import load_dotenv
|
|
3
|
-
import os
|
|
4
|
-
from pathlib import Path
|
|
5
|
-
import logging
|
|
6
|
-
from PyPDF2 import PdfReader
|
|
7
|
-
|
|
8
|
-
load_dotenv()
|
|
9
|
-
|
|
10
|
-
class BaseParser(ABC):
|
|
11
|
-
@abstractmethod
|
|
12
|
-
def parse(self, text: str) -> dict:
|
|
13
|
-
pass
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
class MistralOCRParser(BaseParser):
|
|
17
|
-
def __init__(self, model: str) -> None:
|
|
18
|
-
from mistralai import Mistral
|
|
19
|
-
self.model = model
|
|
20
|
-
self.current_cost: float = 0.0
|
|
21
|
-
self.total_cost_euro: float = 0.0
|
|
22
|
-
|
|
23
|
-
name = "MISTRAL-OCR-API-TOKEN"
|
|
24
|
-
api_key = os.getenv(name)
|
|
25
|
-
if not api_key:
|
|
26
|
-
raise EnvironmentError(f"Missing {name} in .env file.")
|
|
27
|
-
|
|
28
|
-
self.client = Mistral(api_key=api_key)
|
|
29
|
-
|
|
30
|
-
def parse(self, file_path: Path) -> str:
|
|
31
|
-
def upload_pdf(filename):
|
|
32
|
-
uploaded_pdf = self.client.files.upload(
|
|
33
|
-
file={
|
|
34
|
-
"file_name": filename,
|
|
35
|
-
"content": open(filename, "rb"),
|
|
36
|
-
},
|
|
37
|
-
purpose="ocr"
|
|
38
|
-
)
|
|
39
|
-
signed_url = self.client.files.get_signed_url(file_id=uploaded_pdf.id)
|
|
40
|
-
return signed_url.url
|
|
41
|
-
|
|
42
|
-
ocr_response = self.client.ocr.process(
|
|
43
|
-
model=self.model,
|
|
44
|
-
document={
|
|
45
|
-
"type": "document_url",
|
|
46
|
-
"document_url": upload_pdf(file_path),
|
|
47
|
-
},
|
|
48
|
-
include_image_base64=True,
|
|
49
|
-
)
|
|
50
|
-
|
|
51
|
-
self.current_cost = 1/1000 * self._count_pages(file_path)
|
|
52
|
-
self.total_cost_euro += self.current_cost
|
|
53
|
-
|
|
54
|
-
return "\n".join(doc.markdown for doc in ocr_response.pages)
|
|
55
|
-
|
|
56
|
-
@staticmethod
|
|
57
|
-
def _count_pages(file_path: Path) -> int:
|
|
58
|
-
reader = PdfReader(str(file_path))
|
|
59
|
-
return len(reader.pages)
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
class LangChainParser(BaseParser):
|
|
64
|
-
def __init__(self, model: str):
|
|
65
|
-
from langchain.llms import OpenAI
|
|
66
|
-
self.model_name = model
|
|
67
|
-
self.model = OpenAI(model_name=model)
|
|
68
|
-
|
|
69
|
-
def parse(self, text: str) -> dict:
|
|
70
|
-
response = self.model(text)
|
|
71
|
-
return {"source": "LangChain", "output": response}
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
class LlamaParser(BaseParser):
|
|
75
|
-
def __init__(self, result_type: str, mode: bool) -> None:
|
|
76
|
-
logging.info("Initializing LlamaParser...")
|
|
77
|
-
logging.info("Loading LlamaParse package...")
|
|
78
|
-
|
|
79
|
-
from llama_parse import LlamaParse, ResultType
|
|
80
|
-
|
|
81
|
-
if result_type.lower() == "md" or result_type.lower() == "markdown":
|
|
82
|
-
self.result_type = ResultType.MD
|
|
83
|
-
|
|
84
|
-
name = "LLAMA-PARSER-API-TOKEN"
|
|
85
|
-
api_key = os.getenv(name)
|
|
86
|
-
if not api_key:
|
|
87
|
-
raise EnvironmentError(f"Missing {name} in .env file.")
|
|
88
|
-
|
|
89
|
-
logging.info("Initializing LlamaParse parser...")
|
|
90
|
-
self.parser = LlamaParse(
|
|
91
|
-
api_key=api_key,
|
|
92
|
-
result_type=self.result_type,
|
|
93
|
-
premium_mode=mode
|
|
94
|
-
)
|
|
95
|
-
logging.info("LlamaParser initialized successfully.")
|
|
96
|
-
|
|
97
|
-
def parse(self, file_path: Path) -> str:
|
|
98
|
-
logging.info(f"Parsing file: {file_path}")
|
|
99
|
-
documents = self.parser.load_data(str(file_path))
|
|
100
|
-
text = "\n".join(doc.text for doc in documents)
|
|
101
|
-
logging.info(f"Parsing completed. Extracted {len(documents)} documents.")
|
|
102
|
-
return text
|
|
103
|
-
|
|
104
|
-
class HuggingFaceParser(BaseParser):
|
|
105
|
-
def __init__(self, model: str, result_type: str) -> None:
|
|
106
|
-
from transformers import pipeline
|
|
107
|
-
|
|
108
|
-
if result_type.lower() in ("md", "markdown"):
|
|
109
|
-
self.result_type = "markdown"
|
|
110
|
-
else:
|
|
111
|
-
self.result_type = "text"
|
|
112
|
-
|
|
113
|
-
name = "HF-API-TOKEN"
|
|
114
|
-
self.api_key = os.getenv(name)
|
|
115
|
-
if not self.api_key:
|
|
116
|
-
raise EnvironmentError(f"Missing {name} in .env file.")
|
|
117
|
-
|
|
118
|
-
# Example model: microsoft/layoutlmv3-base-finetuned-docvqa
|
|
119
|
-
self.parser = pipeline(
|
|
120
|
-
task="document-question-answering",
|
|
121
|
-
model=model,
|
|
122
|
-
use_auth_token=self.api_key
|
|
123
|
-
)
|
|
124
|
-
|
|
125
|
-
def parse(self, file_path: Path) -> str:
|
|
126
|
-
import fitz
|
|
127
|
-
|
|
128
|
-
pdf_doc = fitz.open(file_path)
|
|
129
|
-
output_parts = []
|
|
130
|
-
|
|
131
|
-
for page in pdf_doc:
|
|
132
|
-
pix = page.get_pixmap(dpi=200)
|
|
133
|
-
img_bytes = pix.tobytes("png")
|
|
134
|
-
response = self.parser(img_bytes, question="Extract all text")
|
|
135
|
-
if response and "answer" in response[0]:
|
|
136
|
-
output_parts.append(response[0]["answer"])
|
|
137
|
-
|
|
138
|
-
if self.result_type == "markdown":
|
|
139
|
-
return "\n\n".join(output_parts)
|
|
140
|
-
else:
|
|
141
|
-
return " ".join(output_parts)
|
|
142
|
-
|
|
143
|
-
# === Step 3: Dynamic factory using string input ===
|
|
144
|
-
class ParserFactory:
|
|
145
|
-
@staticmethod
|
|
146
|
-
def get_parser(provider_model: str, **kwargs) -> BaseParser:
|
|
147
|
-
provider = provider_model.partition(":")[0].lower()
|
|
148
|
-
model = provider_model.partition(":")[2]
|
|
149
|
-
if provider == "langchain":
|
|
150
|
-
return LangChainParser(model=model)
|
|
151
|
-
elif provider == "llamaparser":
|
|
152
|
-
return LlamaParser(kwargs["result_type"], kwargs["mode"])
|
|
153
|
-
elif provider == "huggingface":
|
|
154
|
-
return HuggingFaceParser(model=model, **kwargs)
|
|
155
|
-
elif provider == "mistralocr":
|
|
156
|
-
return MistralOCRParser(model)
|
|
157
|
-
else:
|
|
158
|
-
raise ValueError(f"Unknown parser type: {provider_model}")
|
|
159
|
-
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|