HowdenParser 0.1.11__tar.gz → 0.1.13__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- howdenparser-0.1.13/HowdenParser/parameter/mistralocr.py +5 -0
- {howdenparser-0.1.11 → howdenparser-0.1.13}/HowdenParser/parser.py +46 -24
- {howdenparser-0.1.11 → howdenparser-0.1.13}/PKG-INFO +3 -1
- {howdenparser-0.1.11 → howdenparser-0.1.13}/pyproject.toml +2 -2
- howdenparser-0.1.11/HowdenParser/parameter/mistralocr.py +0 -7
- {howdenparser-0.1.11 → howdenparser-0.1.13}/HowdenParser/__init__.py +0 -0
- {howdenparser-0.1.11 → howdenparser-0.1.13}/HowdenParser/parameter/__init__.py +0 -0
- {howdenparser-0.1.11 → howdenparser-0.1.13}/HowdenParser/parameter/huggingface.py +0 -0
- {howdenparser-0.1.11 → howdenparser-0.1.13}/HowdenParser/parameter/llamaparser.py +0 -0
- {howdenparser-0.1.11 → howdenparser-0.1.13}/README.md +0 -0
|
@@ -2,6 +2,8 @@ from abc import ABC, abstractmethod
|
|
|
2
2
|
from dotenv import load_dotenv
|
|
3
3
|
import os
|
|
4
4
|
from pathlib import Path
|
|
5
|
+
import logging
|
|
6
|
+
from PyPDF2 import PdfReader
|
|
5
7
|
|
|
6
8
|
load_dotenv()
|
|
7
9
|
|
|
@@ -12,14 +14,17 @@ class BaseParser(ABC):
|
|
|
12
14
|
|
|
13
15
|
|
|
14
16
|
class MistralOCRParser(BaseParser):
|
|
15
|
-
def __init__(self,
|
|
17
|
+
def __init__(self, model: str) -> None:
|
|
16
18
|
from mistralai import Mistral
|
|
19
|
+
self.model = model
|
|
20
|
+
self.current_cost: float = 0.0
|
|
21
|
+
self.total_cost_euro: float = 0.0
|
|
22
|
+
|
|
17
23
|
name = "MISTRAL-OCR-API-TOKEN"
|
|
18
24
|
api_key = os.getenv(name)
|
|
19
25
|
if not api_key:
|
|
20
26
|
raise EnvironmentError(f"Missing {name} in .env file.")
|
|
21
27
|
|
|
22
|
-
self.result_type = "md" if result_type.lower() in ("md", "markdown") else "text"
|
|
23
28
|
self.client = Mistral(api_key=api_key)
|
|
24
29
|
|
|
25
30
|
def parse(self, file_path: Path) -> str:
|
|
@@ -35,7 +40,7 @@ class MistralOCRParser(BaseParser):
|
|
|
35
40
|
return signed_url.url
|
|
36
41
|
|
|
37
42
|
ocr_response = self.client.ocr.process(
|
|
38
|
-
model=
|
|
43
|
+
model=self.model,
|
|
39
44
|
document={
|
|
40
45
|
"type": "document_url",
|
|
41
46
|
"document_url": upload_pdf(file_path),
|
|
@@ -43,14 +48,23 @@ class MistralOCRParser(BaseParser):
|
|
|
43
48
|
include_image_base64=True,
|
|
44
49
|
)
|
|
45
50
|
|
|
51
|
+
self.current_cost = 1/1000 * self._count_pages(file_path)
|
|
52
|
+
self.total_cost_euro += self.current_cost
|
|
53
|
+
|
|
46
54
|
return "\n".join(doc.markdown for doc in ocr_response.pages)
|
|
47
55
|
|
|
56
|
+
@staticmethod
|
|
57
|
+
def _count_pages(file_path: Path) -> int:
|
|
58
|
+
reader = PdfReader(str(file_path))
|
|
59
|
+
return len(reader.pages)
|
|
60
|
+
|
|
61
|
+
|
|
48
62
|
|
|
49
63
|
class LangChainParser(BaseParser):
|
|
50
|
-
def __init__(self,
|
|
64
|
+
def __init__(self, model: str):
|
|
51
65
|
from langchain.llms import OpenAI
|
|
52
|
-
self.model_name =
|
|
53
|
-
self.model = OpenAI(model_name=
|
|
66
|
+
self.model_name = model
|
|
67
|
+
self.model = OpenAI(model_name=model)
|
|
54
68
|
|
|
55
69
|
def parse(self, text: str) -> dict:
|
|
56
70
|
response = self.model(text)
|
|
@@ -58,27 +72,37 @@ class LangChainParser(BaseParser):
|
|
|
58
72
|
|
|
59
73
|
|
|
60
74
|
class LlamaParser(BaseParser):
|
|
61
|
-
def __init__(self, result_type: str, mode: bool
|
|
75
|
+
def __init__(self, result_type: str, mode: bool) -> None:
|
|
76
|
+
logging.info("Initializing LlamaParser...")
|
|
77
|
+
logging.info("Loading LlamaParse package...")
|
|
78
|
+
|
|
62
79
|
from llama_parse import LlamaParse, ResultType
|
|
80
|
+
|
|
63
81
|
if result_type.lower() == "md" or result_type.lower() == "markdown":
|
|
64
82
|
self.result_type = ResultType.MD
|
|
83
|
+
|
|
65
84
|
name = "LLAMA-PARSER-API-TOKEN"
|
|
66
|
-
|
|
67
|
-
if not
|
|
85
|
+
api_key = os.getenv(name)
|
|
86
|
+
if not api_key:
|
|
68
87
|
raise EnvironmentError(f"Missing {name} in .env file.")
|
|
69
88
|
|
|
89
|
+
logging.info("Initializing LlamaParse parser...")
|
|
70
90
|
self.parser = LlamaParse(
|
|
71
|
-
api_key=
|
|
91
|
+
api_key=api_key,
|
|
72
92
|
result_type=self.result_type,
|
|
73
93
|
premium_mode=mode
|
|
74
94
|
)
|
|
95
|
+
logging.info("LlamaParser initialized successfully.")
|
|
75
96
|
|
|
76
97
|
def parse(self, file_path: Path) -> str:
|
|
98
|
+
logging.info(f"Parsing file: {file_path}")
|
|
77
99
|
documents = self.parser.load_data(str(file_path))
|
|
78
|
-
|
|
100
|
+
text = "\n".join(doc.text for doc in documents)
|
|
101
|
+
logging.info(f"Parsing completed. Extracted {len(documents)} documents.")
|
|
102
|
+
return text
|
|
79
103
|
|
|
80
104
|
class HuggingFaceParser(BaseParser):
|
|
81
|
-
def __init__(self,
|
|
105
|
+
def __init__(self, model: str, result_type: str) -> None:
|
|
82
106
|
from transformers import pipeline
|
|
83
107
|
|
|
84
108
|
if result_type.lower() in ("md", "markdown"):
|
|
@@ -91,16 +115,15 @@ class HuggingFaceParser(BaseParser):
|
|
|
91
115
|
if not self.api_key:
|
|
92
116
|
raise EnvironmentError(f"Missing {name} in .env file.")
|
|
93
117
|
|
|
94
|
-
# OCR + text extraction pipeline
|
|
95
118
|
# Example model: microsoft/layoutlmv3-base-finetuned-docvqa
|
|
96
119
|
self.parser = pipeline(
|
|
97
120
|
task="document-question-answering",
|
|
98
|
-
model=
|
|
121
|
+
model=model,
|
|
99
122
|
use_auth_token=self.api_key
|
|
100
123
|
)
|
|
101
124
|
|
|
102
125
|
def parse(self, file_path: Path) -> str:
|
|
103
|
-
import fitz
|
|
126
|
+
import fitz
|
|
104
127
|
|
|
105
128
|
pdf_doc = fitz.open(file_path)
|
|
106
129
|
output_parts = []
|
|
@@ -113,7 +136,6 @@ class HuggingFaceParser(BaseParser):
|
|
|
113
136
|
output_parts.append(response[0]["answer"])
|
|
114
137
|
|
|
115
138
|
if self.result_type == "markdown":
|
|
116
|
-
# Here you could add rules to format into markdown
|
|
117
139
|
return "\n\n".join(output_parts)
|
|
118
140
|
else:
|
|
119
141
|
return " ".join(output_parts)
|
|
@@ -121,17 +143,17 @@ class HuggingFaceParser(BaseParser):
|
|
|
121
143
|
# === Step 3: Dynamic factory using string input ===
|
|
122
144
|
class ParserFactory:
|
|
123
145
|
@staticmethod
|
|
124
|
-
def get_parser(
|
|
125
|
-
provider =
|
|
126
|
-
model =
|
|
146
|
+
def get_parser(provider_model: str, **kwargs) -> BaseParser:
|
|
147
|
+
provider = provider_model.partition(":")[0].lower()
|
|
148
|
+
model = provider_model.partition(":")[2]
|
|
127
149
|
if provider == "langchain":
|
|
128
|
-
return LangChainParser(
|
|
150
|
+
return LangChainParser(model=model)
|
|
129
151
|
elif provider == "llamaparser":
|
|
130
|
-
return LlamaParser(
|
|
152
|
+
return LlamaParser(kwargs["result_type"], kwargs["mode"])
|
|
131
153
|
elif provider == "huggingface":
|
|
132
|
-
return HuggingFaceParser(
|
|
154
|
+
return HuggingFaceParser(model=model, **kwargs)
|
|
133
155
|
elif provider == "mistralocr":
|
|
134
|
-
return MistralOCRParser(
|
|
156
|
+
return MistralOCRParser(model)
|
|
135
157
|
else:
|
|
136
|
-
raise ValueError(f"Unknown parser type: {
|
|
158
|
+
raise ValueError(f"Unknown parser type: {provider_model}")
|
|
137
159
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.3
|
|
2
2
|
Name: HowdenParser
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.13
|
|
4
4
|
Summary: A simple configuration manager with Pydantic and JSON export.
|
|
5
5
|
License: MIT
|
|
6
6
|
Keywords: config,configuration,pydantic,json
|
|
@@ -12,9 +12,11 @@ Classifier: Programming Language :: Python :: 3
|
|
|
12
12
|
Classifier: Programming Language :: Python :: 3.12
|
|
13
13
|
Classifier: Programming Language :: Python :: 3.13
|
|
14
14
|
Requires-Dist: fitz (>=0.0.1.dev2,<0.0.2)
|
|
15
|
+
Requires-Dist: howdenconfig (>=0.1.13,<0.2.0)
|
|
15
16
|
Requires-Dist: langchain (>=0.3.27,<0.4.0)
|
|
16
17
|
Requires-Dist: llama-parse (>=0.6.58,<0.7.0)
|
|
17
18
|
Requires-Dist: mistralai (>=1.9.3,<2.0.0)
|
|
19
|
+
Requires-Dist: pypdf2 (>=3.0.1,<4.0.0)
|
|
18
20
|
Requires-Dist: transformers (>=4.55.2,<5.0.0)
|
|
19
21
|
Project-URL: Documentation, https://github.com/yourusername/config
|
|
20
22
|
Project-URL: Homepage, https://github.com/yourusername/config
|
|
@@ -3,7 +3,7 @@ name = "HowdenParser"
|
|
|
3
3
|
description = ""
|
|
4
4
|
readme = "README.md"
|
|
5
5
|
requires-python = ">=3.12,<4.0"
|
|
6
|
-
dependencies = [ "mistralai (>=1.9.3,<2.0.0)", "llama-parse (>=0.6.58,<0.7.0)", "langchain (>=0.3.27,<0.4.0)", "transformers (>=4.55.2,<5.0.0)", "fitz (>=0.0.1.dev2,<0.0.2)",]
|
|
6
|
+
dependencies = [ "mistralai (>=1.9.3,<2.0.0)", "llama-parse (>=0.6.58,<0.7.0)", "langchain (>=0.3.27,<0.4.0)", "transformers (>=4.55.2,<5.0.0)", "fitz (>=0.0.1.dev2,<0.0.2)", "howdenconfig (>=0.1.13,<0.2.0)", "pypdf2 (>=3.0.1,<4.0.0)",]
|
|
7
7
|
[[project.authors]]
|
|
8
8
|
name = "JesperThoftIllemannJ"
|
|
9
9
|
email = "jesper.jaeger@howdendanmark.dk"
|
|
@@ -14,7 +14,7 @@ build-backend = "poetry.core.masonry.api"
|
|
|
14
14
|
|
|
15
15
|
[tool.poetry]
|
|
16
16
|
name = "HowdenParser"
|
|
17
|
-
version = "0.1.
|
|
17
|
+
version = "0.1.13"
|
|
18
18
|
description = "A simple configuration manager with Pydantic and JSON export."
|
|
19
19
|
authors = [ "JesperThoftIllemannJ <jesper.jaeger@howdendanmark.dk>",]
|
|
20
20
|
readme = "README.md"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|