HowdenParser 0.1.13__tar.gz → 0.1.15__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ from .parser import Parser
2
+
3
+ __all__ = ["Parser"]
@@ -0,0 +1,3 @@
1
+ from .huggingface import Parameter as HuggingfaceParameter
2
+ from .mistralocr import Parameter as MistralocrParameter
3
+ from .llamaparser import Parameter as LlamaparserParameter
@@ -0,0 +1,173 @@
1
+ from abc import ABC, abstractmethod
2
+ import os
3
+ import logging
4
+ from pathlib import Path
5
+ from PyPDF2 import PdfReader
6
+ import dotenv
7
+
8
+ dotenv.load_dotenv()
9
+
10
+
11
+ class BaseParser(ABC):
12
+ _registry: dict[str, type["BaseParser"]] = {}
13
+
14
+ def __init_subclass__(cls, name: str | None = None, **kwargs):
15
+ """Automatically register subclasses under a key."""
16
+ super().__init_subclass__(**kwargs)
17
+ key = name or cls.__name__.lower().replace("parser", "")
18
+ BaseParser._registry[key] = cls
19
+ BaseParser._registry.pop("", None)
20
+
21
+ @abstractmethod
22
+ def parse(self, text: str):
23
+ pass
24
+
25
+
26
+ class Parser(BaseParser):
27
+ """Factory + registry interface for all parsers."""
28
+
29
+ @classmethod
30
+ def available_parsers(cls) -> dict[str, list[str]]:
31
+ """Return registered parsers and their init arguments."""
32
+ import inspect
33
+ result = {}
34
+ for name, parser_cls in BaseParser._registry.items():
35
+ sig = inspect.signature(parser_cls.__init__)
36
+ result[name] = [p for p in sig.parameters if p != "self"]
37
+ result.pop('', None)
38
+ return result
39
+
40
+ @classmethod
41
+ def create(cls, config: dict | None = None, **kwargs) -> BaseParser:
42
+ provider = kwargs["provider_and_model"].split(":")[0].lower()
43
+ model = kwargs["provider_and_model"].split(":")[1].lower()
44
+ if provider not in BaseParser._registry:
45
+ raise ValueError(f"Unknown parser '{provider}'. "
46
+ f"Available: {cls.available_parsers()}")
47
+
48
+ parser_cls = BaseParser._registry[provider]
49
+ import inspect
50
+ merged_args = {**(config or {}), **kwargs}
51
+ if "model" in inspect.signature(parser_cls.__init__).parameters:
52
+ merged_args["model"] = model
53
+
54
+ # Remove keys not in constructor
55
+ sig = inspect.signature(parser_cls.__init__)
56
+ valid_args = {k: v for k, v in merged_args.items() if k in sig.parameters and k != "self"}
57
+
58
+ return parser_cls(**valid_args)
59
+
60
+
61
+
62
+
63
+ @abstractmethod
64
+ def parse(self, text: str):
65
+ pass
66
+
67
+
68
+ # --- Parsers ---
69
+ class MistralOCRParser(BaseParser, name="mistralocr"):
70
+ def __init__(self,provider_and_model:str) -> None:
71
+ from mistralai import Mistral
72
+
73
+ self.model = provider_and_model.split(":")[1]
74
+ self.current_cost: float = 0.0
75
+ self.total_cost_euro: float = 0.0
76
+
77
+ api_key = os.getenv("MISTRAL-OCR-API-TOKEN")
78
+ if not api_key:
79
+ raise EnvironmentError("Missing MISTRAL-OCR-API-TOKEN in .env file.")
80
+
81
+ self.client = Mistral(api_key=api_key)
82
+
83
+ def parse(self, file_path: Path) -> str:
84
+ def upload_pdf(filename):
85
+ uploaded_pdf = self.client.files.upload(
86
+ file={"file_name": filename, "content": open(filename, "rb")},
87
+ purpose="ocr"
88
+ )
89
+ signed_url = self.client.files.get_signed_url(file_id=uploaded_pdf.id)
90
+ return signed_url.url
91
+
92
+ ocr_response = self.client.ocr.process(
93
+ model=self.model,
94
+ document={"type": "document_url", "document_url": upload_pdf(file_path)},
95
+ include_image_base64=True,
96
+ )
97
+
98
+ self.current_cost = 1 / 1000 * self._count_pages(file_path)
99
+ self.total_cost_euro += self.current_cost
100
+
101
+ return "\n".join(doc.markdown for doc in ocr_response.pages)
102
+
103
+ @staticmethod
104
+ def _count_pages(file_path: Path) -> int:
105
+ reader = PdfReader(str(file_path))
106
+ return len(reader.pages)
107
+
108
+
109
+ class LangChainParser(BaseParser, name="langchain"):
110
+ def __init__(self, model: str):
111
+ from langchain.llms import OpenAI
112
+ self.model_name = model
113
+ self.model = OpenAI(model_name=model)
114
+
115
+ def parse(self, text: str) -> dict:
116
+ response = self.model(text)
117
+ return {"source": "LangChain", "output": response}
118
+
119
+
120
+ class LlamaParser(BaseParser, name="llamaparser"):
121
+ def __init__(self, result_type: str, mode: bool) -> None:
122
+ logging.info("Initializing LlamaParser...")
123
+
124
+ from llama_parse import LlamaParse, ResultType
125
+
126
+ if result_type.lower() in ("md", "markdown"):
127
+ self.result_type = ResultType.MD
128
+
129
+ api_key = os.getenv("LLAMA-PARSER-API-TOKEN")
130
+ if not api_key:
131
+ raise EnvironmentError("Missing LLAMA-PARSER-API-TOKEN in .env file.")
132
+
133
+ self.parser = LlamaParse(api_key=api_key, result_type=self.result_type, premium_mode=mode)
134
+
135
+ def parse(self, file_path: Path) -> str:
136
+ documents = self.parser.load_data(str(file_path))
137
+ return "\n".join(doc.text for doc in documents)
138
+
139
+
140
+ class HuggingFaceParser(BaseParser, name="huggingface"):
141
+ def __init__(self, model: str, result_type: str) -> None:
142
+ from transformers import pipeline
143
+
144
+ if result_type.lower() in ("md", "markdown"):
145
+ self.result_type = "markdown"
146
+ else:
147
+ self.result_type = "text"
148
+
149
+ api_key = os.getenv("HF-API-TOKEN")
150
+ if not api_key:
151
+ raise EnvironmentError("Missing HF-API-TOKEN in .env file.")
152
+
153
+ self.parser = pipeline(
154
+ task="document-question-answering",
155
+ model=model,
156
+ use_auth_token=api_key
157
+ )
158
+
159
+ def parse(self, file_path: Path) -> str:
160
+ import fitz
161
+ pdf_doc = fitz.open(file_path)
162
+ output_parts = []
163
+
164
+ for page in pdf_doc:
165
+ pix = page.get_pixmap(dpi=200)
166
+ img_bytes = pix.tobytes("png")
167
+ response = self.parser(img_bytes, question="Extract all text")
168
+ if response and "answer" in response[0]:
169
+ output_parts.append(response[0]["answer"])
170
+
171
+ return "\n\n".join(output_parts) if self.result_type == "markdown" else " ".join(output_parts)
172
+
173
+
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: HowdenParser
3
- Version: 0.1.13
3
+ Version: 0.1.15
4
4
  Summary: A simple configuration manager with Pydantic and JSON export.
5
5
  License: MIT
6
6
  Keywords: config,configuration,pydantic,json
@@ -14,7 +14,7 @@ build-backend = "poetry.core.masonry.api"
14
14
 
15
15
  [tool.poetry]
16
16
  name = "HowdenParser"
17
- version = "0.1.13"
17
+ version = "0.1.15"
18
18
  description = "A simple configuration manager with Pydantic and JSON export."
19
19
  authors = [ "JesperThoftIllemannJ <jesper.jaeger@howdendanmark.dk>",]
20
20
  readme = "README.md"
@@ -1,3 +0,0 @@
1
- from .parser import ParserFactory
2
-
3
- __all__ = ["ParserFactory"]
@@ -1,15 +0,0 @@
1
- import os
2
- import importlib
3
-
4
- # Get all .py files in this folder except __init__.py
5
- current_dir = os.path.dirname(__file__)
6
- for filename in os.listdir(current_dir):
7
- if filename.endswith(".py") and filename != "__init__.py":
8
- module_name = filename[:-3] # remove .py
9
- module = importlib.import_module(f".{module_name}", package=__name__)
10
-
11
- # Add all classes from the module to the package namespace
12
- for attr_name in dir(module):
13
- attr = getattr(module, attr_name)
14
- if isinstance(attr, type): # only classes
15
- globals()[attr_name] = attr
@@ -1,159 +0,0 @@
1
- from abc import ABC, abstractmethod
2
- from dotenv import load_dotenv
3
- import os
4
- from pathlib import Path
5
- import logging
6
- from PyPDF2 import PdfReader
7
-
8
- load_dotenv()
9
-
10
- class BaseParser(ABC):
11
- @abstractmethod
12
- def parse(self, text: str) -> dict:
13
- pass
14
-
15
-
16
- class MistralOCRParser(BaseParser):
17
- def __init__(self, model: str) -> None:
18
- from mistralai import Mistral
19
- self.model = model
20
- self.current_cost: float = 0.0
21
- self.total_cost_euro: float = 0.0
22
-
23
- name = "MISTRAL-OCR-API-TOKEN"
24
- api_key = os.getenv(name)
25
- if not api_key:
26
- raise EnvironmentError(f"Missing {name} in .env file.")
27
-
28
- self.client = Mistral(api_key=api_key)
29
-
30
- def parse(self, file_path: Path) -> str:
31
- def upload_pdf(filename):
32
- uploaded_pdf = self.client.files.upload(
33
- file={
34
- "file_name": filename,
35
- "content": open(filename, "rb"),
36
- },
37
- purpose="ocr"
38
- )
39
- signed_url = self.client.files.get_signed_url(file_id=uploaded_pdf.id)
40
- return signed_url.url
41
-
42
- ocr_response = self.client.ocr.process(
43
- model=self.model,
44
- document={
45
- "type": "document_url",
46
- "document_url": upload_pdf(file_path),
47
- },
48
- include_image_base64=True,
49
- )
50
-
51
- self.current_cost = 1/1000 * self._count_pages(file_path)
52
- self.total_cost_euro += self.current_cost
53
-
54
- return "\n".join(doc.markdown for doc in ocr_response.pages)
55
-
56
- @staticmethod
57
- def _count_pages(file_path: Path) -> int:
58
- reader = PdfReader(str(file_path))
59
- return len(reader.pages)
60
-
61
-
62
-
63
- class LangChainParser(BaseParser):
64
- def __init__(self, model: str):
65
- from langchain.llms import OpenAI
66
- self.model_name = model
67
- self.model = OpenAI(model_name=model)
68
-
69
- def parse(self, text: str) -> dict:
70
- response = self.model(text)
71
- return {"source": "LangChain", "output": response}
72
-
73
-
74
- class LlamaParser(BaseParser):
75
- def __init__(self, result_type: str, mode: bool) -> None:
76
- logging.info("Initializing LlamaParser...")
77
- logging.info("Loading LlamaParse package...")
78
-
79
- from llama_parse import LlamaParse, ResultType
80
-
81
- if result_type.lower() == "md" or result_type.lower() == "markdown":
82
- self.result_type = ResultType.MD
83
-
84
- name = "LLAMA-PARSER-API-TOKEN"
85
- api_key = os.getenv(name)
86
- if not api_key:
87
- raise EnvironmentError(f"Missing {name} in .env file.")
88
-
89
- logging.info("Initializing LlamaParse parser...")
90
- self.parser = LlamaParse(
91
- api_key=api_key,
92
- result_type=self.result_type,
93
- premium_mode=mode
94
- )
95
- logging.info("LlamaParser initialized successfully.")
96
-
97
- def parse(self, file_path: Path) -> str:
98
- logging.info(f"Parsing file: {file_path}")
99
- documents = self.parser.load_data(str(file_path))
100
- text = "\n".join(doc.text for doc in documents)
101
- logging.info(f"Parsing completed. Extracted {len(documents)} documents.")
102
- return text
103
-
104
- class HuggingFaceParser(BaseParser):
105
- def __init__(self, model: str, result_type: str) -> None:
106
- from transformers import pipeline
107
-
108
- if result_type.lower() in ("md", "markdown"):
109
- self.result_type = "markdown"
110
- else:
111
- self.result_type = "text"
112
-
113
- name = "HF-API-TOKEN"
114
- self.api_key = os.getenv(name)
115
- if not self.api_key:
116
- raise EnvironmentError(f"Missing {name} in .env file.")
117
-
118
- # Example model: microsoft/layoutlmv3-base-finetuned-docvqa
119
- self.parser = pipeline(
120
- task="document-question-answering",
121
- model=model,
122
- use_auth_token=self.api_key
123
- )
124
-
125
- def parse(self, file_path: Path) -> str:
126
- import fitz
127
-
128
- pdf_doc = fitz.open(file_path)
129
- output_parts = []
130
-
131
- for page in pdf_doc:
132
- pix = page.get_pixmap(dpi=200)
133
- img_bytes = pix.tobytes("png")
134
- response = self.parser(img_bytes, question="Extract all text")
135
- if response and "answer" in response[0]:
136
- output_parts.append(response[0]["answer"])
137
-
138
- if self.result_type == "markdown":
139
- return "\n\n".join(output_parts)
140
- else:
141
- return " ".join(output_parts)
142
-
143
- # === Step 3: Dynamic factory using string input ===
144
- class ParserFactory:
145
- @staticmethod
146
- def get_parser(provider_model: str, **kwargs) -> BaseParser:
147
- provider = provider_model.partition(":")[0].lower()
148
- model = provider_model.partition(":")[2]
149
- if provider == "langchain":
150
- return LangChainParser(model=model)
151
- elif provider == "llamaparser":
152
- return LlamaParser(kwargs["result_type"], kwargs["mode"])
153
- elif provider == "huggingface":
154
- return HuggingFaceParser(model=model, **kwargs)
155
- elif provider == "mistralocr":
156
- return MistralOCRParser(model)
157
- else:
158
- raise ValueError(f"Unknown parser type: {provider_model}")
159
-
File without changes