graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,220 @@
1
+ from logging import Logger
2
+ from langchain_core.callbacks.base import BaseCallbackHandler
3
+ from langchain_core.outputs.chat_generation import ChatGeneration
4
+ from langchain_core.outputs.llm_result import LLMResult
5
+
6
+ from typing import Any
7
+
8
+ GEMINI_PRO_INPUT_COST_PER_1K_TOKENS = 0.0001
9
+ GEMINI_PRO_OUTPUT_COST_PER_1K_TOKENS = 0.0004
10
+
11
+ #per_k
12
+ # storage per hour
13
+ cost_table = {
14
+ "gemini-2.0-flash": {"input": 0.0001, "output": 0.0004, "cache" : 0.0, "storage_per_hour": 0.0},
15
+ "gemini-1.5-pro": {"input": 0.001250, "output": 0.005, "cache" : 0.0, "storage_per_hour": 0.0},
16
+ "gemini-2.5-flash-preview-04-17": {"input": 0.000150, "output": 0.0035, "cache" : 0.0000375, "storage_per_hour": 0.0010},
17
+ "gemini-2.5-flash": {"input": 0.000300, "output": 0.0025, "cache" : 0.0000375, "storage_per_hour": 0.0010},
18
+ "gemini-2.5-pro-preview-03-25": {"input": 0.001250, "output": 0.0100, "cache": 0.00031, "storage_per_hour": 0.0045},
19
+ "gemini-2.5-pro": {"input": 0.001250, "output": 0.0100, "cache": 0.00031, "storage_per_hour": 0.0045},
20
+ "gemini-2.5-flash-lite": {"input": 0.0001, "output": 0.0040, "cache": 0.00025, "storage_per_hour": 0.001}
21
+ }
22
+ keys = list(cost_table.keys())
23
+ for k in keys:
24
+ if k.startswith('models/'):
25
+ pass
26
+ else:
27
+ cost_table['models/' + k] = cost_table[k]
28
+
29
+ def calculate_gemini_cost(input_tokens, output_tokens, cached_tokens, model_name):
30
+ """Calculates the cost based on Gemini Pro pricing."""
31
+ cost = cost_table.get(model_name, {"input": 1.250/1000000, "output": 5.0/1000000, "cache" : 0.0, "storage_per_hour": 0.0}) # prodential to assume high priced model
32
+ input_cost = ((input_tokens-cached_tokens) / 1000) * cost['input']
33
+ cache_cost = cached_tokens / 1000 * cost['cache']
34
+ output_cost = (output_tokens / 1000) * cost['output']
35
+ total_cost = input_cost + output_cost + cache_cost
36
+ return total_cost
37
+ import time
38
+ class GeminiCostCallbackHandler(BaseCallbackHandler):
39
+ """A custom callback handler to track Gemini API costs."""
40
+
41
+ def __init__(self):
42
+ super().__init__()
43
+ self.total_input_tokens = 0
44
+ self.total_output_tokens = 0
45
+ self.cache_tokens = 0
46
+ self.reasoning_tokens = 0
47
+ self.total_cost = 0.0
48
+ self.usage_history = []
49
+ self.run_start_time = None
50
+ self.run_end_time = None
51
+ def on_llm_start(self, serialized, prompts, *, run_id, parent_run_id = None, tags = None, metadata = None, **kwargs):
52
+ self.run_start_time = time.time()
53
+ return super().on_llm_start(serialized, prompts, run_id=run_id, parent_run_id=parent_run_id, tags=tags, metadata=metadata, **kwargs)
54
+ def on_llm_end(self, response: LLMResult, **kwargs: Any) -> None:
55
+ """Called at the end of an LLM call."""
56
+ self.run_end_time = time.time()
57
+ for generation in response.generations:
58
+ # The 'generation' is a list of ChatGeneration or Generation objects
59
+ for gen in generation:
60
+ # Check if the generation object is a ChatGeneration instance
61
+ # and has the 'usage_metadata' attribute.
62
+ if isinstance(gen, ChatGeneration) and hasattr(gen, 'message'):
63
+ message = gen.message
64
+ if hasattr(message, 'usage_metadata'):
65
+ usage_metadata = message.usage_metadata
66
+ if usage_metadata is not None:
67
+
68
+
69
+ input_tokens = usage_metadata.get("input_tokens", 0)
70
+ output_tokens = usage_metadata.get("output_tokens", 0)
71
+
72
+ try:
73
+ cached_tokens = usage_metadata['input_token_details']['cache_read']
74
+ except KeyError:
75
+ cached_tokens = 0
76
+ output_tokens = usage_metadata.get("output_tokens", 0)
77
+ try:
78
+ reasoning_tokens = usage_metadata['output_token_details']['reasoning']
79
+ except KeyError:
80
+ reasoning_tokens = 0
81
+
82
+ if input_tokens > 0 or output_tokens > 0:
83
+ cost = calculate_gemini_cost(input_tokens, output_tokens, cached_tokens, model_name = gen.generation_info['model_name'])
84
+ self.total_input_tokens += input_tokens
85
+ self.total_output_tokens += output_tokens
86
+ self.reasoning_tokens += reasoning_tokens
87
+ self.cache_tokens += cached_tokens
88
+ self.total_cost += cost
89
+ self.usage_history.append({
90
+ "model_name": gen.generation_info['model_name'],
91
+ "input_tokens" : input_tokens,
92
+ "output_tokens":output_tokens,
93
+ "cached_tokens":cached_tokens,
94
+ "reasoning_tokens":reasoning_tokens,
95
+ "cost":cost,
96
+ "start_time": self.run_start_time,
97
+ "end_time" : self.run_end_time
98
+
99
+ })
100
+
101
+ def reset(self):
102
+ """Resets the counters."""
103
+ self.total_input_tokens = 0
104
+ self.total_output_tokens = 0
105
+ self.total_cost = 0.0
106
+
107
+ def __repr__(self):
108
+ return (
109
+ f"Total Input Tokens: {self.total_input_tokens}\n"
110
+ f"Total Output Tokens: {self.total_output_tokens}\n"
111
+ f"Total Cost: ${self.total_cost:.8f}"
112
+ )
113
+ def model_dump(self):
114
+ return (
115
+ {"input_tokens": self.total_input_tokens,
116
+ "output_tokens": self.total_output_tokens,
117
+ "total_cost": self.total_cost,
118
+ 'usage_history' : self.usage_history}
119
+ )
120
+ from contextlib import contextmanager
121
+
122
+ @contextmanager
123
+ def get_gemini_callback_cost():
124
+ """A context manager to track Gemini API costs for a block of code."""
125
+ # Create an instance of the handler
126
+ """_summary_
127
+
128
+ Yields:
129
+ _type_: _description_
130
+
131
+ _usage_
132
+ with get_gemini_callback_cost() as cb:
133
+ result = chain.invoke(
134
+ {"city": "Paris"},
135
+ config={"callbacks": [cb]} # Pass the yielded handler to the chain
136
+ )
137
+ print("\n--- Inside Context Manager ---")
138
+ print(result.content)
139
+
140
+ # You can access the cost immediately after the call
141
+ print("\n--- Cost After First Call ---")
142
+ print(cb)
143
+
144
+ # The state is preserved even after the `with` block exits
145
+ print("\n--- Final Cost from Context Manager ---")
146
+ print(cb)
147
+
148
+ """
149
+ callback_handler = GeminiCostCallbackHandler()
150
+ try:
151
+ # Yield the handler so it can be used inside the 'with' block
152
+ # and its state can be accessed after the block.
153
+ yield callback_handler
154
+ finally:
155
+ # The code inside the 'with' block has finished.
156
+ # The handler now holds the final cost.
157
+ pass
158
+
159
+ class PromptCostTokenLogger(BaseCallbackHandler):
160
+ def __init__(self, logger: Logger):
161
+ self.cost_token_logger: Logger = logger
162
+ self.total_input_tokens = 0
163
+ self.total_reasoning_tokens = 0
164
+ self.total_cached_tokens = 0
165
+ self.total_output_tokens = 0
166
+ self.total_cost = 0
167
+ def on_llm_end(self, response, *, run_id, parent_run_id = None, **kwargs):
168
+ # self.cost_token_logger.info(response.response_metadata)
169
+ for g in response.generations:
170
+ for gg in g:
171
+
172
+ to_log = {'response_metadata': None, 'usage_metadata': None}
173
+ if hasattr(gg.message, "response_metadata"):
174
+ to_log["response_metadata"] = f"{gg.message.response_metadata}"
175
+ if hasattr(gg.message, "usage_metadata"):
176
+ to_log["usage_metadata"] = f"{gg.message.usage_metadata}"
177
+ if to_log:
178
+ self.cost_token_logger.info(str(to_log))
179
+ # example "{'input_tokens': 106170, 'output_tokens': 7652, 'total_tokens': 118594, 'input_token_details': {'cache_read': 106164}, 'output_token_details': {'reasoning': 4772}}"
180
+ for generation in response.generations:
181
+ # The 'generation' is a list of ChatGeneration or Generation objects
182
+ for gen in generation:
183
+ # Check if the generation object is a ChatGeneration instance
184
+ # and has the 'usage_metadata' attribute.
185
+ if isinstance(gen, ChatGeneration) and hasattr(gen.message, 'usage_metadata') and gen.message.usage_metadata is not None:
186
+
187
+ usage_metadata = gen.message.usage_metadata
188
+
189
+ input_tokens = usage_metadata.get("input_tokens", 0)
190
+ try:
191
+ cached_tokens = usage_metadata['input_token_details']['cache_read']
192
+ except KeyError:
193
+ cached_tokens = 0
194
+ output_tokens = usage_metadata.get("output_tokens", 0)
195
+ try:
196
+ reasoning_tokens = usage_metadata['output_token_details']['reasoning']
197
+ except KeyError:
198
+ reasoning_tokens = 0
199
+ if input_tokens > 0 or output_tokens > 0:
200
+ cost = calculate_gemini_cost(input_tokens, output_tokens, cached_tokens, model_name = gen.generation_info['model_name'])
201
+ self.total_input_tokens += input_tokens
202
+ self.total_cached_tokens += cached_tokens
203
+ self.total_output_tokens += output_tokens
204
+ self.total_reasoning_tokens += reasoning_tokens
205
+ self.total_cost += cost
206
+ return super().on_llm_end(response, run_id=run_id, parent_run_id=parent_run_id, **kwargs)
207
+ def on_llm_error(self, error, *, run_id, parent_run_id = None, **kwargs):
208
+
209
+ return super().on_llm_error(error, run_id=run_id, parent_run_id=parent_run_id, **kwargs)
210
+ class PromptTokenCounter(BaseCallbackHandler):
211
+ def on_llm_start(self, serialized: dict, prompts: list[str], **kwargs):
212
+ for prompt in prompts:
213
+ token_count = self.count_tokens(prompt)
214
+ print(f"Prompt: {prompt}")
215
+ print(f"Token count: {token_count}")
216
+
217
+ def count_tokens(self, text: str) -> int:
218
+ # Implement your token counting logic here
219
+ # For example, a simple approximation:
220
+ return len(text.split())
@@ -0,0 +1,135 @@
1
+
2
+ #### log utils
3
+
4
+ import logging
5
+ import sqlite3
6
+ import threading
7
+ import os
8
+ import traceback
9
+ def safe_format_exception(exc: Exception, base_path: str = None):
10
+ """Format exception with paths relative to project root."""
11
+ if base_path is None:
12
+ base_path = os.getcwd() # default to current working directory
13
+
14
+ lines = []
15
+ for line in traceback.format_exception(type(exc), exc, exc.__traceback__):
16
+ if line.startswith(' File'):
17
+ parts = line.split('"')
18
+ if len(parts) >= 3:
19
+ full_path = parts[1]
20
+ try:
21
+ relative_path = os.path.relpath(full_path, base_path)
22
+ line = line.replace(full_path, relative_path)
23
+ except ValueError:
24
+ # relpath failed (different drive?), keep original
25
+ pass
26
+ lines.append(line)
27
+
28
+ return ''.join(lines)
29
+
30
+ def trace_logger_hierarchy(logger):
31
+ while logger:
32
+ print(f"Logger Name: {logger.name}")
33
+ print(f" Level: {logging.getLevelName(logger.level)}")
34
+ print(f" Handlers: {logger.handlers}")
35
+ print(f" Propagate: {logger.propagate}")
36
+ print("-" * 40)
37
+ logger = logger.parent
38
+
39
+ class SQLiteHandler(logging.Handler):
40
+ """
41
+ Custom logging handler that writes log records to an SQLite database,
42
+ including filename and line number information.
43
+ """
44
+
45
+ def __init__(self, db_path):
46
+ """
47
+ Initializes the handler with the database path.
48
+ Ensures the log table exists.
49
+ """
50
+ super().__init__()
51
+ self.db_path = db_path
52
+ self.lock = threading.Lock()
53
+ self._initialize_database()
54
+ # Set up a formatter to format the log records
55
+ self.formatter = logging.Formatter('%(asctime)s', '%Y-%m-%d %H:%M:%S')
56
+
57
+ def _initialize_database(self):
58
+ """
59
+ Creates the logs table if it doesn't already exist.
60
+ """
61
+ with sqlite3.connect(self.db_path) as conn:
62
+ cursor = conn.cursor()
63
+ cursor.execute('''
64
+ CREATE TABLE IF NOT EXISTS logs (
65
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
66
+ timestamp TEXT,
67
+ level TEXT,
68
+ module TEXT,
69
+ filename TEXT,
70
+ line_number INTEGER,
71
+ message TEXT
72
+ )
73
+ ''')
74
+ conn.commit()
75
+
76
+ def emit(self, record):
77
+ """
78
+ Inserts a new log record into the database.
79
+ """
80
+ try:
81
+ # Ensure the record is formatted to populate all fields
82
+ self.format(record)
83
+ # Format the timestamp using the formatter
84
+ timestamp = self.formatter.formatTime(record)
85
+ #with self.lock:
86
+ with sqlite3.connect(self.db_path, timeout=10) as conn:
87
+ cursor = conn.cursor()
88
+ cursor.execute('''
89
+ INSERT INTO logs (timestamp, level, module, filename, line_number, message)
90
+ VALUES (?, ?, ?, ?, ?, ?)
91
+ ''', (timestamp, record.levelname, record.module,
92
+ record.filename, record.lineno, record.getMessage()))
93
+ conn.commit()
94
+ except Exception:
95
+ self.handleError(record)
96
+ def __del__(self):
97
+ """
98
+ Destructor to perform a WAL checkpoint when the handler is destroyed.
99
+ """
100
+ try:
101
+ with sqlite3.connect(self.db_path) as conn:
102
+ conn.execute('PRAGMA wal_checkpoint;')
103
+ conn.commit()
104
+ except Exception as e:
105
+ print(f"Error during WAL checkpoint: {e}")
106
+
107
+
108
+ """_summary_
109
+
110
+ usage:
111
+
112
+ import logging
113
+
114
+ # Configure the logger
115
+ logger = logging.getLogger(__name__)
116
+ logger.setLevel(logging.DEBUG)
117
+
118
+ # Create the SQLite logging handler
119
+ sqlite_handler = SQLiteHandler('application_logs.db')
120
+ sqlite_handler.setLevel(logging.DEBUG)
121
+
122
+ # Add the handler to the logger
123
+ logger.addHandler(sqlite_handler)
124
+
125
+ # Register the handler's close method with the logging shutdown
126
+ logging.shutdown = sqlite_handler.close
127
+
128
+ # Log messages
129
+ logger.info('This is an info message.')
130
+ logger.error('This is an error message.')
131
+
132
+ # When the application is terminating
133
+ logging.shutdown()
134
+
135
+ """