graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
from logging import Logger
|
|
2
|
+
from langchain_core.callbacks.base import BaseCallbackHandler
|
|
3
|
+
from langchain_core.outputs.chat_generation import ChatGeneration
|
|
4
|
+
from langchain_core.outputs.llm_result import LLMResult
|
|
5
|
+
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
GEMINI_PRO_INPUT_COST_PER_1K_TOKENS = 0.0001
|
|
9
|
+
GEMINI_PRO_OUTPUT_COST_PER_1K_TOKENS = 0.0004
|
|
10
|
+
|
|
11
|
+
#per_k
|
|
12
|
+
# storage per hour
|
|
13
|
+
cost_table = {
|
|
14
|
+
"gemini-2.0-flash": {"input": 0.0001, "output": 0.0004, "cache" : 0.0, "storage_per_hour": 0.0},
|
|
15
|
+
"gemini-1.5-pro": {"input": 0.001250, "output": 0.005, "cache" : 0.0, "storage_per_hour": 0.0},
|
|
16
|
+
"gemini-2.5-flash-preview-04-17": {"input": 0.000150, "output": 0.0035, "cache" : 0.0000375, "storage_per_hour": 0.0010},
|
|
17
|
+
"gemini-2.5-flash": {"input": 0.000300, "output": 0.0025, "cache" : 0.0000375, "storage_per_hour": 0.0010},
|
|
18
|
+
"gemini-2.5-pro-preview-03-25": {"input": 0.001250, "output": 0.0100, "cache": 0.00031, "storage_per_hour": 0.0045},
|
|
19
|
+
"gemini-2.5-pro": {"input": 0.001250, "output": 0.0100, "cache": 0.00031, "storage_per_hour": 0.0045},
|
|
20
|
+
"gemini-2.5-flash-lite": {"input": 0.0001, "output": 0.0040, "cache": 0.00025, "storage_per_hour": 0.001}
|
|
21
|
+
}
|
|
22
|
+
keys = list(cost_table.keys())
|
|
23
|
+
for k in keys:
|
|
24
|
+
if k.startswith('models/'):
|
|
25
|
+
pass
|
|
26
|
+
else:
|
|
27
|
+
cost_table['models/' + k] = cost_table[k]
|
|
28
|
+
|
|
29
|
+
def calculate_gemini_cost(input_tokens, output_tokens, cached_tokens, model_name):
|
|
30
|
+
"""Calculates the cost based on Gemini Pro pricing."""
|
|
31
|
+
cost = cost_table.get(model_name, {"input": 1.250/1000000, "output": 5.0/1000000, "cache" : 0.0, "storage_per_hour": 0.0}) # prodential to assume high priced model
|
|
32
|
+
input_cost = ((input_tokens-cached_tokens) / 1000) * cost['input']
|
|
33
|
+
cache_cost = cached_tokens / 1000 * cost['cache']
|
|
34
|
+
output_cost = (output_tokens / 1000) * cost['output']
|
|
35
|
+
total_cost = input_cost + output_cost + cache_cost
|
|
36
|
+
return total_cost
|
|
37
|
+
import time
|
|
38
|
+
class GeminiCostCallbackHandler(BaseCallbackHandler):
|
|
39
|
+
"""A custom callback handler to track Gemini API costs."""
|
|
40
|
+
|
|
41
|
+
def __init__(self):
|
|
42
|
+
super().__init__()
|
|
43
|
+
self.total_input_tokens = 0
|
|
44
|
+
self.total_output_tokens = 0
|
|
45
|
+
self.cache_tokens = 0
|
|
46
|
+
self.reasoning_tokens = 0
|
|
47
|
+
self.total_cost = 0.0
|
|
48
|
+
self.usage_history = []
|
|
49
|
+
self.run_start_time = None
|
|
50
|
+
self.run_end_time = None
|
|
51
|
+
def on_llm_start(self, serialized, prompts, *, run_id, parent_run_id = None, tags = None, metadata = None, **kwargs):
|
|
52
|
+
self.run_start_time = time.time()
|
|
53
|
+
return super().on_llm_start(serialized, prompts, run_id=run_id, parent_run_id=parent_run_id, tags=tags, metadata=metadata, **kwargs)
|
|
54
|
+
def on_llm_end(self, response: LLMResult, **kwargs: Any) -> None:
|
|
55
|
+
"""Called at the end of an LLM call."""
|
|
56
|
+
self.run_end_time = time.time()
|
|
57
|
+
for generation in response.generations:
|
|
58
|
+
# The 'generation' is a list of ChatGeneration or Generation objects
|
|
59
|
+
for gen in generation:
|
|
60
|
+
# Check if the generation object is a ChatGeneration instance
|
|
61
|
+
# and has the 'usage_metadata' attribute.
|
|
62
|
+
if isinstance(gen, ChatGeneration) and hasattr(gen, 'message'):
|
|
63
|
+
message = gen.message
|
|
64
|
+
if hasattr(message, 'usage_metadata'):
|
|
65
|
+
usage_metadata = message.usage_metadata
|
|
66
|
+
if usage_metadata is not None:
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
input_tokens = usage_metadata.get("input_tokens", 0)
|
|
70
|
+
output_tokens = usage_metadata.get("output_tokens", 0)
|
|
71
|
+
|
|
72
|
+
try:
|
|
73
|
+
cached_tokens = usage_metadata['input_token_details']['cache_read']
|
|
74
|
+
except KeyError:
|
|
75
|
+
cached_tokens = 0
|
|
76
|
+
output_tokens = usage_metadata.get("output_tokens", 0)
|
|
77
|
+
try:
|
|
78
|
+
reasoning_tokens = usage_metadata['output_token_details']['reasoning']
|
|
79
|
+
except KeyError:
|
|
80
|
+
reasoning_tokens = 0
|
|
81
|
+
|
|
82
|
+
if input_tokens > 0 or output_tokens > 0:
|
|
83
|
+
cost = calculate_gemini_cost(input_tokens, output_tokens, cached_tokens, model_name = gen.generation_info['model_name'])
|
|
84
|
+
self.total_input_tokens += input_tokens
|
|
85
|
+
self.total_output_tokens += output_tokens
|
|
86
|
+
self.reasoning_tokens += reasoning_tokens
|
|
87
|
+
self.cache_tokens += cached_tokens
|
|
88
|
+
self.total_cost += cost
|
|
89
|
+
self.usage_history.append({
|
|
90
|
+
"model_name": gen.generation_info['model_name'],
|
|
91
|
+
"input_tokens" : input_tokens,
|
|
92
|
+
"output_tokens":output_tokens,
|
|
93
|
+
"cached_tokens":cached_tokens,
|
|
94
|
+
"reasoning_tokens":reasoning_tokens,
|
|
95
|
+
"cost":cost,
|
|
96
|
+
"start_time": self.run_start_time,
|
|
97
|
+
"end_time" : self.run_end_time
|
|
98
|
+
|
|
99
|
+
})
|
|
100
|
+
|
|
101
|
+
def reset(self):
|
|
102
|
+
"""Resets the counters."""
|
|
103
|
+
self.total_input_tokens = 0
|
|
104
|
+
self.total_output_tokens = 0
|
|
105
|
+
self.total_cost = 0.0
|
|
106
|
+
|
|
107
|
+
def __repr__(self):
|
|
108
|
+
return (
|
|
109
|
+
f"Total Input Tokens: {self.total_input_tokens}\n"
|
|
110
|
+
f"Total Output Tokens: {self.total_output_tokens}\n"
|
|
111
|
+
f"Total Cost: ${self.total_cost:.8f}"
|
|
112
|
+
)
|
|
113
|
+
def model_dump(self):
|
|
114
|
+
return (
|
|
115
|
+
{"input_tokens": self.total_input_tokens,
|
|
116
|
+
"output_tokens": self.total_output_tokens,
|
|
117
|
+
"total_cost": self.total_cost,
|
|
118
|
+
'usage_history' : self.usage_history}
|
|
119
|
+
)
|
|
120
|
+
from contextlib import contextmanager
|
|
121
|
+
|
|
122
|
+
@contextmanager
|
|
123
|
+
def get_gemini_callback_cost():
|
|
124
|
+
"""A context manager to track Gemini API costs for a block of code."""
|
|
125
|
+
# Create an instance of the handler
|
|
126
|
+
"""_summary_
|
|
127
|
+
|
|
128
|
+
Yields:
|
|
129
|
+
_type_: _description_
|
|
130
|
+
|
|
131
|
+
_usage_
|
|
132
|
+
with get_gemini_callback_cost() as cb:
|
|
133
|
+
result = chain.invoke(
|
|
134
|
+
{"city": "Paris"},
|
|
135
|
+
config={"callbacks": [cb]} # Pass the yielded handler to the chain
|
|
136
|
+
)
|
|
137
|
+
print("\n--- Inside Context Manager ---")
|
|
138
|
+
print(result.content)
|
|
139
|
+
|
|
140
|
+
# You can access the cost immediately after the call
|
|
141
|
+
print("\n--- Cost After First Call ---")
|
|
142
|
+
print(cb)
|
|
143
|
+
|
|
144
|
+
# The state is preserved even after the `with` block exits
|
|
145
|
+
print("\n--- Final Cost from Context Manager ---")
|
|
146
|
+
print(cb)
|
|
147
|
+
|
|
148
|
+
"""
|
|
149
|
+
callback_handler = GeminiCostCallbackHandler()
|
|
150
|
+
try:
|
|
151
|
+
# Yield the handler so it can be used inside the 'with' block
|
|
152
|
+
# and its state can be accessed after the block.
|
|
153
|
+
yield callback_handler
|
|
154
|
+
finally:
|
|
155
|
+
# The code inside the 'with' block has finished.
|
|
156
|
+
# The handler now holds the final cost.
|
|
157
|
+
pass
|
|
158
|
+
|
|
159
|
+
class PromptCostTokenLogger(BaseCallbackHandler):
|
|
160
|
+
def __init__(self, logger: Logger):
|
|
161
|
+
self.cost_token_logger: Logger = logger
|
|
162
|
+
self.total_input_tokens = 0
|
|
163
|
+
self.total_reasoning_tokens = 0
|
|
164
|
+
self.total_cached_tokens = 0
|
|
165
|
+
self.total_output_tokens = 0
|
|
166
|
+
self.total_cost = 0
|
|
167
|
+
def on_llm_end(self, response, *, run_id, parent_run_id = None, **kwargs):
|
|
168
|
+
# self.cost_token_logger.info(response.response_metadata)
|
|
169
|
+
for g in response.generations:
|
|
170
|
+
for gg in g:
|
|
171
|
+
|
|
172
|
+
to_log = {'response_metadata': None, 'usage_metadata': None}
|
|
173
|
+
if hasattr(gg.message, "response_metadata"):
|
|
174
|
+
to_log["response_metadata"] = f"{gg.message.response_metadata}"
|
|
175
|
+
if hasattr(gg.message, "usage_metadata"):
|
|
176
|
+
to_log["usage_metadata"] = f"{gg.message.usage_metadata}"
|
|
177
|
+
if to_log:
|
|
178
|
+
self.cost_token_logger.info(str(to_log))
|
|
179
|
+
# example "{'input_tokens': 106170, 'output_tokens': 7652, 'total_tokens': 118594, 'input_token_details': {'cache_read': 106164}, 'output_token_details': {'reasoning': 4772}}"
|
|
180
|
+
for generation in response.generations:
|
|
181
|
+
# The 'generation' is a list of ChatGeneration or Generation objects
|
|
182
|
+
for gen in generation:
|
|
183
|
+
# Check if the generation object is a ChatGeneration instance
|
|
184
|
+
# and has the 'usage_metadata' attribute.
|
|
185
|
+
if isinstance(gen, ChatGeneration) and hasattr(gen.message, 'usage_metadata') and gen.message.usage_metadata is not None:
|
|
186
|
+
|
|
187
|
+
usage_metadata = gen.message.usage_metadata
|
|
188
|
+
|
|
189
|
+
input_tokens = usage_metadata.get("input_tokens", 0)
|
|
190
|
+
try:
|
|
191
|
+
cached_tokens = usage_metadata['input_token_details']['cache_read']
|
|
192
|
+
except KeyError:
|
|
193
|
+
cached_tokens = 0
|
|
194
|
+
output_tokens = usage_metadata.get("output_tokens", 0)
|
|
195
|
+
try:
|
|
196
|
+
reasoning_tokens = usage_metadata['output_token_details']['reasoning']
|
|
197
|
+
except KeyError:
|
|
198
|
+
reasoning_tokens = 0
|
|
199
|
+
if input_tokens > 0 or output_tokens > 0:
|
|
200
|
+
cost = calculate_gemini_cost(input_tokens, output_tokens, cached_tokens, model_name = gen.generation_info['model_name'])
|
|
201
|
+
self.total_input_tokens += input_tokens
|
|
202
|
+
self.total_cached_tokens += cached_tokens
|
|
203
|
+
self.total_output_tokens += output_tokens
|
|
204
|
+
self.total_reasoning_tokens += reasoning_tokens
|
|
205
|
+
self.total_cost += cost
|
|
206
|
+
return super().on_llm_end(response, run_id=run_id, parent_run_id=parent_run_id, **kwargs)
|
|
207
|
+
def on_llm_error(self, error, *, run_id, parent_run_id = None, **kwargs):
|
|
208
|
+
|
|
209
|
+
return super().on_llm_error(error, run_id=run_id, parent_run_id=parent_run_id, **kwargs)
|
|
210
|
+
class PromptTokenCounter(BaseCallbackHandler):
|
|
211
|
+
def on_llm_start(self, serialized: dict, prompts: list[str], **kwargs):
|
|
212
|
+
for prompt in prompts:
|
|
213
|
+
token_count = self.count_tokens(prompt)
|
|
214
|
+
print(f"Prompt: {prompt}")
|
|
215
|
+
print(f"Token count: {token_count}")
|
|
216
|
+
|
|
217
|
+
def count_tokens(self, text: str) -> int:
|
|
218
|
+
# Implement your token counting logic here
|
|
219
|
+
# For example, a simple approximation:
|
|
220
|
+
return len(text.split())
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
|
|
2
|
+
#### log utils
|
|
3
|
+
|
|
4
|
+
import logging
|
|
5
|
+
import sqlite3
|
|
6
|
+
import threading
|
|
7
|
+
import os
|
|
8
|
+
import traceback
|
|
9
|
+
def safe_format_exception(exc: Exception, base_path: str = None):
|
|
10
|
+
"""Format exception with paths relative to project root."""
|
|
11
|
+
if base_path is None:
|
|
12
|
+
base_path = os.getcwd() # default to current working directory
|
|
13
|
+
|
|
14
|
+
lines = []
|
|
15
|
+
for line in traceback.format_exception(type(exc), exc, exc.__traceback__):
|
|
16
|
+
if line.startswith(' File'):
|
|
17
|
+
parts = line.split('"')
|
|
18
|
+
if len(parts) >= 3:
|
|
19
|
+
full_path = parts[1]
|
|
20
|
+
try:
|
|
21
|
+
relative_path = os.path.relpath(full_path, base_path)
|
|
22
|
+
line = line.replace(full_path, relative_path)
|
|
23
|
+
except ValueError:
|
|
24
|
+
# relpath failed (different drive?), keep original
|
|
25
|
+
pass
|
|
26
|
+
lines.append(line)
|
|
27
|
+
|
|
28
|
+
return ''.join(lines)
|
|
29
|
+
|
|
30
|
+
def trace_logger_hierarchy(logger):
|
|
31
|
+
while logger:
|
|
32
|
+
print(f"Logger Name: {logger.name}")
|
|
33
|
+
print(f" Level: {logging.getLevelName(logger.level)}")
|
|
34
|
+
print(f" Handlers: {logger.handlers}")
|
|
35
|
+
print(f" Propagate: {logger.propagate}")
|
|
36
|
+
print("-" * 40)
|
|
37
|
+
logger = logger.parent
|
|
38
|
+
|
|
39
|
+
class SQLiteHandler(logging.Handler):
|
|
40
|
+
"""
|
|
41
|
+
Custom logging handler that writes log records to an SQLite database,
|
|
42
|
+
including filename and line number information.
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
def __init__(self, db_path):
|
|
46
|
+
"""
|
|
47
|
+
Initializes the handler with the database path.
|
|
48
|
+
Ensures the log table exists.
|
|
49
|
+
"""
|
|
50
|
+
super().__init__()
|
|
51
|
+
self.db_path = db_path
|
|
52
|
+
self.lock = threading.Lock()
|
|
53
|
+
self._initialize_database()
|
|
54
|
+
# Set up a formatter to format the log records
|
|
55
|
+
self.formatter = logging.Formatter('%(asctime)s', '%Y-%m-%d %H:%M:%S')
|
|
56
|
+
|
|
57
|
+
def _initialize_database(self):
|
|
58
|
+
"""
|
|
59
|
+
Creates the logs table if it doesn't already exist.
|
|
60
|
+
"""
|
|
61
|
+
with sqlite3.connect(self.db_path) as conn:
|
|
62
|
+
cursor = conn.cursor()
|
|
63
|
+
cursor.execute('''
|
|
64
|
+
CREATE TABLE IF NOT EXISTS logs (
|
|
65
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
66
|
+
timestamp TEXT,
|
|
67
|
+
level TEXT,
|
|
68
|
+
module TEXT,
|
|
69
|
+
filename TEXT,
|
|
70
|
+
line_number INTEGER,
|
|
71
|
+
message TEXT
|
|
72
|
+
)
|
|
73
|
+
''')
|
|
74
|
+
conn.commit()
|
|
75
|
+
|
|
76
|
+
def emit(self, record):
|
|
77
|
+
"""
|
|
78
|
+
Inserts a new log record into the database.
|
|
79
|
+
"""
|
|
80
|
+
try:
|
|
81
|
+
# Ensure the record is formatted to populate all fields
|
|
82
|
+
self.format(record)
|
|
83
|
+
# Format the timestamp using the formatter
|
|
84
|
+
timestamp = self.formatter.formatTime(record)
|
|
85
|
+
#with self.lock:
|
|
86
|
+
with sqlite3.connect(self.db_path, timeout=10) as conn:
|
|
87
|
+
cursor = conn.cursor()
|
|
88
|
+
cursor.execute('''
|
|
89
|
+
INSERT INTO logs (timestamp, level, module, filename, line_number, message)
|
|
90
|
+
VALUES (?, ?, ?, ?, ?, ?)
|
|
91
|
+
''', (timestamp, record.levelname, record.module,
|
|
92
|
+
record.filename, record.lineno, record.getMessage()))
|
|
93
|
+
conn.commit()
|
|
94
|
+
except Exception:
|
|
95
|
+
self.handleError(record)
|
|
96
|
+
def __del__(self):
|
|
97
|
+
"""
|
|
98
|
+
Destructor to perform a WAL checkpoint when the handler is destroyed.
|
|
99
|
+
"""
|
|
100
|
+
try:
|
|
101
|
+
with sqlite3.connect(self.db_path) as conn:
|
|
102
|
+
conn.execute('PRAGMA wal_checkpoint;')
|
|
103
|
+
conn.commit()
|
|
104
|
+
except Exception as e:
|
|
105
|
+
print(f"Error during WAL checkpoint: {e}")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
"""_summary_
|
|
109
|
+
|
|
110
|
+
usage:
|
|
111
|
+
|
|
112
|
+
import logging
|
|
113
|
+
|
|
114
|
+
# Configure the logger
|
|
115
|
+
logger = logging.getLogger(__name__)
|
|
116
|
+
logger.setLevel(logging.DEBUG)
|
|
117
|
+
|
|
118
|
+
# Create the SQLite logging handler
|
|
119
|
+
sqlite_handler = SQLiteHandler('application_logs.db')
|
|
120
|
+
sqlite_handler.setLevel(logging.DEBUG)
|
|
121
|
+
|
|
122
|
+
# Add the handler to the logger
|
|
123
|
+
logger.addHandler(sqlite_handler)
|
|
124
|
+
|
|
125
|
+
# Register the handler's close method with the logging shutdown
|
|
126
|
+
logging.shutdown = sqlite_handler.close
|
|
127
|
+
|
|
128
|
+
# Log messages
|
|
129
|
+
logger.info('This is an info message.')
|
|
130
|
+
logger.error('This is an error message.')
|
|
131
|
+
|
|
132
|
+
# When the application is terminating
|
|
133
|
+
logging.shutdown()
|
|
134
|
+
|
|
135
|
+
"""
|