mcp-win-stdio-excel-db 0.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_win_stdio_excel_db-0.2.5/.gitignore +63 -0
- mcp_win_stdio_excel_db-0.2.5/PKG-INFO +25 -0
- mcp_win_stdio_excel_db-0.2.5/README.md +10 -0
- mcp_win_stdio_excel_db-0.2.5/pyproject.toml +31 -0
- mcp_win_stdio_excel_db-0.2.5/src/mcp_win_stdio/excel_db/__init__.py +6 -0
- mcp_win_stdio_excel_db-0.2.5/src/mcp_win_stdio/excel_db/__main__.py +5 -0
- mcp_win_stdio_excel_db-0.2.5/src/mcp_win_stdio/excel_db/auditor.py +402 -0
- mcp_win_stdio_excel_db-0.2.5/src/mcp_win_stdio/excel_db/cli.py +24 -0
- mcp_win_stdio_excel_db-0.2.5/src/mcp_win_stdio/excel_db/engine.py +129 -0
- mcp_win_stdio_excel_db-0.2.5/src/mcp_win_stdio/excel_db/guide.py +67 -0
- mcp_win_stdio_excel_db-0.2.5/src/mcp_win_stdio/excel_db/migrator.py +301 -0
- mcp_win_stdio_excel_db-0.2.5/src/mcp_win_stdio/excel_db/server.py +430 -0
- mcp_win_stdio_excel_db-0.2.5/src/mcp_win_stdio/excel_db/stream.py +401 -0
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.py[cod]
|
|
3
|
+
*$py.class
|
|
4
|
+
*.so
|
|
5
|
+
.Python
|
|
6
|
+
build/
|
|
7
|
+
develop-eggs/
|
|
8
|
+
dist/
|
|
9
|
+
downloads/
|
|
10
|
+
eggs/
|
|
11
|
+
.eggs/
|
|
12
|
+
lib/
|
|
13
|
+
lib64/
|
|
14
|
+
parts/
|
|
15
|
+
sdist/
|
|
16
|
+
var/
|
|
17
|
+
wheels/
|
|
18
|
+
share/python-wheels/
|
|
19
|
+
*.egg-info/
|
|
20
|
+
.installed.cfg
|
|
21
|
+
*.egg
|
|
22
|
+
MANIFEST
|
|
23
|
+
|
|
24
|
+
*.manifest
|
|
25
|
+
*.spec
|
|
26
|
+
api.txt
|
|
27
|
+
pypi-api.txt
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
pip-log.txt
|
|
31
|
+
pip-delete-this-directory.txt
|
|
32
|
+
pypi api.txt
|
|
33
|
+
htmlcov/
|
|
34
|
+
.tox/
|
|
35
|
+
.nox/
|
|
36
|
+
.coverage
|
|
37
|
+
.coverage.*
|
|
38
|
+
.cache
|
|
39
|
+
nosetests.xml
|
|
40
|
+
coverage.xml
|
|
41
|
+
*.cover
|
|
42
|
+
*.py,cover
|
|
43
|
+
.hypothesis/
|
|
44
|
+
.pytest_cache/
|
|
45
|
+
cover/
|
|
46
|
+
|
|
47
|
+
*.mo
|
|
48
|
+
*.pot
|
|
49
|
+
|
|
50
|
+
.env
|
|
51
|
+
.venv
|
|
52
|
+
env/
|
|
53
|
+
venv/
|
|
54
|
+
ENV/
|
|
55
|
+
env.bak/
|
|
56
|
+
venv.bak/
|
|
57
|
+
|
|
58
|
+
.idea/
|
|
59
|
+
.vscode/
|
|
60
|
+
*.swp
|
|
61
|
+
*.swo
|
|
62
|
+
|
|
63
|
+
*.log
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: mcp-win-stdio-excel-db
|
|
3
|
+
Version: 0.2.5
|
|
4
|
+
Summary: High-Performance Excel & Database Power Pipeline MCP Server: Zero-Context Streaming, Cross-Source Joins, and Diff Auditing.
|
|
5
|
+
Author: Mohan Kumar Indala
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Keywords: analytics,database,duckdb,etl,excel,mcp,pipeline,sql,stdio,streaming
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Requires-Dist: mcp-win-stdio>=0.2.5
|
|
10
|
+
Requires-Dist: mcp>=1.2.0
|
|
11
|
+
Requires-Dist: openpyxl>=3.1.0
|
|
12
|
+
Requires-Dist: pandas>=2.0.0
|
|
13
|
+
Requires-Dist: sqlalchemy>=2.0.0
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
|
|
16
|
+
# mcp-win-stdio-excel-db
|
|
17
|
+
|
|
18
|
+
High-Performance Power Pipeline MCP Server combining SQL Databases and Excel Spreadsheets.
|
|
19
|
+
|
|
20
|
+
## Features
|
|
21
|
+
* **Direct DB to Excel Streaming (`db_to_excel_stream`)**: Zero-context LLM token cost for massive queries.
|
|
22
|
+
* **Bulk Excel to DB Upsert (`excel_to_db_upsert`)**: Stream spreadsheets directly into database tables with type inference and chunking.
|
|
23
|
+
* **Cross-Source Join Engine (`query_unified_sources`)**: Run unified SQL joining SQL DB tables and `.xlsx` sheets in memory.
|
|
24
|
+
* **Reconciliation Auditor (`reconcile_db_vs_excel`)**: Identify data drift and discrepancies between database records and spreadsheets.
|
|
25
|
+
* **Template Populator (`db_to_excel_template`)**: Inject database query results into styled corporate Excel workbooks.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
# mcp-win-stdio-excel-db
|
|
2
|
+
|
|
3
|
+
High-Performance Power Pipeline MCP Server combining SQL Databases and Excel Spreadsheets.
|
|
4
|
+
|
|
5
|
+
## Features
|
|
6
|
+
* **Direct DB to Excel Streaming (`db_to_excel_stream`)**: Zero-context LLM token cost for massive queries.
|
|
7
|
+
* **Bulk Excel to DB Upsert (`excel_to_db_upsert`)**: Stream spreadsheets directly into database tables with type inference and chunking.
|
|
8
|
+
* **Cross-Source Join Engine (`query_unified_sources`)**: Run unified SQL joining SQL DB tables and `.xlsx` sheets in memory.
|
|
9
|
+
* **Reconciliation Auditor (`reconcile_db_vs_excel`)**: Identify data drift and discrepancies between database records and spreadsheets.
|
|
10
|
+
* **Template Populator (`db_to_excel_template`)**: Inject database query results into styled corporate Excel workbooks.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "mcp-win-stdio-excel-db"
|
|
7
|
+
version = "0.2.5"
|
|
8
|
+
description = "High-Performance Excel & Database Power Pipeline MCP Server: Zero-Context Streaming, Cross-Source Joins, and Diff Auditing."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
authors = [
|
|
13
|
+
{ name = "Mohan Kumar Indala" }
|
|
14
|
+
]
|
|
15
|
+
keywords = ["mcp", "excel", "database", "pipeline", "etl", "duckdb", "sql", "streaming", "analytics", "stdio"]
|
|
16
|
+
|
|
17
|
+
dependencies = [
|
|
18
|
+
"mcp-win-stdio>=0.2.5",
|
|
19
|
+
"mcp>=1.2.0",
|
|
20
|
+
"pandas>=2.0.0",
|
|
21
|
+
"openpyxl>=3.1.0",
|
|
22
|
+
"sqlalchemy>=2.0.0",
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
[project.scripts]
|
|
26
|
+
mcp-win-stdio-excel-db = "mcp_win_stdio.excel_db.cli:main"
|
|
27
|
+
mws-excel-db = "mcp_win_stdio.excel_db.cli:main"
|
|
28
|
+
|
|
29
|
+
[tool.hatch.build.targets.wheel]
|
|
30
|
+
packages = ["src/mcp_win_stdio"]
|
|
31
|
+
|
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Automated Data Auditor and Reconciliation Engine for comparing Database tables
|
|
4
|
+
and Excel spreadsheets with tolerance, column mappings, and multi-tab diff reports.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import math
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any, Dict, List, Optional, Union
|
|
11
|
+
|
|
12
|
+
import pandas as pd
|
|
13
|
+
from openpyxl import Workbook
|
|
14
|
+
from openpyxl.styles import Border, Font, PatternFill, Side
|
|
15
|
+
from openpyxl.utils import get_column_letter
|
|
16
|
+
from sqlalchemy import create_engine, text
|
|
17
|
+
|
|
18
|
+
from mcp_win_stdio.excel_db.stream import resolve_sqlalchemy_url
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _clean_val(v: Any) -> Any:
|
|
22
|
+
"""Format primitives safely for JSON serialization and comparison."""
|
|
23
|
+
if pd.isna(v):
|
|
24
|
+
return None
|
|
25
|
+
elif hasattr(v, "isoformat"):
|
|
26
|
+
return v.isoformat()
|
|
27
|
+
elif isinstance(v, (float, int)) and (math.isnan(v) or math.isinf(v)):
|
|
28
|
+
return None
|
|
29
|
+
return v
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _load_dataset(source: Union[str, Dict[str, Any]], db_resolver_func=None) -> pd.DataFrame:
|
|
33
|
+
"""
|
|
34
|
+
Load a dataset from either a file path (Excel/CSV) or a configuration dict (DB/Excel).
|
|
35
|
+
"""
|
|
36
|
+
if isinstance(source, str):
|
|
37
|
+
path = Path(source).resolve()
|
|
38
|
+
if not path.exists():
|
|
39
|
+
raise FileNotFoundError(f"Source file not found: {path}")
|
|
40
|
+
if path.suffix.lower() == ".csv":
|
|
41
|
+
df = pd.read_csv(path)
|
|
42
|
+
else:
|
|
43
|
+
df = pd.read_excel(path)
|
|
44
|
+
|
|
45
|
+
if len(df) > 0 and (df.columns.dtype == "int64" or all(str(c).isdigit() for c in df.columns)):
|
|
46
|
+
new_cols = [str(x).strip() for x in df.iloc[0]]
|
|
47
|
+
df = df.iloc[1:].reset_index(drop=True)
|
|
48
|
+
df.columns = new_cols
|
|
49
|
+
return df
|
|
50
|
+
|
|
51
|
+
if not isinstance(source, dict):
|
|
52
|
+
raise ValueError(f"Invalid source specification: {source}")
|
|
53
|
+
|
|
54
|
+
s_type = source.get("type", "excel").lower()
|
|
55
|
+
|
|
56
|
+
if s_type == "db":
|
|
57
|
+
conn_id = source.get("connection")
|
|
58
|
+
raw_sql = source.get("query")
|
|
59
|
+
if not raw_sql:
|
|
60
|
+
raise ValueError("Database source must specify 'query'.")
|
|
61
|
+
db_url = db_resolver_func(conn_id) if db_resolver_func else conn_id
|
|
62
|
+
resolved_url = resolve_sqlalchemy_url(db_url)
|
|
63
|
+
engine = create_engine(resolved_url)
|
|
64
|
+
with engine.connect() as conn:
|
|
65
|
+
return pd.read_sql_query(text(raw_sql), conn)
|
|
66
|
+
|
|
67
|
+
elif s_type in ("excel", "xlsx"):
|
|
68
|
+
path = Path(source.get("path", "")).resolve()
|
|
69
|
+
if not path.exists():
|
|
70
|
+
raise FileNotFoundError(f"Source Excel file not found: {path}")
|
|
71
|
+
sheet = source.get("sheet", 0)
|
|
72
|
+
return pd.read_excel(path, sheet_name=sheet)
|
|
73
|
+
|
|
74
|
+
elif s_type == "csv":
|
|
75
|
+
path = Path(source.get("path", "")).resolve()
|
|
76
|
+
if not path.exists():
|
|
77
|
+
raise FileNotFoundError(f"Source CSV file not found: {path}")
|
|
78
|
+
return pd.read_csv(path)
|
|
79
|
+
|
|
80
|
+
else:
|
|
81
|
+
raise ValueError(f"Unsupported source type: '{s_type}'.")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def compare_master_datasets(
|
|
85
|
+
source_a: Union[str, Dict[str, Any]],
|
|
86
|
+
source_b: Union[str, Dict[str, Any]],
|
|
87
|
+
key_columns: List[str],
|
|
88
|
+
column_mapping: Optional[Dict[str, str]] = None,
|
|
89
|
+
compare_columns: Optional[List[str]] = None,
|
|
90
|
+
numeric_tolerance: float = 0.001,
|
|
91
|
+
ignore_whitespace_case: bool = True,
|
|
92
|
+
output_report_path: Optional[str] = None,
|
|
93
|
+
db_resolver_func=None,
|
|
94
|
+
) -> Dict[str, Any]:
|
|
95
|
+
"""
|
|
96
|
+
Deterministically compare two master datasets (Excel vs Excel, DB vs Excel, or DB vs DB).
|
|
97
|
+
Detects new records, missing records, and field-level discrepancies with tolerance.
|
|
98
|
+
|
|
99
|
+
Args:
|
|
100
|
+
source_a: Baseline master (file path or dict specification).
|
|
101
|
+
source_b: Comparison / New master (file path or dict specification).
|
|
102
|
+
key_columns: List of primary key column names (as named in source_a).
|
|
103
|
+
column_mapping: Optional map of source_b column names to source_a column names
|
|
104
|
+
(e.g. {"Material Number": "material_number", "Unit Rate": "sale_price"}).
|
|
105
|
+
compare_columns: Optional subset of source_a column names to compare.
|
|
106
|
+
numeric_tolerance: Maximum absolute difference considered equal for numbers (default 0.001).
|
|
107
|
+
ignore_whitespace_case: If True, trims whitespace and ignores case for string comparison.
|
|
108
|
+
output_report_path: Optional path to save a styled multi-tab Excel audit report.
|
|
109
|
+
db_resolver_func: Function to resolve connection names if DB sources are used.
|
|
110
|
+
"""
|
|
111
|
+
t_start = datetime.now()
|
|
112
|
+
|
|
113
|
+
# 1. Load DataFrames
|
|
114
|
+
df_a = _load_dataset(source_a, db_resolver_func)
|
|
115
|
+
df_b = _load_dataset(source_b, db_resolver_func)
|
|
116
|
+
|
|
117
|
+
# 2. Normalize and apply column mapping to source_b
|
|
118
|
+
df_a.columns = [str(c).strip() for c in df_a.columns]
|
|
119
|
+
df_b.columns = [str(c).strip() for c in df_b.columns]
|
|
120
|
+
|
|
121
|
+
if column_mapping:
|
|
122
|
+
# Rename source_b columns to match source_a naming convention
|
|
123
|
+
rename_dict = {}
|
|
124
|
+
for b_col, a_col in column_mapping.items():
|
|
125
|
+
b_clean = str(b_col).strip()
|
|
126
|
+
a_clean = str(a_col).strip()
|
|
127
|
+
if b_clean in df_b.columns:
|
|
128
|
+
rename_dict[b_clean] = a_clean
|
|
129
|
+
df_b = df_b.rename(columns=rename_dict)
|
|
130
|
+
|
|
131
|
+
# 3. Validate Key Columns
|
|
132
|
+
for k in key_columns:
|
|
133
|
+
if k not in df_a.columns:
|
|
134
|
+
raise ValueError(f"Key column '{k}' not found in Source A. Available: {list(df_a.columns)}")
|
|
135
|
+
if k not in df_b.columns:
|
|
136
|
+
raise ValueError(f"Key column '{k}' not found in Source B. Available: {list(df_b.columns)}")
|
|
137
|
+
|
|
138
|
+
# 4. Determine columns to compare
|
|
139
|
+
if not compare_columns:
|
|
140
|
+
shared_cols = [c for c in df_a.columns if c in df_b.columns and c not in key_columns]
|
|
141
|
+
else:
|
|
142
|
+
shared_cols = [c for c in compare_columns if c in df_a.columns and c in df_b.columns and c not in key_columns]
|
|
143
|
+
|
|
144
|
+
# Clean key column values to strings for robust joining
|
|
145
|
+
for k in key_columns:
|
|
146
|
+
df_a[k] = df_a[k].astype(str).str.strip()
|
|
147
|
+
df_b[k] = df_b[k].astype(str).str.strip()
|
|
148
|
+
|
|
149
|
+
# 5. Full Outer Join on Key Columns
|
|
150
|
+
merged = pd.merge(df_a, df_b, on=key_columns, how="outer", suffixes=("_a", "_b"), indicator=True)
|
|
151
|
+
|
|
152
|
+
only_in_a = merged[merged["_merge"] == "left_only"].copy()
|
|
153
|
+
only_in_b = merged[merged["_merge"] == "right_only"].copy()
|
|
154
|
+
in_both = merged[merged["_merge"] == "both"].copy()
|
|
155
|
+
|
|
156
|
+
# 6. Detailed Field Comparison
|
|
157
|
+
discrepancies = []
|
|
158
|
+
col_discrepancy_counts: Dict[str, int] = {c: 0 for c in shared_cols}
|
|
159
|
+
|
|
160
|
+
for _, row in in_both.iterrows():
|
|
161
|
+
row_diffs = {}
|
|
162
|
+
for col in shared_cols:
|
|
163
|
+
val_a = row.get(f"{col}_a")
|
|
164
|
+
val_b = row.get(f"{col}_b")
|
|
165
|
+
|
|
166
|
+
# Handle nulls
|
|
167
|
+
is_na_a = pd.isna(val_a)
|
|
168
|
+
is_na_b = pd.isna(val_b)
|
|
169
|
+
if is_na_a and is_na_b:
|
|
170
|
+
continue
|
|
171
|
+
if is_na_a != is_na_b:
|
|
172
|
+
row_diffs[col] = {
|
|
173
|
+
"source_a": _clean_val(val_a),
|
|
174
|
+
"source_b": _clean_val(val_b),
|
|
175
|
+
"delta": None,
|
|
176
|
+
"type": "null_mismatch",
|
|
177
|
+
}
|
|
178
|
+
col_discrepancy_counts[col] += 1
|
|
179
|
+
continue
|
|
180
|
+
|
|
181
|
+
# Numeric comparison
|
|
182
|
+
is_num_a = isinstance(val_a, (int, float)) and not isinstance(val_a, bool)
|
|
183
|
+
is_num_b = isinstance(val_b, (int, float)) and not isinstance(val_b, bool)
|
|
184
|
+
|
|
185
|
+
if is_num_a and is_num_b:
|
|
186
|
+
delta = float(val_b) - float(val_a)
|
|
187
|
+
if abs(delta) > numeric_tolerance:
|
|
188
|
+
pct = round((delta / float(val_a) * 100), 2) if float(val_a) != 0 else None
|
|
189
|
+
row_diffs[col] = {
|
|
190
|
+
"source_a": _clean_val(val_a),
|
|
191
|
+
"source_b": _clean_val(val_b),
|
|
192
|
+
"delta": round(delta, 4),
|
|
193
|
+
"pct_change": pct,
|
|
194
|
+
"type": "numeric_diff",
|
|
195
|
+
}
|
|
196
|
+
col_discrepancy_counts[col] += 1
|
|
197
|
+
else:
|
|
198
|
+
# String / Boolean comparison
|
|
199
|
+
s_a = str(val_a).strip()
|
|
200
|
+
s_b = str(val_b).strip()
|
|
201
|
+
if ignore_whitespace_case:
|
|
202
|
+
s_a_cmp = s_a.lower()
|
|
203
|
+
s_b_cmp = s_b.lower()
|
|
204
|
+
else:
|
|
205
|
+
s_a_cmp = s_a
|
|
206
|
+
s_b_cmp = s_b
|
|
207
|
+
|
|
208
|
+
if s_a_cmp != s_b_cmp:
|
|
209
|
+
row_diffs[col] = {
|
|
210
|
+
"source_a": _clean_val(val_a),
|
|
211
|
+
"source_b": _clean_val(val_b),
|
|
212
|
+
"delta": None,
|
|
213
|
+
"type": "text_diff",
|
|
214
|
+
}
|
|
215
|
+
col_discrepancy_counts[col] += 1
|
|
216
|
+
|
|
217
|
+
if row_diffs:
|
|
218
|
+
keys_dict = {k: row[k] for k in key_columns}
|
|
219
|
+
discrepancies.append({"keys": keys_dict, "differences": row_diffs})
|
|
220
|
+
|
|
221
|
+
identical_records_count = len(in_both) - len(discrepancies)
|
|
222
|
+
|
|
223
|
+
# 7. Generate Multi-Tab Styled Audit Report
|
|
224
|
+
report_file_path = None
|
|
225
|
+
if output_report_path:
|
|
226
|
+
out_p = Path(output_report_path).resolve()
|
|
227
|
+
out_p.parent.mkdir(parents=True, exist_ok=True)
|
|
228
|
+
wb = Workbook()
|
|
229
|
+
|
|
230
|
+
# Styles
|
|
231
|
+
font_title = Font(name="Calibri", size=14, bold=True, color="1F497D")
|
|
232
|
+
font_header = Font(name="Calibri", size=11, bold=True, color="FFFFFF")
|
|
233
|
+
font_bold = Font(name="Calibri", size=11, bold=True)
|
|
234
|
+
fill_header_navy = PatternFill(start_color="1F497D", end_color="1F497D", fill_type="solid")
|
|
235
|
+
fill_header_green = PatternFill(start_color="27AE60", end_color="27AE60", fill_type="solid")
|
|
236
|
+
fill_header_orange = PatternFill(start_color="E67E22", end_color="E67E22", fill_type="solid")
|
|
237
|
+
fill_diff_yellow = PatternFill(start_color="FFF3CD", end_color="FFF3CD", fill_type="solid")
|
|
238
|
+
fill_diff_red = PatternFill(start_color="F8D7DA", end_color="F8D7DA", fill_type="solid")
|
|
239
|
+
border_thin = Border(
|
|
240
|
+
left=Side(style="thin", color="D3D3D3"),
|
|
241
|
+
right=Side(style="thin", color="D3D3D3"),
|
|
242
|
+
top=Side(style="thin", color="D3D3D3"),
|
|
243
|
+
bottom=Side(style="thin", color="D3D3D3"),
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
# Tab 1: Overview
|
|
247
|
+
ws_overview = wb.active
|
|
248
|
+
ws_overview.title = "Audit Overview"
|
|
249
|
+
ws_overview.append(["Master Data Reconciliation Audit Report", ""])
|
|
250
|
+
ws_overview["A1"].font = font_title
|
|
251
|
+
ws_overview.append([])
|
|
252
|
+
ws_overview.append(["Metric", "Count / Value"])
|
|
253
|
+
ws_overview["A3"].font = font_header
|
|
254
|
+
ws_overview["B3"].font = font_header
|
|
255
|
+
ws_overview["A3"].fill = fill_header_navy
|
|
256
|
+
ws_overview["B3"].fill = fill_header_navy
|
|
257
|
+
|
|
258
|
+
metrics = [
|
|
259
|
+
("Audit Execution Timestamp", datetime.now().isoformat()),
|
|
260
|
+
("Source A Total Records", len(df_a)),
|
|
261
|
+
("Source B Total Records", len(df_b)),
|
|
262
|
+
("Matched Primary Keys", len(in_both)),
|
|
263
|
+
("Identical Records (Within Tolerance)", identical_records_count),
|
|
264
|
+
("Records with Value Discrepancies", len(discrepancies)),
|
|
265
|
+
("New Records (Source B Only)", len(only_in_b)),
|
|
266
|
+
("Missing Records (Source A Only)", len(only_in_a)),
|
|
267
|
+
("Numeric Tolerance Applied", numeric_tolerance),
|
|
268
|
+
]
|
|
269
|
+
for m_name, m_val in metrics:
|
|
270
|
+
ws_overview.append([m_name, m_val])
|
|
271
|
+
|
|
272
|
+
ws_overview.append([])
|
|
273
|
+
ws_overview.append(["Field Name", "Mismatched Rows Count"])
|
|
274
|
+
ws_overview["A15"].font = font_header
|
|
275
|
+
ws_overview["B15"].font = font_header
|
|
276
|
+
ws_overview["A15"].fill = fill_header_navy
|
|
277
|
+
ws_overview["B15"].fill = fill_header_navy
|
|
278
|
+
for fld, cnt in col_discrepancy_counts.items():
|
|
279
|
+
ws_overview.append([fld, cnt])
|
|
280
|
+
|
|
281
|
+
# Tab 2: Discrepancies
|
|
282
|
+
if discrepancies:
|
|
283
|
+
ws_diff = wb.create_sheet("Field Differences")
|
|
284
|
+
diff_headers = list(key_columns) + [
|
|
285
|
+
"Field",
|
|
286
|
+
"Source A (Baseline)",
|
|
287
|
+
"Source B (New)",
|
|
288
|
+
"Delta",
|
|
289
|
+
"% Change",
|
|
290
|
+
"Diff Type",
|
|
291
|
+
]
|
|
292
|
+
ws_diff.append(diff_headers)
|
|
293
|
+
for c_idx in range(1, len(diff_headers) + 1):
|
|
294
|
+
cell = ws_diff.cell(row=1, column=c_idx)
|
|
295
|
+
cell.font = font_header
|
|
296
|
+
cell.fill = fill_header_navy
|
|
297
|
+
|
|
298
|
+
for item in discrepancies:
|
|
299
|
+
k_vals = [item["keys"][k] for k in key_columns]
|
|
300
|
+
for fld, diff_info in item["differences"].items():
|
|
301
|
+
row_vals = k_vals + [
|
|
302
|
+
fld,
|
|
303
|
+
str(diff_info["source_a"]),
|
|
304
|
+
str(diff_info["source_b"]),
|
|
305
|
+
diff_info.get("delta"),
|
|
306
|
+
diff_info.get("pct_change"),
|
|
307
|
+
diff_info.get("type"),
|
|
308
|
+
]
|
|
309
|
+
ws_diff.append(row_vals)
|
|
310
|
+
# highlight
|
|
311
|
+
curr_row = ws_diff.max_row
|
|
312
|
+
for c_idx in range(1, len(row_vals) + 1):
|
|
313
|
+
ws_diff.cell(row=curr_row, column=c_idx).fill = fill_diff_yellow
|
|
314
|
+
|
|
315
|
+
# Tab 3: New Records in B
|
|
316
|
+
if len(only_in_b) > 0:
|
|
317
|
+
ws_new = wb.create_sheet("New Records")
|
|
318
|
+
b_cols = [c for c in df_b.columns if not c.endswith("_a")]
|
|
319
|
+
clean_b_cols = [c.replace("_b", "") for c in b_cols]
|
|
320
|
+
ws_new.append(clean_b_cols)
|
|
321
|
+
for c_idx in range(1, len(clean_b_cols) + 1):
|
|
322
|
+
cell = ws_new.cell(row=1, column=c_idx)
|
|
323
|
+
cell.font = font_header
|
|
324
|
+
cell.fill = fill_header_green
|
|
325
|
+
|
|
326
|
+
for _, row in only_in_b.iterrows():
|
|
327
|
+
ws_new.append([_clean_val(row.get(f"{c}_b", row.get(c))) for c in clean_b_cols])
|
|
328
|
+
|
|
329
|
+
# Tab 4: Missing Records in B
|
|
330
|
+
if len(only_in_a) > 0:
|
|
331
|
+
ws_miss = wb.create_sheet("Missing Records")
|
|
332
|
+
a_cols = [c for c in df_a.columns if not c.endswith("_b")]
|
|
333
|
+
clean_a_cols = [c.replace("_a", "") for c in a_cols]
|
|
334
|
+
ws_miss.append(clean_a_cols)
|
|
335
|
+
for c_idx in range(1, len(clean_a_cols) + 1):
|
|
336
|
+
cell = ws_miss.cell(row=1, column=c_idx)
|
|
337
|
+
cell.font = font_header
|
|
338
|
+
cell.fill = fill_header_orange
|
|
339
|
+
|
|
340
|
+
for _, row in only_in_a.iterrows():
|
|
341
|
+
ws_miss.append([_clean_val(row.get(f"{c}_a", row.get(c))) for c in clean_a_cols])
|
|
342
|
+
|
|
343
|
+
# Auto-fit columns across all sheets
|
|
344
|
+
for sheet in wb.worksheets:
|
|
345
|
+
for col in sheet.columns:
|
|
346
|
+
max_len = max(len(str(c.value or "")) for c in col)
|
|
347
|
+
col_letter = get_column_letter(col[0].column)
|
|
348
|
+
sheet.column_dimensions[col_letter].width = min(max(max_len + 3, 12), 45)
|
|
349
|
+
|
|
350
|
+
wb.save(out_p)
|
|
351
|
+
wb.close()
|
|
352
|
+
report_file_path = str(out_p)
|
|
353
|
+
|
|
354
|
+
duration = (datetime.now() - t_start).total_seconds()
|
|
355
|
+
|
|
356
|
+
return {
|
|
357
|
+
"status": "success",
|
|
358
|
+
"summary": {
|
|
359
|
+
"source_a_records": len(df_a),
|
|
360
|
+
"source_b_records": len(df_b),
|
|
361
|
+
"matched_keys": len(in_both),
|
|
362
|
+
"matching_records": identical_records_count,
|
|
363
|
+
"modified_records": len(discrepancies),
|
|
364
|
+
"new_records_in_b": len(only_in_b),
|
|
365
|
+
"missing_records_in_b": len(only_in_a),
|
|
366
|
+
},
|
|
367
|
+
"total_source_a_records": len(df_a),
|
|
368
|
+
"total_source_b_records": len(df_b),
|
|
369
|
+
"matched_primary_keys": len(in_both),
|
|
370
|
+
"identical_records_count": identical_records_count,
|
|
371
|
+
"records_with_discrepancies_count": len(discrepancies),
|
|
372
|
+
"new_records_in_source_b_count": len(only_in_b),
|
|
373
|
+
"missing_records_in_source_b_count": len(only_in_a),
|
|
374
|
+
"column_discrepancy_counts": col_discrepancy_counts,
|
|
375
|
+
"sample_discrepancies": discrepancies[:10],
|
|
376
|
+
"audit_report_file": report_file_path,
|
|
377
|
+
"duration_seconds": round(duration, 3),
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def reconcile_db_vs_excel(
|
|
382
|
+
sql_query: str,
|
|
383
|
+
excel_path: str,
|
|
384
|
+
key_columns: List[str],
|
|
385
|
+
db_url_or_config: Union[str, Dict[str, Any]],
|
|
386
|
+
sheet_name: Optional[Union[str, int]] = 0,
|
|
387
|
+
compare_columns: Optional[List[str]] = None,
|
|
388
|
+
output_report_path: Optional[str] = None,
|
|
389
|
+
) -> Dict[str, Any]:
|
|
390
|
+
"""
|
|
391
|
+
Backwards-compatible wrapper delegating to compare_master_datasets.
|
|
392
|
+
"""
|
|
393
|
+
db_source = {"type": "db", "connection": db_url_or_config, "query": sql_query}
|
|
394
|
+
excel_source = {"type": "excel", "path": excel_path, "sheet": sheet_name}
|
|
395
|
+
return compare_master_datasets(
|
|
396
|
+
source_a=db_source,
|
|
397
|
+
source_b=excel_source,
|
|
398
|
+
key_columns=key_columns,
|
|
399
|
+
compare_columns=compare_columns,
|
|
400
|
+
output_report_path=output_report_path,
|
|
401
|
+
db_resolver_func=lambda c: c,
|
|
402
|
+
)
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
CLI entry point for mcp-win-stdio-excel-db.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
|
|
8
|
+
from mcp_win_stdio.excel_db.server import mcp
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def main():
|
|
12
|
+
parser = argparse.ArgumentParser(description="mcp-win-stdio-excel-db MCP Server")
|
|
13
|
+
parser.add_argument("--transport", default="stdio", choices=["stdio", "sse"], help="MCP transport")
|
|
14
|
+
parser.add_argument("--port", type=int, default=8001, help="Port for SSE transport")
|
|
15
|
+
args = parser.parse_args()
|
|
16
|
+
|
|
17
|
+
if args.transport == "sse":
|
|
18
|
+
mcp.run(transport="sse", port=args.port)
|
|
19
|
+
else:
|
|
20
|
+
mcp.run(transport="stdio")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
if __name__ == "__main__":
|
|
24
|
+
main()
|