mcp-win-stdio-excel-db 0.2.5__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_win_stdio/excel_db/__init__.py +6 -0
- mcp_win_stdio/excel_db/__main__.py +5 -0
- mcp_win_stdio/excel_db/auditor.py +402 -0
- mcp_win_stdio/excel_db/cli.py +24 -0
- mcp_win_stdio/excel_db/engine.py +129 -0
- mcp_win_stdio/excel_db/guide.py +67 -0
- mcp_win_stdio/excel_db/migrator.py +301 -0
- mcp_win_stdio/excel_db/server.py +430 -0
- mcp_win_stdio/excel_db/stream.py +401 -0
- mcp_win_stdio_excel_db-0.2.5.dist-info/METADATA +25 -0
- mcp_win_stdio_excel_db-0.2.5.dist-info/RECORD +13 -0
- mcp_win_stdio_excel_db-0.2.5.dist-info/WHEEL +4 -0
- mcp_win_stdio_excel_db-0.2.5.dist-info/entry_points.txt +3 -0
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Automated Data Auditor and Reconciliation Engine for comparing Database tables
|
|
4
|
+
and Excel spreadsheets with tolerance, column mappings, and multi-tab diff reports.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import math
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any, Dict, List, Optional, Union
|
|
11
|
+
|
|
12
|
+
import pandas as pd
|
|
13
|
+
from openpyxl import Workbook
|
|
14
|
+
from openpyxl.styles import Border, Font, PatternFill, Side
|
|
15
|
+
from openpyxl.utils import get_column_letter
|
|
16
|
+
from sqlalchemy import create_engine, text
|
|
17
|
+
|
|
18
|
+
from mcp_win_stdio.excel_db.stream import resolve_sqlalchemy_url
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _clean_val(v: Any) -> Any:
|
|
22
|
+
"""Format primitives safely for JSON serialization and comparison."""
|
|
23
|
+
if pd.isna(v):
|
|
24
|
+
return None
|
|
25
|
+
elif hasattr(v, "isoformat"):
|
|
26
|
+
return v.isoformat()
|
|
27
|
+
elif isinstance(v, (float, int)) and (math.isnan(v) or math.isinf(v)):
|
|
28
|
+
return None
|
|
29
|
+
return v
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _load_dataset(source: Union[str, Dict[str, Any]], db_resolver_func=None) -> pd.DataFrame:
|
|
33
|
+
"""
|
|
34
|
+
Load a dataset from either a file path (Excel/CSV) or a configuration dict (DB/Excel).
|
|
35
|
+
"""
|
|
36
|
+
if isinstance(source, str):
|
|
37
|
+
path = Path(source).resolve()
|
|
38
|
+
if not path.exists():
|
|
39
|
+
raise FileNotFoundError(f"Source file not found: {path}")
|
|
40
|
+
if path.suffix.lower() == ".csv":
|
|
41
|
+
df = pd.read_csv(path)
|
|
42
|
+
else:
|
|
43
|
+
df = pd.read_excel(path)
|
|
44
|
+
|
|
45
|
+
if len(df) > 0 and (df.columns.dtype == "int64" or all(str(c).isdigit() for c in df.columns)):
|
|
46
|
+
new_cols = [str(x).strip() for x in df.iloc[0]]
|
|
47
|
+
df = df.iloc[1:].reset_index(drop=True)
|
|
48
|
+
df.columns = new_cols
|
|
49
|
+
return df
|
|
50
|
+
|
|
51
|
+
if not isinstance(source, dict):
|
|
52
|
+
raise ValueError(f"Invalid source specification: {source}")
|
|
53
|
+
|
|
54
|
+
s_type = source.get("type", "excel").lower()
|
|
55
|
+
|
|
56
|
+
if s_type == "db":
|
|
57
|
+
conn_id = source.get("connection")
|
|
58
|
+
raw_sql = source.get("query")
|
|
59
|
+
if not raw_sql:
|
|
60
|
+
raise ValueError("Database source must specify 'query'.")
|
|
61
|
+
db_url = db_resolver_func(conn_id) if db_resolver_func else conn_id
|
|
62
|
+
resolved_url = resolve_sqlalchemy_url(db_url)
|
|
63
|
+
engine = create_engine(resolved_url)
|
|
64
|
+
with engine.connect() as conn:
|
|
65
|
+
return pd.read_sql_query(text(raw_sql), conn)
|
|
66
|
+
|
|
67
|
+
elif s_type in ("excel", "xlsx"):
|
|
68
|
+
path = Path(source.get("path", "")).resolve()
|
|
69
|
+
if not path.exists():
|
|
70
|
+
raise FileNotFoundError(f"Source Excel file not found: {path}")
|
|
71
|
+
sheet = source.get("sheet", 0)
|
|
72
|
+
return pd.read_excel(path, sheet_name=sheet)
|
|
73
|
+
|
|
74
|
+
elif s_type == "csv":
|
|
75
|
+
path = Path(source.get("path", "")).resolve()
|
|
76
|
+
if not path.exists():
|
|
77
|
+
raise FileNotFoundError(f"Source CSV file not found: {path}")
|
|
78
|
+
return pd.read_csv(path)
|
|
79
|
+
|
|
80
|
+
else:
|
|
81
|
+
raise ValueError(f"Unsupported source type: '{s_type}'.")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def compare_master_datasets(
|
|
85
|
+
source_a: Union[str, Dict[str, Any]],
|
|
86
|
+
source_b: Union[str, Dict[str, Any]],
|
|
87
|
+
key_columns: List[str],
|
|
88
|
+
column_mapping: Optional[Dict[str, str]] = None,
|
|
89
|
+
compare_columns: Optional[List[str]] = None,
|
|
90
|
+
numeric_tolerance: float = 0.001,
|
|
91
|
+
ignore_whitespace_case: bool = True,
|
|
92
|
+
output_report_path: Optional[str] = None,
|
|
93
|
+
db_resolver_func=None,
|
|
94
|
+
) -> Dict[str, Any]:
|
|
95
|
+
"""
|
|
96
|
+
Deterministically compare two master datasets (Excel vs Excel, DB vs Excel, or DB vs DB).
|
|
97
|
+
Detects new records, missing records, and field-level discrepancies with tolerance.
|
|
98
|
+
|
|
99
|
+
Args:
|
|
100
|
+
source_a: Baseline master (file path or dict specification).
|
|
101
|
+
source_b: Comparison / New master (file path or dict specification).
|
|
102
|
+
key_columns: List of primary key column names (as named in source_a).
|
|
103
|
+
column_mapping: Optional map of source_b column names to source_a column names
|
|
104
|
+
(e.g. {"Material Number": "material_number", "Unit Rate": "sale_price"}).
|
|
105
|
+
compare_columns: Optional subset of source_a column names to compare.
|
|
106
|
+
numeric_tolerance: Maximum absolute difference considered equal for numbers (default 0.001).
|
|
107
|
+
ignore_whitespace_case: If True, trims whitespace and ignores case for string comparison.
|
|
108
|
+
output_report_path: Optional path to save a styled multi-tab Excel audit report.
|
|
109
|
+
db_resolver_func: Function to resolve connection names if DB sources are used.
|
|
110
|
+
"""
|
|
111
|
+
t_start = datetime.now()
|
|
112
|
+
|
|
113
|
+
# 1. Load DataFrames
|
|
114
|
+
df_a = _load_dataset(source_a, db_resolver_func)
|
|
115
|
+
df_b = _load_dataset(source_b, db_resolver_func)
|
|
116
|
+
|
|
117
|
+
# 2. Normalize and apply column mapping to source_b
|
|
118
|
+
df_a.columns = [str(c).strip() for c in df_a.columns]
|
|
119
|
+
df_b.columns = [str(c).strip() for c in df_b.columns]
|
|
120
|
+
|
|
121
|
+
if column_mapping:
|
|
122
|
+
# Rename source_b columns to match source_a naming convention
|
|
123
|
+
rename_dict = {}
|
|
124
|
+
for b_col, a_col in column_mapping.items():
|
|
125
|
+
b_clean = str(b_col).strip()
|
|
126
|
+
a_clean = str(a_col).strip()
|
|
127
|
+
if b_clean in df_b.columns:
|
|
128
|
+
rename_dict[b_clean] = a_clean
|
|
129
|
+
df_b = df_b.rename(columns=rename_dict)
|
|
130
|
+
|
|
131
|
+
# 3. Validate Key Columns
|
|
132
|
+
for k in key_columns:
|
|
133
|
+
if k not in df_a.columns:
|
|
134
|
+
raise ValueError(f"Key column '{k}' not found in Source A. Available: {list(df_a.columns)}")
|
|
135
|
+
if k not in df_b.columns:
|
|
136
|
+
raise ValueError(f"Key column '{k}' not found in Source B. Available: {list(df_b.columns)}")
|
|
137
|
+
|
|
138
|
+
# 4. Determine columns to compare
|
|
139
|
+
if not compare_columns:
|
|
140
|
+
shared_cols = [c for c in df_a.columns if c in df_b.columns and c not in key_columns]
|
|
141
|
+
else:
|
|
142
|
+
shared_cols = [c for c in compare_columns if c in df_a.columns and c in df_b.columns and c not in key_columns]
|
|
143
|
+
|
|
144
|
+
# Clean key column values to strings for robust joining
|
|
145
|
+
for k in key_columns:
|
|
146
|
+
df_a[k] = df_a[k].astype(str).str.strip()
|
|
147
|
+
df_b[k] = df_b[k].astype(str).str.strip()
|
|
148
|
+
|
|
149
|
+
# 5. Full Outer Join on Key Columns
|
|
150
|
+
merged = pd.merge(df_a, df_b, on=key_columns, how="outer", suffixes=("_a", "_b"), indicator=True)
|
|
151
|
+
|
|
152
|
+
only_in_a = merged[merged["_merge"] == "left_only"].copy()
|
|
153
|
+
only_in_b = merged[merged["_merge"] == "right_only"].copy()
|
|
154
|
+
in_both = merged[merged["_merge"] == "both"].copy()
|
|
155
|
+
|
|
156
|
+
# 6. Detailed Field Comparison
|
|
157
|
+
discrepancies = []
|
|
158
|
+
col_discrepancy_counts: Dict[str, int] = {c: 0 for c in shared_cols}
|
|
159
|
+
|
|
160
|
+
for _, row in in_both.iterrows():
|
|
161
|
+
row_diffs = {}
|
|
162
|
+
for col in shared_cols:
|
|
163
|
+
val_a = row.get(f"{col}_a")
|
|
164
|
+
val_b = row.get(f"{col}_b")
|
|
165
|
+
|
|
166
|
+
# Handle nulls
|
|
167
|
+
is_na_a = pd.isna(val_a)
|
|
168
|
+
is_na_b = pd.isna(val_b)
|
|
169
|
+
if is_na_a and is_na_b:
|
|
170
|
+
continue
|
|
171
|
+
if is_na_a != is_na_b:
|
|
172
|
+
row_diffs[col] = {
|
|
173
|
+
"source_a": _clean_val(val_a),
|
|
174
|
+
"source_b": _clean_val(val_b),
|
|
175
|
+
"delta": None,
|
|
176
|
+
"type": "null_mismatch",
|
|
177
|
+
}
|
|
178
|
+
col_discrepancy_counts[col] += 1
|
|
179
|
+
continue
|
|
180
|
+
|
|
181
|
+
# Numeric comparison
|
|
182
|
+
is_num_a = isinstance(val_a, (int, float)) and not isinstance(val_a, bool)
|
|
183
|
+
is_num_b = isinstance(val_b, (int, float)) and not isinstance(val_b, bool)
|
|
184
|
+
|
|
185
|
+
if is_num_a and is_num_b:
|
|
186
|
+
delta = float(val_b) - float(val_a)
|
|
187
|
+
if abs(delta) > numeric_tolerance:
|
|
188
|
+
pct = round((delta / float(val_a) * 100), 2) if float(val_a) != 0 else None
|
|
189
|
+
row_diffs[col] = {
|
|
190
|
+
"source_a": _clean_val(val_a),
|
|
191
|
+
"source_b": _clean_val(val_b),
|
|
192
|
+
"delta": round(delta, 4),
|
|
193
|
+
"pct_change": pct,
|
|
194
|
+
"type": "numeric_diff",
|
|
195
|
+
}
|
|
196
|
+
col_discrepancy_counts[col] += 1
|
|
197
|
+
else:
|
|
198
|
+
# String / Boolean comparison
|
|
199
|
+
s_a = str(val_a).strip()
|
|
200
|
+
s_b = str(val_b).strip()
|
|
201
|
+
if ignore_whitespace_case:
|
|
202
|
+
s_a_cmp = s_a.lower()
|
|
203
|
+
s_b_cmp = s_b.lower()
|
|
204
|
+
else:
|
|
205
|
+
s_a_cmp = s_a
|
|
206
|
+
s_b_cmp = s_b
|
|
207
|
+
|
|
208
|
+
if s_a_cmp != s_b_cmp:
|
|
209
|
+
row_diffs[col] = {
|
|
210
|
+
"source_a": _clean_val(val_a),
|
|
211
|
+
"source_b": _clean_val(val_b),
|
|
212
|
+
"delta": None,
|
|
213
|
+
"type": "text_diff",
|
|
214
|
+
}
|
|
215
|
+
col_discrepancy_counts[col] += 1
|
|
216
|
+
|
|
217
|
+
if row_diffs:
|
|
218
|
+
keys_dict = {k: row[k] for k in key_columns}
|
|
219
|
+
discrepancies.append({"keys": keys_dict, "differences": row_diffs})
|
|
220
|
+
|
|
221
|
+
identical_records_count = len(in_both) - len(discrepancies)
|
|
222
|
+
|
|
223
|
+
# 7. Generate Multi-Tab Styled Audit Report
|
|
224
|
+
report_file_path = None
|
|
225
|
+
if output_report_path:
|
|
226
|
+
out_p = Path(output_report_path).resolve()
|
|
227
|
+
out_p.parent.mkdir(parents=True, exist_ok=True)
|
|
228
|
+
wb = Workbook()
|
|
229
|
+
|
|
230
|
+
# Styles
|
|
231
|
+
font_title = Font(name="Calibri", size=14, bold=True, color="1F497D")
|
|
232
|
+
font_header = Font(name="Calibri", size=11, bold=True, color="FFFFFF")
|
|
233
|
+
font_bold = Font(name="Calibri", size=11, bold=True)
|
|
234
|
+
fill_header_navy = PatternFill(start_color="1F497D", end_color="1F497D", fill_type="solid")
|
|
235
|
+
fill_header_green = PatternFill(start_color="27AE60", end_color="27AE60", fill_type="solid")
|
|
236
|
+
fill_header_orange = PatternFill(start_color="E67E22", end_color="E67E22", fill_type="solid")
|
|
237
|
+
fill_diff_yellow = PatternFill(start_color="FFF3CD", end_color="FFF3CD", fill_type="solid")
|
|
238
|
+
fill_diff_red = PatternFill(start_color="F8D7DA", end_color="F8D7DA", fill_type="solid")
|
|
239
|
+
border_thin = Border(
|
|
240
|
+
left=Side(style="thin", color="D3D3D3"),
|
|
241
|
+
right=Side(style="thin", color="D3D3D3"),
|
|
242
|
+
top=Side(style="thin", color="D3D3D3"),
|
|
243
|
+
bottom=Side(style="thin", color="D3D3D3"),
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
# Tab 1: Overview
|
|
247
|
+
ws_overview = wb.active
|
|
248
|
+
ws_overview.title = "Audit Overview"
|
|
249
|
+
ws_overview.append(["Master Data Reconciliation Audit Report", ""])
|
|
250
|
+
ws_overview["A1"].font = font_title
|
|
251
|
+
ws_overview.append([])
|
|
252
|
+
ws_overview.append(["Metric", "Count / Value"])
|
|
253
|
+
ws_overview["A3"].font = font_header
|
|
254
|
+
ws_overview["B3"].font = font_header
|
|
255
|
+
ws_overview["A3"].fill = fill_header_navy
|
|
256
|
+
ws_overview["B3"].fill = fill_header_navy
|
|
257
|
+
|
|
258
|
+
metrics = [
|
|
259
|
+
("Audit Execution Timestamp", datetime.now().isoformat()),
|
|
260
|
+
("Source A Total Records", len(df_a)),
|
|
261
|
+
("Source B Total Records", len(df_b)),
|
|
262
|
+
("Matched Primary Keys", len(in_both)),
|
|
263
|
+
("Identical Records (Within Tolerance)", identical_records_count),
|
|
264
|
+
("Records with Value Discrepancies", len(discrepancies)),
|
|
265
|
+
("New Records (Source B Only)", len(only_in_b)),
|
|
266
|
+
("Missing Records (Source A Only)", len(only_in_a)),
|
|
267
|
+
("Numeric Tolerance Applied", numeric_tolerance),
|
|
268
|
+
]
|
|
269
|
+
for m_name, m_val in metrics:
|
|
270
|
+
ws_overview.append([m_name, m_val])
|
|
271
|
+
|
|
272
|
+
ws_overview.append([])
|
|
273
|
+
ws_overview.append(["Field Name", "Mismatched Rows Count"])
|
|
274
|
+
ws_overview["A15"].font = font_header
|
|
275
|
+
ws_overview["B15"].font = font_header
|
|
276
|
+
ws_overview["A15"].fill = fill_header_navy
|
|
277
|
+
ws_overview["B15"].fill = fill_header_navy
|
|
278
|
+
for fld, cnt in col_discrepancy_counts.items():
|
|
279
|
+
ws_overview.append([fld, cnt])
|
|
280
|
+
|
|
281
|
+
# Tab 2: Discrepancies
|
|
282
|
+
if discrepancies:
|
|
283
|
+
ws_diff = wb.create_sheet("Field Differences")
|
|
284
|
+
diff_headers = list(key_columns) + [
|
|
285
|
+
"Field",
|
|
286
|
+
"Source A (Baseline)",
|
|
287
|
+
"Source B (New)",
|
|
288
|
+
"Delta",
|
|
289
|
+
"% Change",
|
|
290
|
+
"Diff Type",
|
|
291
|
+
]
|
|
292
|
+
ws_diff.append(diff_headers)
|
|
293
|
+
for c_idx in range(1, len(diff_headers) + 1):
|
|
294
|
+
cell = ws_diff.cell(row=1, column=c_idx)
|
|
295
|
+
cell.font = font_header
|
|
296
|
+
cell.fill = fill_header_navy
|
|
297
|
+
|
|
298
|
+
for item in discrepancies:
|
|
299
|
+
k_vals = [item["keys"][k] for k in key_columns]
|
|
300
|
+
for fld, diff_info in item["differences"].items():
|
|
301
|
+
row_vals = k_vals + [
|
|
302
|
+
fld,
|
|
303
|
+
str(diff_info["source_a"]),
|
|
304
|
+
str(diff_info["source_b"]),
|
|
305
|
+
diff_info.get("delta"),
|
|
306
|
+
diff_info.get("pct_change"),
|
|
307
|
+
diff_info.get("type"),
|
|
308
|
+
]
|
|
309
|
+
ws_diff.append(row_vals)
|
|
310
|
+
# highlight
|
|
311
|
+
curr_row = ws_diff.max_row
|
|
312
|
+
for c_idx in range(1, len(row_vals) + 1):
|
|
313
|
+
ws_diff.cell(row=curr_row, column=c_idx).fill = fill_diff_yellow
|
|
314
|
+
|
|
315
|
+
# Tab 3: New Records in B
|
|
316
|
+
if len(only_in_b) > 0:
|
|
317
|
+
ws_new = wb.create_sheet("New Records")
|
|
318
|
+
b_cols = [c for c in df_b.columns if not c.endswith("_a")]
|
|
319
|
+
clean_b_cols = [c.replace("_b", "") for c in b_cols]
|
|
320
|
+
ws_new.append(clean_b_cols)
|
|
321
|
+
for c_idx in range(1, len(clean_b_cols) + 1):
|
|
322
|
+
cell = ws_new.cell(row=1, column=c_idx)
|
|
323
|
+
cell.font = font_header
|
|
324
|
+
cell.fill = fill_header_green
|
|
325
|
+
|
|
326
|
+
for _, row in only_in_b.iterrows():
|
|
327
|
+
ws_new.append([_clean_val(row.get(f"{c}_b", row.get(c))) for c in clean_b_cols])
|
|
328
|
+
|
|
329
|
+
# Tab 4: Missing Records in B
|
|
330
|
+
if len(only_in_a) > 0:
|
|
331
|
+
ws_miss = wb.create_sheet("Missing Records")
|
|
332
|
+
a_cols = [c for c in df_a.columns if not c.endswith("_b")]
|
|
333
|
+
clean_a_cols = [c.replace("_a", "") for c in a_cols]
|
|
334
|
+
ws_miss.append(clean_a_cols)
|
|
335
|
+
for c_idx in range(1, len(clean_a_cols) + 1):
|
|
336
|
+
cell = ws_miss.cell(row=1, column=c_idx)
|
|
337
|
+
cell.font = font_header
|
|
338
|
+
cell.fill = fill_header_orange
|
|
339
|
+
|
|
340
|
+
for _, row in only_in_a.iterrows():
|
|
341
|
+
ws_miss.append([_clean_val(row.get(f"{c}_a", row.get(c))) for c in clean_a_cols])
|
|
342
|
+
|
|
343
|
+
# Auto-fit columns across all sheets
|
|
344
|
+
for sheet in wb.worksheets:
|
|
345
|
+
for col in sheet.columns:
|
|
346
|
+
max_len = max(len(str(c.value or "")) for c in col)
|
|
347
|
+
col_letter = get_column_letter(col[0].column)
|
|
348
|
+
sheet.column_dimensions[col_letter].width = min(max(max_len + 3, 12), 45)
|
|
349
|
+
|
|
350
|
+
wb.save(out_p)
|
|
351
|
+
wb.close()
|
|
352
|
+
report_file_path = str(out_p)
|
|
353
|
+
|
|
354
|
+
duration = (datetime.now() - t_start).total_seconds()
|
|
355
|
+
|
|
356
|
+
return {
|
|
357
|
+
"status": "success",
|
|
358
|
+
"summary": {
|
|
359
|
+
"source_a_records": len(df_a),
|
|
360
|
+
"source_b_records": len(df_b),
|
|
361
|
+
"matched_keys": len(in_both),
|
|
362
|
+
"matching_records": identical_records_count,
|
|
363
|
+
"modified_records": len(discrepancies),
|
|
364
|
+
"new_records_in_b": len(only_in_b),
|
|
365
|
+
"missing_records_in_b": len(only_in_a),
|
|
366
|
+
},
|
|
367
|
+
"total_source_a_records": len(df_a),
|
|
368
|
+
"total_source_b_records": len(df_b),
|
|
369
|
+
"matched_primary_keys": len(in_both),
|
|
370
|
+
"identical_records_count": identical_records_count,
|
|
371
|
+
"records_with_discrepancies_count": len(discrepancies),
|
|
372
|
+
"new_records_in_source_b_count": len(only_in_b),
|
|
373
|
+
"missing_records_in_source_b_count": len(only_in_a),
|
|
374
|
+
"column_discrepancy_counts": col_discrepancy_counts,
|
|
375
|
+
"sample_discrepancies": discrepancies[:10],
|
|
376
|
+
"audit_report_file": report_file_path,
|
|
377
|
+
"duration_seconds": round(duration, 3),
|
|
378
|
+
}
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def reconcile_db_vs_excel(
|
|
382
|
+
sql_query: str,
|
|
383
|
+
excel_path: str,
|
|
384
|
+
key_columns: List[str],
|
|
385
|
+
db_url_or_config: Union[str, Dict[str, Any]],
|
|
386
|
+
sheet_name: Optional[Union[str, int]] = 0,
|
|
387
|
+
compare_columns: Optional[List[str]] = None,
|
|
388
|
+
output_report_path: Optional[str] = None,
|
|
389
|
+
) -> Dict[str, Any]:
|
|
390
|
+
"""
|
|
391
|
+
Backwards-compatible wrapper delegating to compare_master_datasets.
|
|
392
|
+
"""
|
|
393
|
+
db_source = {"type": "db", "connection": db_url_or_config, "query": sql_query}
|
|
394
|
+
excel_source = {"type": "excel", "path": excel_path, "sheet": sheet_name}
|
|
395
|
+
return compare_master_datasets(
|
|
396
|
+
source_a=db_source,
|
|
397
|
+
source_b=excel_source,
|
|
398
|
+
key_columns=key_columns,
|
|
399
|
+
compare_columns=compare_columns,
|
|
400
|
+
output_report_path=output_report_path,
|
|
401
|
+
db_resolver_func=lambda c: c,
|
|
402
|
+
)
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
CLI entry point for mcp-win-stdio-excel-db.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
|
|
8
|
+
from mcp_win_stdio.excel_db.server import mcp
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def main():
|
|
12
|
+
parser = argparse.ArgumentParser(description="mcp-win-stdio-excel-db MCP Server")
|
|
13
|
+
parser.add_argument("--transport", default="stdio", choices=["stdio", "sse"], help="MCP transport")
|
|
14
|
+
parser.add_argument("--port", type=int, default=8001, help="Port for SSE transport")
|
|
15
|
+
args = parser.parse_args()
|
|
16
|
+
|
|
17
|
+
if args.transport == "sse":
|
|
18
|
+
mcp.run(transport="sse", port=args.port)
|
|
19
|
+
else:
|
|
20
|
+
mcp.run(transport="stdio")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
if __name__ == "__main__":
|
|
24
|
+
main()
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Cross-Source Join Engine for executing unified SQL queries across SQL Databases and Excel Workbooks.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
import re
|
|
7
|
+
import sqlite3
|
|
8
|
+
from datetime import datetime
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any, Dict, List, Optional
|
|
11
|
+
|
|
12
|
+
import pandas as pd
|
|
13
|
+
from sqlalchemy import create_engine, text
|
|
14
|
+
|
|
15
|
+
from mcp_win_stdio.excel_db.stream import resolve_sqlalchemy_url
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def run_cross_source_query(
|
|
19
|
+
sources: List[Dict[str, Any]],
|
|
20
|
+
transformation_sql: str,
|
|
21
|
+
target: Optional[Dict[str, Any]] = None,
|
|
22
|
+
db_resolver_func=None,
|
|
23
|
+
) -> Dict[str, Any]:
|
|
24
|
+
"""
|
|
25
|
+
Execute a unified SQL query joining SQL databases and Excel spreadsheets in memory.
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
sources: List of data source definitions:
|
|
29
|
+
- {"name": "orders", "type": "db", "connection": "postgres", "query": "SELECT * FROM orders"}
|
|
30
|
+
- {"name": "clients", "type": "excel", "path": "Z:/data/clients.xlsx", "sheet": "VIP"}
|
|
31
|
+
transformation_sql: The SQL query to join/transform data across sources.
|
|
32
|
+
target: Optional export target:
|
|
33
|
+
- {"type": "excel", "path": "Z:/out.xlsx", "sheet": "Report"}
|
|
34
|
+
- {"type": "db", "connection": "mysql", "table": "report_tbl"}
|
|
35
|
+
db_resolver_func: Function to resolve connection names to db URLs.
|
|
36
|
+
"""
|
|
37
|
+
t_start = datetime.now()
|
|
38
|
+
mem_conn = sqlite3.connect(":memory:")
|
|
39
|
+
loaded_sources_info = []
|
|
40
|
+
|
|
41
|
+
try:
|
|
42
|
+
# 1. Ingest all sources into in-memory engine
|
|
43
|
+
for src in sources:
|
|
44
|
+
s_name = src.get("name")
|
|
45
|
+
s_type = src.get("type", "db").lower()
|
|
46
|
+
|
|
47
|
+
if not s_name:
|
|
48
|
+
raise ValueError("Each source must specify a unique 'name' table alias.")
|
|
49
|
+
|
|
50
|
+
if s_type == "db":
|
|
51
|
+
conn_id = src.get("connection")
|
|
52
|
+
raw_sql = src.get("query")
|
|
53
|
+
if not raw_sql:
|
|
54
|
+
raise ValueError(f"Source '{s_name}' is missing 'query'.")
|
|
55
|
+
|
|
56
|
+
# Resolve DB URL
|
|
57
|
+
db_url = db_resolver_func(conn_id) if db_resolver_func else conn_id
|
|
58
|
+
resolved_url = resolve_sqlalchemy_url(db_url)
|
|
59
|
+
eng = create_engine(resolved_url)
|
|
60
|
+
|
|
61
|
+
with eng.connect() as c:
|
|
62
|
+
df = pd.read_sql_query(text(raw_sql), c)
|
|
63
|
+
|
|
64
|
+
elif s_type in ("excel", "xlsx", "csv"):
|
|
65
|
+
file_path = Path(src.get("path", "")).resolve()
|
|
66
|
+
if not file_path.exists():
|
|
67
|
+
raise FileNotFoundError(f"Source file not found: {file_path}")
|
|
68
|
+
|
|
69
|
+
if file_path.suffix.lower() == ".csv":
|
|
70
|
+
df = pd.read_csv(file_path)
|
|
71
|
+
else:
|
|
72
|
+
sheet = src.get("sheet", 0)
|
|
73
|
+
df = pd.read_excel(file_path, sheet_name=sheet)
|
|
74
|
+
|
|
75
|
+
else:
|
|
76
|
+
raise ValueError(f"Unsupported source type: '{s_type}'. Supported: 'db', 'excel', 'csv'.")
|
|
77
|
+
|
|
78
|
+
# Clean column headers for SQLite
|
|
79
|
+
df.columns = [re.sub(r"[^a-zA-Z0-9_]", "_", str(col).strip()) for col in df.columns]
|
|
80
|
+
df.to_sql(s_name, mem_conn, index=False, if_exists="replace")
|
|
81
|
+
loaded_sources_info.append(f"{s_name} ({len(df)} rows, {len(df.columns)} cols)")
|
|
82
|
+
|
|
83
|
+
# 2. Execute unified transformation query
|
|
84
|
+
result_df = pd.read_sql_query(transformation_sql, mem_conn)
|
|
85
|
+
res_rows, res_cols = result_df.shape
|
|
86
|
+
|
|
87
|
+
# 3. Export to target or return summary
|
|
88
|
+
output_summary = {}
|
|
89
|
+
if target:
|
|
90
|
+
t_type = target.get("type", "excel").lower()
|
|
91
|
+
if t_type == "excel":
|
|
92
|
+
out_path = Path(target.get("path", "unified_report.xlsx")).resolve()
|
|
93
|
+
out_path.parent.mkdir(parents=True, exist_ok=True)
|
|
94
|
+
sheet_name = target.get("sheet", "Report")
|
|
95
|
+
|
|
96
|
+
with pd.ExcelWriter(out_path, engine="openpyxl") as writer:
|
|
97
|
+
result_df.to_excel(writer, sheet_name=sheet_name, index=False)
|
|
98
|
+
output_summary = {"type": "excel", "path": str(out_path), "sheet": sheet_name, "rows_written": res_rows}
|
|
99
|
+
|
|
100
|
+
elif t_type == "db":
|
|
101
|
+
conn_id = target.get("connection")
|
|
102
|
+
tbl_name = target.get("table", "pipeline_results")
|
|
103
|
+
if_exists = target.get("if_exists", "append")
|
|
104
|
+
|
|
105
|
+
db_url = db_resolver_func(conn_id) if db_resolver_func else conn_id
|
|
106
|
+
resolved_url = resolve_sqlalchemy_url(db_url)
|
|
107
|
+
eng = create_engine(resolved_url)
|
|
108
|
+
|
|
109
|
+
result_df.to_sql(tbl_name, con=eng, if_exists=if_exists, index=False, chunksize=1000)
|
|
110
|
+
output_summary = {"type": "db", "table": tbl_name, "rows_inserted": res_rows}
|
|
111
|
+
|
|
112
|
+
duration = (datetime.now() - t_start).total_seconds()
|
|
113
|
+
|
|
114
|
+
# If no target specified, return small sample preview (up to 5 rows)
|
|
115
|
+
sample_preview = result_df.head(5).to_dict(orient="records") if not target else None
|
|
116
|
+
|
|
117
|
+
return {
|
|
118
|
+
"status": "success",
|
|
119
|
+
"sources_loaded": loaded_sources_info,
|
|
120
|
+
"result_rows": res_rows,
|
|
121
|
+
"result_columns": res_cols,
|
|
122
|
+
"columns": list(result_df.columns),
|
|
123
|
+
"output_target": output_summary if target else "In-Memory Preview",
|
|
124
|
+
"preview_sample": sample_preview,
|
|
125
|
+
"duration_seconds": round(duration, 3),
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
finally:
|
|
129
|
+
mem_conn.close()
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Interactive and printable guide for the Excel-DB High-Speed Pipeline MCP Server.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
EXCEL_DB_GUIDE = """
|
|
6
|
+
# ================================================================
|
|
7
|
+
# EXCEL-DB MCP (HIGH-SPEED PIPELINES & RECONCILIATION) - USER GUIDE
|
|
8
|
+
# ================================================================
|
|
9
|
+
|
|
10
|
+
The Excel-DB MCP server provides 9 specialized tools for zero-context streaming,
|
|
11
|
+
cross-source SQL queries, automated master dataset reconciliation, and transactional
|
|
12
|
+
database migrations between Microsoft Excel and relational databases (PostgreSQL, MySQL, SQLite).
|
|
13
|
+
|
|
14
|
+
----------------------------------------------------------------
|
|
15
|
+
1. TOOL SUMMARY (9 TOOLS)
|
|
16
|
+
----------------------------------------------------------------
|
|
17
|
+
* Streaming & Direct ETL:
|
|
18
|
+
- db_to_excel_stream(sql_query, target_excel_path, ...):
|
|
19
|
+
Directly streams database query results to an Excel file with chunking,
|
|
20
|
+
optional table styling, and auto-fit column widths without loading rows into LLM context.
|
|
21
|
+
- excel_to_db_upsert(excel_path, target_table, chunk_size, ...):
|
|
22
|
+
Bulk streams Excel worksheet rows directly into a database table in chunks with append/replace modes.
|
|
23
|
+
- db_to_excel_template(sql_query, template_excel_path, output_excel_path, ...):
|
|
24
|
+
Injects query results into pre-styled Excel template workbooks, preserving charts, logos, and macros.
|
|
25
|
+
|
|
26
|
+
* Master Data Comparison & Audit:
|
|
27
|
+
- compare_master_datasets(source_a, source_b, key_columns, column_mapping, tolerance, output_report_path):
|
|
28
|
+
Deterministically compares two master datasets (Excel vs Excel, DB vs Excel, DB vs DB).
|
|
29
|
+
Detects additions, removals, and field discrepancies with floating-point tolerance.
|
|
30
|
+
Generates styled 4-tab Excel audit workbooks with KPI summary cards and cell diff highlights.
|
|
31
|
+
- reconcile_db_vs_excel(sql_query, excel_path, key_columns, ...):
|
|
32
|
+
Audits a live SQL database query against an Excel spreadsheet to spot missing records or discrepancies.
|
|
33
|
+
|
|
34
|
+
* Database Migration & Synchronization:
|
|
35
|
+
- generate_master_migration_plan(excel_path, target_table, key_columns, column_mapping_json, fk_lookups_json, ...):
|
|
36
|
+
Generates atomic, transactional PostgreSQL/MySQL migration SQL scripts from an Excel master.
|
|
37
|
+
Includes pre-flight foreign key validation (resolving codes like 'ROL' to UUIDs against live DB) and UPSERT handling.
|
|
38
|
+
- sync_master_to_db(excel_path, target_table, key_columns, column_mapping_json, fk_lookups_json, dry_run=True, ...):
|
|
39
|
+
Executes master data synchronization against a live database with full transactional safety.
|
|
40
|
+
Dry-run mode (default) simulates the entire migration and verifies all constraints without committing.
|
|
41
|
+
|
|
42
|
+
* Unified Analytics & Custom Python Pipelines:
|
|
43
|
+
- query_unified_sources(pipeline_spec_json):
|
|
44
|
+
Executes in-memory DuckDB/SQLite SQL queries across multiple heterogeneous sources (joining DB tables and Excel files).
|
|
45
|
+
- py_template_pipeline(template_excel_path, output_excel_path, pipeline_spec_json):
|
|
46
|
+
Executes custom Python/Pandas logic on multiple sources and writes resulting DataFrames into template sheets as styled tables.
|
|
47
|
+
|
|
48
|
+
----------------------------------------------------------------
|
|
49
|
+
2. BEST PRACTICES & LLM PROMPT RECIPES
|
|
50
|
+
----------------------------------------------------------------
|
|
51
|
+
* Reconciling Master Data:
|
|
52
|
+
"Compare masters/NewCatalog.xlsx against showreel_dev database table props_management.materials,
|
|
53
|
+
key on 'material_number', map 'SKU' -> 'material_number', and generate an audit report at reports/diff.xlsx."
|
|
54
|
+
|
|
55
|
+
* Safe Transactional Migration:
|
|
56
|
+
"Run sync_master_to_db with dry_run=True first to verify foreign keys and check constraint violations.
|
|
57
|
+
Only rerun with dry_run=False after inspecting the dry-run summary."
|
|
58
|
+
|
|
59
|
+
* Zero-Context Streaming for Large Datasets:
|
|
60
|
+
"Never load 50,000 rows into prompt context. Always use db_to_excel_stream to pipe query results directly into Excel."
|
|
61
|
+
# ================================================================
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def print_excel_db_guide() -> None:
|
|
66
|
+
"""Print the formatted Excel-DB MCP guide to stdout."""
|
|
67
|
+
print(EXCEL_DB_GUIDE.strip())
|