mcp-win-stdio-excel-db 0.2.5__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ """
2
+ Excel & DB Power Engine MCP Server for mcp-win-stdio.
3
+ High-Speed Streaming, Cross-Source Joins, and Diff Auditing.
4
+ """
5
+
6
+ __version__ = "0.2.5"
@@ -0,0 +1,5 @@
1
+ #!/usr/bin/env python3
2
+ from mcp_win_stdio.excel_db.cli import main
3
+
4
+ if __name__ == "__main__":
5
+ main()
@@ -0,0 +1,402 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Automated Data Auditor and Reconciliation Engine for comparing Database tables
4
+ and Excel spreadsheets with tolerance, column mappings, and multi-tab diff reports.
5
+ """
6
+
7
+ import math
8
+ from datetime import datetime
9
+ from pathlib import Path
10
+ from typing import Any, Dict, List, Optional, Union
11
+
12
+ import pandas as pd
13
+ from openpyxl import Workbook
14
+ from openpyxl.styles import Border, Font, PatternFill, Side
15
+ from openpyxl.utils import get_column_letter
16
+ from sqlalchemy import create_engine, text
17
+
18
+ from mcp_win_stdio.excel_db.stream import resolve_sqlalchemy_url
19
+
20
+
21
+ def _clean_val(v: Any) -> Any:
22
+ """Format primitives safely for JSON serialization and comparison."""
23
+ if pd.isna(v):
24
+ return None
25
+ elif hasattr(v, "isoformat"):
26
+ return v.isoformat()
27
+ elif isinstance(v, (float, int)) and (math.isnan(v) or math.isinf(v)):
28
+ return None
29
+ return v
30
+
31
+
32
+ def _load_dataset(source: Union[str, Dict[str, Any]], db_resolver_func=None) -> pd.DataFrame:
33
+ """
34
+ Load a dataset from either a file path (Excel/CSV) or a configuration dict (DB/Excel).
35
+ """
36
+ if isinstance(source, str):
37
+ path = Path(source).resolve()
38
+ if not path.exists():
39
+ raise FileNotFoundError(f"Source file not found: {path}")
40
+ if path.suffix.lower() == ".csv":
41
+ df = pd.read_csv(path)
42
+ else:
43
+ df = pd.read_excel(path)
44
+
45
+ if len(df) > 0 and (df.columns.dtype == "int64" or all(str(c).isdigit() for c in df.columns)):
46
+ new_cols = [str(x).strip() for x in df.iloc[0]]
47
+ df = df.iloc[1:].reset_index(drop=True)
48
+ df.columns = new_cols
49
+ return df
50
+
51
+ if not isinstance(source, dict):
52
+ raise ValueError(f"Invalid source specification: {source}")
53
+
54
+ s_type = source.get("type", "excel").lower()
55
+
56
+ if s_type == "db":
57
+ conn_id = source.get("connection")
58
+ raw_sql = source.get("query")
59
+ if not raw_sql:
60
+ raise ValueError("Database source must specify 'query'.")
61
+ db_url = db_resolver_func(conn_id) if db_resolver_func else conn_id
62
+ resolved_url = resolve_sqlalchemy_url(db_url)
63
+ engine = create_engine(resolved_url)
64
+ with engine.connect() as conn:
65
+ return pd.read_sql_query(text(raw_sql), conn)
66
+
67
+ elif s_type in ("excel", "xlsx"):
68
+ path = Path(source.get("path", "")).resolve()
69
+ if not path.exists():
70
+ raise FileNotFoundError(f"Source Excel file not found: {path}")
71
+ sheet = source.get("sheet", 0)
72
+ return pd.read_excel(path, sheet_name=sheet)
73
+
74
+ elif s_type == "csv":
75
+ path = Path(source.get("path", "")).resolve()
76
+ if not path.exists():
77
+ raise FileNotFoundError(f"Source CSV file not found: {path}")
78
+ return pd.read_csv(path)
79
+
80
+ else:
81
+ raise ValueError(f"Unsupported source type: '{s_type}'.")
82
+
83
+
84
+ def compare_master_datasets(
85
+ source_a: Union[str, Dict[str, Any]],
86
+ source_b: Union[str, Dict[str, Any]],
87
+ key_columns: List[str],
88
+ column_mapping: Optional[Dict[str, str]] = None,
89
+ compare_columns: Optional[List[str]] = None,
90
+ numeric_tolerance: float = 0.001,
91
+ ignore_whitespace_case: bool = True,
92
+ output_report_path: Optional[str] = None,
93
+ db_resolver_func=None,
94
+ ) -> Dict[str, Any]:
95
+ """
96
+ Deterministically compare two master datasets (Excel vs Excel, DB vs Excel, or DB vs DB).
97
+ Detects new records, missing records, and field-level discrepancies with tolerance.
98
+
99
+ Args:
100
+ source_a: Baseline master (file path or dict specification).
101
+ source_b: Comparison / New master (file path or dict specification).
102
+ key_columns: List of primary key column names (as named in source_a).
103
+ column_mapping: Optional map of source_b column names to source_a column names
104
+ (e.g. {"Material Number": "material_number", "Unit Rate": "sale_price"}).
105
+ compare_columns: Optional subset of source_a column names to compare.
106
+ numeric_tolerance: Maximum absolute difference considered equal for numbers (default 0.001).
107
+ ignore_whitespace_case: If True, trims whitespace and ignores case for string comparison.
108
+ output_report_path: Optional path to save a styled multi-tab Excel audit report.
109
+ db_resolver_func: Function to resolve connection names if DB sources are used.
110
+ """
111
+ t_start = datetime.now()
112
+
113
+ # 1. Load DataFrames
114
+ df_a = _load_dataset(source_a, db_resolver_func)
115
+ df_b = _load_dataset(source_b, db_resolver_func)
116
+
117
+ # 2. Normalize and apply column mapping to source_b
118
+ df_a.columns = [str(c).strip() for c in df_a.columns]
119
+ df_b.columns = [str(c).strip() for c in df_b.columns]
120
+
121
+ if column_mapping:
122
+ # Rename source_b columns to match source_a naming convention
123
+ rename_dict = {}
124
+ for b_col, a_col in column_mapping.items():
125
+ b_clean = str(b_col).strip()
126
+ a_clean = str(a_col).strip()
127
+ if b_clean in df_b.columns:
128
+ rename_dict[b_clean] = a_clean
129
+ df_b = df_b.rename(columns=rename_dict)
130
+
131
+ # 3. Validate Key Columns
132
+ for k in key_columns:
133
+ if k not in df_a.columns:
134
+ raise ValueError(f"Key column '{k}' not found in Source A. Available: {list(df_a.columns)}")
135
+ if k not in df_b.columns:
136
+ raise ValueError(f"Key column '{k}' not found in Source B. Available: {list(df_b.columns)}")
137
+
138
+ # 4. Determine columns to compare
139
+ if not compare_columns:
140
+ shared_cols = [c for c in df_a.columns if c in df_b.columns and c not in key_columns]
141
+ else:
142
+ shared_cols = [c for c in compare_columns if c in df_a.columns and c in df_b.columns and c not in key_columns]
143
+
144
+ # Clean key column values to strings for robust joining
145
+ for k in key_columns:
146
+ df_a[k] = df_a[k].astype(str).str.strip()
147
+ df_b[k] = df_b[k].astype(str).str.strip()
148
+
149
+ # 5. Full Outer Join on Key Columns
150
+ merged = pd.merge(df_a, df_b, on=key_columns, how="outer", suffixes=("_a", "_b"), indicator=True)
151
+
152
+ only_in_a = merged[merged["_merge"] == "left_only"].copy()
153
+ only_in_b = merged[merged["_merge"] == "right_only"].copy()
154
+ in_both = merged[merged["_merge"] == "both"].copy()
155
+
156
+ # 6. Detailed Field Comparison
157
+ discrepancies = []
158
+ col_discrepancy_counts: Dict[str, int] = {c: 0 for c in shared_cols}
159
+
160
+ for _, row in in_both.iterrows():
161
+ row_diffs = {}
162
+ for col in shared_cols:
163
+ val_a = row.get(f"{col}_a")
164
+ val_b = row.get(f"{col}_b")
165
+
166
+ # Handle nulls
167
+ is_na_a = pd.isna(val_a)
168
+ is_na_b = pd.isna(val_b)
169
+ if is_na_a and is_na_b:
170
+ continue
171
+ if is_na_a != is_na_b:
172
+ row_diffs[col] = {
173
+ "source_a": _clean_val(val_a),
174
+ "source_b": _clean_val(val_b),
175
+ "delta": None,
176
+ "type": "null_mismatch",
177
+ }
178
+ col_discrepancy_counts[col] += 1
179
+ continue
180
+
181
+ # Numeric comparison
182
+ is_num_a = isinstance(val_a, (int, float)) and not isinstance(val_a, bool)
183
+ is_num_b = isinstance(val_b, (int, float)) and not isinstance(val_b, bool)
184
+
185
+ if is_num_a and is_num_b:
186
+ delta = float(val_b) - float(val_a)
187
+ if abs(delta) > numeric_tolerance:
188
+ pct = round((delta / float(val_a) * 100), 2) if float(val_a) != 0 else None
189
+ row_diffs[col] = {
190
+ "source_a": _clean_val(val_a),
191
+ "source_b": _clean_val(val_b),
192
+ "delta": round(delta, 4),
193
+ "pct_change": pct,
194
+ "type": "numeric_diff",
195
+ }
196
+ col_discrepancy_counts[col] += 1
197
+ else:
198
+ # String / Boolean comparison
199
+ s_a = str(val_a).strip()
200
+ s_b = str(val_b).strip()
201
+ if ignore_whitespace_case:
202
+ s_a_cmp = s_a.lower()
203
+ s_b_cmp = s_b.lower()
204
+ else:
205
+ s_a_cmp = s_a
206
+ s_b_cmp = s_b
207
+
208
+ if s_a_cmp != s_b_cmp:
209
+ row_diffs[col] = {
210
+ "source_a": _clean_val(val_a),
211
+ "source_b": _clean_val(val_b),
212
+ "delta": None,
213
+ "type": "text_diff",
214
+ }
215
+ col_discrepancy_counts[col] += 1
216
+
217
+ if row_diffs:
218
+ keys_dict = {k: row[k] for k in key_columns}
219
+ discrepancies.append({"keys": keys_dict, "differences": row_diffs})
220
+
221
+ identical_records_count = len(in_both) - len(discrepancies)
222
+
223
+ # 7. Generate Multi-Tab Styled Audit Report
224
+ report_file_path = None
225
+ if output_report_path:
226
+ out_p = Path(output_report_path).resolve()
227
+ out_p.parent.mkdir(parents=True, exist_ok=True)
228
+ wb = Workbook()
229
+
230
+ # Styles
231
+ font_title = Font(name="Calibri", size=14, bold=True, color="1F497D")
232
+ font_header = Font(name="Calibri", size=11, bold=True, color="FFFFFF")
233
+ font_bold = Font(name="Calibri", size=11, bold=True)
234
+ fill_header_navy = PatternFill(start_color="1F497D", end_color="1F497D", fill_type="solid")
235
+ fill_header_green = PatternFill(start_color="27AE60", end_color="27AE60", fill_type="solid")
236
+ fill_header_orange = PatternFill(start_color="E67E22", end_color="E67E22", fill_type="solid")
237
+ fill_diff_yellow = PatternFill(start_color="FFF3CD", end_color="FFF3CD", fill_type="solid")
238
+ fill_diff_red = PatternFill(start_color="F8D7DA", end_color="F8D7DA", fill_type="solid")
239
+ border_thin = Border(
240
+ left=Side(style="thin", color="D3D3D3"),
241
+ right=Side(style="thin", color="D3D3D3"),
242
+ top=Side(style="thin", color="D3D3D3"),
243
+ bottom=Side(style="thin", color="D3D3D3"),
244
+ )
245
+
246
+ # Tab 1: Overview
247
+ ws_overview = wb.active
248
+ ws_overview.title = "Audit Overview"
249
+ ws_overview.append(["Master Data Reconciliation Audit Report", ""])
250
+ ws_overview["A1"].font = font_title
251
+ ws_overview.append([])
252
+ ws_overview.append(["Metric", "Count / Value"])
253
+ ws_overview["A3"].font = font_header
254
+ ws_overview["B3"].font = font_header
255
+ ws_overview["A3"].fill = fill_header_navy
256
+ ws_overview["B3"].fill = fill_header_navy
257
+
258
+ metrics = [
259
+ ("Audit Execution Timestamp", datetime.now().isoformat()),
260
+ ("Source A Total Records", len(df_a)),
261
+ ("Source B Total Records", len(df_b)),
262
+ ("Matched Primary Keys", len(in_both)),
263
+ ("Identical Records (Within Tolerance)", identical_records_count),
264
+ ("Records with Value Discrepancies", len(discrepancies)),
265
+ ("New Records (Source B Only)", len(only_in_b)),
266
+ ("Missing Records (Source A Only)", len(only_in_a)),
267
+ ("Numeric Tolerance Applied", numeric_tolerance),
268
+ ]
269
+ for m_name, m_val in metrics:
270
+ ws_overview.append([m_name, m_val])
271
+
272
+ ws_overview.append([])
273
+ ws_overview.append(["Field Name", "Mismatched Rows Count"])
274
+ ws_overview["A15"].font = font_header
275
+ ws_overview["B15"].font = font_header
276
+ ws_overview["A15"].fill = fill_header_navy
277
+ ws_overview["B15"].fill = fill_header_navy
278
+ for fld, cnt in col_discrepancy_counts.items():
279
+ ws_overview.append([fld, cnt])
280
+
281
+ # Tab 2: Discrepancies
282
+ if discrepancies:
283
+ ws_diff = wb.create_sheet("Field Differences")
284
+ diff_headers = list(key_columns) + [
285
+ "Field",
286
+ "Source A (Baseline)",
287
+ "Source B (New)",
288
+ "Delta",
289
+ "% Change",
290
+ "Diff Type",
291
+ ]
292
+ ws_diff.append(diff_headers)
293
+ for c_idx in range(1, len(diff_headers) + 1):
294
+ cell = ws_diff.cell(row=1, column=c_idx)
295
+ cell.font = font_header
296
+ cell.fill = fill_header_navy
297
+
298
+ for item in discrepancies:
299
+ k_vals = [item["keys"][k] for k in key_columns]
300
+ for fld, diff_info in item["differences"].items():
301
+ row_vals = k_vals + [
302
+ fld,
303
+ str(diff_info["source_a"]),
304
+ str(diff_info["source_b"]),
305
+ diff_info.get("delta"),
306
+ diff_info.get("pct_change"),
307
+ diff_info.get("type"),
308
+ ]
309
+ ws_diff.append(row_vals)
310
+ # highlight
311
+ curr_row = ws_diff.max_row
312
+ for c_idx in range(1, len(row_vals) + 1):
313
+ ws_diff.cell(row=curr_row, column=c_idx).fill = fill_diff_yellow
314
+
315
+ # Tab 3: New Records in B
316
+ if len(only_in_b) > 0:
317
+ ws_new = wb.create_sheet("New Records")
318
+ b_cols = [c for c in df_b.columns if not c.endswith("_a")]
319
+ clean_b_cols = [c.replace("_b", "") for c in b_cols]
320
+ ws_new.append(clean_b_cols)
321
+ for c_idx in range(1, len(clean_b_cols) + 1):
322
+ cell = ws_new.cell(row=1, column=c_idx)
323
+ cell.font = font_header
324
+ cell.fill = fill_header_green
325
+
326
+ for _, row in only_in_b.iterrows():
327
+ ws_new.append([_clean_val(row.get(f"{c}_b", row.get(c))) for c in clean_b_cols])
328
+
329
+ # Tab 4: Missing Records in B
330
+ if len(only_in_a) > 0:
331
+ ws_miss = wb.create_sheet("Missing Records")
332
+ a_cols = [c for c in df_a.columns if not c.endswith("_b")]
333
+ clean_a_cols = [c.replace("_a", "") for c in a_cols]
334
+ ws_miss.append(clean_a_cols)
335
+ for c_idx in range(1, len(clean_a_cols) + 1):
336
+ cell = ws_miss.cell(row=1, column=c_idx)
337
+ cell.font = font_header
338
+ cell.fill = fill_header_orange
339
+
340
+ for _, row in only_in_a.iterrows():
341
+ ws_miss.append([_clean_val(row.get(f"{c}_a", row.get(c))) for c in clean_a_cols])
342
+
343
+ # Auto-fit columns across all sheets
344
+ for sheet in wb.worksheets:
345
+ for col in sheet.columns:
346
+ max_len = max(len(str(c.value or "")) for c in col)
347
+ col_letter = get_column_letter(col[0].column)
348
+ sheet.column_dimensions[col_letter].width = min(max(max_len + 3, 12), 45)
349
+
350
+ wb.save(out_p)
351
+ wb.close()
352
+ report_file_path = str(out_p)
353
+
354
+ duration = (datetime.now() - t_start).total_seconds()
355
+
356
+ return {
357
+ "status": "success",
358
+ "summary": {
359
+ "source_a_records": len(df_a),
360
+ "source_b_records": len(df_b),
361
+ "matched_keys": len(in_both),
362
+ "matching_records": identical_records_count,
363
+ "modified_records": len(discrepancies),
364
+ "new_records_in_b": len(only_in_b),
365
+ "missing_records_in_b": len(only_in_a),
366
+ },
367
+ "total_source_a_records": len(df_a),
368
+ "total_source_b_records": len(df_b),
369
+ "matched_primary_keys": len(in_both),
370
+ "identical_records_count": identical_records_count,
371
+ "records_with_discrepancies_count": len(discrepancies),
372
+ "new_records_in_source_b_count": len(only_in_b),
373
+ "missing_records_in_source_b_count": len(only_in_a),
374
+ "column_discrepancy_counts": col_discrepancy_counts,
375
+ "sample_discrepancies": discrepancies[:10],
376
+ "audit_report_file": report_file_path,
377
+ "duration_seconds": round(duration, 3),
378
+ }
379
+
380
+
381
+ def reconcile_db_vs_excel(
382
+ sql_query: str,
383
+ excel_path: str,
384
+ key_columns: List[str],
385
+ db_url_or_config: Union[str, Dict[str, Any]],
386
+ sheet_name: Optional[Union[str, int]] = 0,
387
+ compare_columns: Optional[List[str]] = None,
388
+ output_report_path: Optional[str] = None,
389
+ ) -> Dict[str, Any]:
390
+ """
391
+ Backwards-compatible wrapper delegating to compare_master_datasets.
392
+ """
393
+ db_source = {"type": "db", "connection": db_url_or_config, "query": sql_query}
394
+ excel_source = {"type": "excel", "path": excel_path, "sheet": sheet_name}
395
+ return compare_master_datasets(
396
+ source_a=db_source,
397
+ source_b=excel_source,
398
+ key_columns=key_columns,
399
+ compare_columns=compare_columns,
400
+ output_report_path=output_report_path,
401
+ db_resolver_func=lambda c: c,
402
+ )
@@ -0,0 +1,24 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ CLI entry point for mcp-win-stdio-excel-db.
4
+ """
5
+
6
+ import argparse
7
+
8
+ from mcp_win_stdio.excel_db.server import mcp
9
+
10
+
11
+ def main():
12
+ parser = argparse.ArgumentParser(description="mcp-win-stdio-excel-db MCP Server")
13
+ parser.add_argument("--transport", default="stdio", choices=["stdio", "sse"], help="MCP transport")
14
+ parser.add_argument("--port", type=int, default=8001, help="Port for SSE transport")
15
+ args = parser.parse_args()
16
+
17
+ if args.transport == "sse":
18
+ mcp.run(transport="sse", port=args.port)
19
+ else:
20
+ mcp.run(transport="stdio")
21
+
22
+
23
+ if __name__ == "__main__":
24
+ main()
@@ -0,0 +1,129 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Cross-Source Join Engine for executing unified SQL queries across SQL Databases and Excel Workbooks.
4
+ """
5
+
6
+ import re
7
+ import sqlite3
8
+ from datetime import datetime
9
+ from pathlib import Path
10
+ from typing import Any, Dict, List, Optional
11
+
12
+ import pandas as pd
13
+ from sqlalchemy import create_engine, text
14
+
15
+ from mcp_win_stdio.excel_db.stream import resolve_sqlalchemy_url
16
+
17
+
18
+ def run_cross_source_query(
19
+ sources: List[Dict[str, Any]],
20
+ transformation_sql: str,
21
+ target: Optional[Dict[str, Any]] = None,
22
+ db_resolver_func=None,
23
+ ) -> Dict[str, Any]:
24
+ """
25
+ Execute a unified SQL query joining SQL databases and Excel spreadsheets in memory.
26
+
27
+ Args:
28
+ sources: List of data source definitions:
29
+ - {"name": "orders", "type": "db", "connection": "postgres", "query": "SELECT * FROM orders"}
30
+ - {"name": "clients", "type": "excel", "path": "Z:/data/clients.xlsx", "sheet": "VIP"}
31
+ transformation_sql: The SQL query to join/transform data across sources.
32
+ target: Optional export target:
33
+ - {"type": "excel", "path": "Z:/out.xlsx", "sheet": "Report"}
34
+ - {"type": "db", "connection": "mysql", "table": "report_tbl"}
35
+ db_resolver_func: Function to resolve connection names to db URLs.
36
+ """
37
+ t_start = datetime.now()
38
+ mem_conn = sqlite3.connect(":memory:")
39
+ loaded_sources_info = []
40
+
41
+ try:
42
+ # 1. Ingest all sources into in-memory engine
43
+ for src in sources:
44
+ s_name = src.get("name")
45
+ s_type = src.get("type", "db").lower()
46
+
47
+ if not s_name:
48
+ raise ValueError("Each source must specify a unique 'name' table alias.")
49
+
50
+ if s_type == "db":
51
+ conn_id = src.get("connection")
52
+ raw_sql = src.get("query")
53
+ if not raw_sql:
54
+ raise ValueError(f"Source '{s_name}' is missing 'query'.")
55
+
56
+ # Resolve DB URL
57
+ db_url = db_resolver_func(conn_id) if db_resolver_func else conn_id
58
+ resolved_url = resolve_sqlalchemy_url(db_url)
59
+ eng = create_engine(resolved_url)
60
+
61
+ with eng.connect() as c:
62
+ df = pd.read_sql_query(text(raw_sql), c)
63
+
64
+ elif s_type in ("excel", "xlsx", "csv"):
65
+ file_path = Path(src.get("path", "")).resolve()
66
+ if not file_path.exists():
67
+ raise FileNotFoundError(f"Source file not found: {file_path}")
68
+
69
+ if file_path.suffix.lower() == ".csv":
70
+ df = pd.read_csv(file_path)
71
+ else:
72
+ sheet = src.get("sheet", 0)
73
+ df = pd.read_excel(file_path, sheet_name=sheet)
74
+
75
+ else:
76
+ raise ValueError(f"Unsupported source type: '{s_type}'. Supported: 'db', 'excel', 'csv'.")
77
+
78
+ # Clean column headers for SQLite
79
+ df.columns = [re.sub(r"[^a-zA-Z0-9_]", "_", str(col).strip()) for col in df.columns]
80
+ df.to_sql(s_name, mem_conn, index=False, if_exists="replace")
81
+ loaded_sources_info.append(f"{s_name} ({len(df)} rows, {len(df.columns)} cols)")
82
+
83
+ # 2. Execute unified transformation query
84
+ result_df = pd.read_sql_query(transformation_sql, mem_conn)
85
+ res_rows, res_cols = result_df.shape
86
+
87
+ # 3. Export to target or return summary
88
+ output_summary = {}
89
+ if target:
90
+ t_type = target.get("type", "excel").lower()
91
+ if t_type == "excel":
92
+ out_path = Path(target.get("path", "unified_report.xlsx")).resolve()
93
+ out_path.parent.mkdir(parents=True, exist_ok=True)
94
+ sheet_name = target.get("sheet", "Report")
95
+
96
+ with pd.ExcelWriter(out_path, engine="openpyxl") as writer:
97
+ result_df.to_excel(writer, sheet_name=sheet_name, index=False)
98
+ output_summary = {"type": "excel", "path": str(out_path), "sheet": sheet_name, "rows_written": res_rows}
99
+
100
+ elif t_type == "db":
101
+ conn_id = target.get("connection")
102
+ tbl_name = target.get("table", "pipeline_results")
103
+ if_exists = target.get("if_exists", "append")
104
+
105
+ db_url = db_resolver_func(conn_id) if db_resolver_func else conn_id
106
+ resolved_url = resolve_sqlalchemy_url(db_url)
107
+ eng = create_engine(resolved_url)
108
+
109
+ result_df.to_sql(tbl_name, con=eng, if_exists=if_exists, index=False, chunksize=1000)
110
+ output_summary = {"type": "db", "table": tbl_name, "rows_inserted": res_rows}
111
+
112
+ duration = (datetime.now() - t_start).total_seconds()
113
+
114
+ # If no target specified, return small sample preview (up to 5 rows)
115
+ sample_preview = result_df.head(5).to_dict(orient="records") if not target else None
116
+
117
+ return {
118
+ "status": "success",
119
+ "sources_loaded": loaded_sources_info,
120
+ "result_rows": res_rows,
121
+ "result_columns": res_cols,
122
+ "columns": list(result_df.columns),
123
+ "output_target": output_summary if target else "In-Memory Preview",
124
+ "preview_sample": sample_preview,
125
+ "duration_seconds": round(duration, 3),
126
+ }
127
+
128
+ finally:
129
+ mem_conn.close()
@@ -0,0 +1,67 @@
1
+ """
2
+ Interactive and printable guide for the Excel-DB High-Speed Pipeline MCP Server.
3
+ """
4
+
5
+ EXCEL_DB_GUIDE = """
6
+ # ================================================================
7
+ # EXCEL-DB MCP (HIGH-SPEED PIPELINES & RECONCILIATION) - USER GUIDE
8
+ # ================================================================
9
+
10
+ The Excel-DB MCP server provides 9 specialized tools for zero-context streaming,
11
+ cross-source SQL queries, automated master dataset reconciliation, and transactional
12
+ database migrations between Microsoft Excel and relational databases (PostgreSQL, MySQL, SQLite).
13
+
14
+ ----------------------------------------------------------------
15
+ 1. TOOL SUMMARY (9 TOOLS)
16
+ ----------------------------------------------------------------
17
+ * Streaming & Direct ETL:
18
+ - db_to_excel_stream(sql_query, target_excel_path, ...):
19
+ Directly streams database query results to an Excel file with chunking,
20
+ optional table styling, and auto-fit column widths without loading rows into LLM context.
21
+ - excel_to_db_upsert(excel_path, target_table, chunk_size, ...):
22
+ Bulk streams Excel worksheet rows directly into a database table in chunks with append/replace modes.
23
+ - db_to_excel_template(sql_query, template_excel_path, output_excel_path, ...):
24
+ Injects query results into pre-styled Excel template workbooks, preserving charts, logos, and macros.
25
+
26
+ * Master Data Comparison & Audit:
27
+ - compare_master_datasets(source_a, source_b, key_columns, column_mapping, tolerance, output_report_path):
28
+ Deterministically compares two master datasets (Excel vs Excel, DB vs Excel, DB vs DB).
29
+ Detects additions, removals, and field discrepancies with floating-point tolerance.
30
+ Generates styled 4-tab Excel audit workbooks with KPI summary cards and cell diff highlights.
31
+ - reconcile_db_vs_excel(sql_query, excel_path, key_columns, ...):
32
+ Audits a live SQL database query against an Excel spreadsheet to spot missing records or discrepancies.
33
+
34
+ * Database Migration & Synchronization:
35
+ - generate_master_migration_plan(excel_path, target_table, key_columns, column_mapping_json, fk_lookups_json, ...):
36
+ Generates atomic, transactional PostgreSQL/MySQL migration SQL scripts from an Excel master.
37
+ Includes pre-flight foreign key validation (resolving codes like 'ROL' to UUIDs against live DB) and UPSERT handling.
38
+ - sync_master_to_db(excel_path, target_table, key_columns, column_mapping_json, fk_lookups_json, dry_run=True, ...):
39
+ Executes master data synchronization against a live database with full transactional safety.
40
+ Dry-run mode (default) simulates the entire migration and verifies all constraints without committing.
41
+
42
+ * Unified Analytics & Custom Python Pipelines:
43
+ - query_unified_sources(pipeline_spec_json):
44
+ Executes in-memory DuckDB/SQLite SQL queries across multiple heterogeneous sources (joining DB tables and Excel files).
45
+ - py_template_pipeline(template_excel_path, output_excel_path, pipeline_spec_json):
46
+ Executes custom Python/Pandas logic on multiple sources and writes resulting DataFrames into template sheets as styled tables.
47
+
48
+ ----------------------------------------------------------------
49
+ 2. BEST PRACTICES & LLM PROMPT RECIPES
50
+ ----------------------------------------------------------------
51
+ * Reconciling Master Data:
52
+ "Compare masters/NewCatalog.xlsx against showreel_dev database table props_management.materials,
53
+ key on 'material_number', map 'SKU' -> 'material_number', and generate an audit report at reports/diff.xlsx."
54
+
55
+ * Safe Transactional Migration:
56
+ "Run sync_master_to_db with dry_run=True first to verify foreign keys and check constraint violations.
57
+ Only rerun with dry_run=False after inspecting the dry-run summary."
58
+
59
+ * Zero-Context Streaming for Large Datasets:
60
+ "Never load 50,000 rows into prompt context. Always use db_to_excel_stream to pipe query results directly into Excel."
61
+ # ================================================================
62
+ """
63
+
64
+
65
+ def print_excel_db_guide() -> None:
66
+ """Print the formatted Excel-DB MCP guide to stdout."""
67
+ print(EXCEL_DB_GUIDE.strip())