sheetdiff 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sheetdiff/__init__.py +8 -0
- sheetdiff/cli.py +42 -0
- sheetdiff/config.py +127 -0
- sheetdiff/core.py +356 -0
- sheetdiff/storage.py +201 -0
- sheetdiff/templates/diff.html +511 -0
- sheetdiff/templates/index.html +325 -0
- sheetdiff/webapp.py +176 -0
- sheetdiff-0.1.0.dist-info/METADATA +190 -0
- sheetdiff-0.1.0.dist-info/RECORD +14 -0
- sheetdiff-0.1.0.dist-info/WHEEL +5 -0
- sheetdiff-0.1.0.dist-info/entry_points.txt +2 -0
- sheetdiff-0.1.0.dist-info/licenses/LICENSE +21 -0
- sheetdiff-0.1.0.dist-info/top_level.txt +1 -0
sheetdiff/__init__.py
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
from .core import diff_sheets, format_unified, diff_rows_full, diff_multi_grid, compare_workbooks
|
|
2
|
+
from . import config
|
|
3
|
+
from .storage import resolve as resolve_source, resolve_many as resolve_sources, StorageError
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"diff_sheets", "format_unified", "diff_rows_full", "diff_multi_grid", "compare_workbooks",
|
|
7
|
+
"config", "resolve_source", "resolve_sources", "StorageError",
|
|
8
|
+
]
|
sheetdiff/cli.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
import argparse
|
|
2
|
+
import sys
|
|
3
|
+
import tempfile
|
|
4
|
+
|
|
5
|
+
from . import config, storage
|
|
6
|
+
from .core import diff_sheets, format_unified, diff_workbook, format_workbook
|
|
7
|
+
|
|
8
|
+
def main():
|
|
9
|
+
p = argparse.ArgumentParser(prog="sheetdiff")
|
|
10
|
+
p.add_argument("left", help="local path or remote URI (s3://, minio://, r2://, garage://, az://)")
|
|
11
|
+
p.add_argument("right", help="local path or remote URI (s3://, minio://, r2://, garage://, az://)")
|
|
12
|
+
p.add_argument("--key", help="column name to use as row key")
|
|
13
|
+
p.add_argument("--sheet", help="sheet name for Excel files")
|
|
14
|
+
p.add_argument("--all-sheets", action="store_true", help="compare all sheets in workbooks")
|
|
15
|
+
p.add_argument(
|
|
16
|
+
"--max-remote-mb", type=int, default=None,
|
|
17
|
+
help=f"override the remote download size cap (default: {config.MAX_REMOTE_MB} MB, "
|
|
18
|
+
"also settable via SHEETDIFF_MAX_REMOTE_MB)",
|
|
19
|
+
)
|
|
20
|
+
args = p.parse_args()
|
|
21
|
+
|
|
22
|
+
max_bytes = args.max_remote_mb * 1024 * 1024 if args.max_remote_mb else None
|
|
23
|
+
|
|
24
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmp:
|
|
25
|
+
try:
|
|
26
|
+
left_path = storage.resolve(args.left, tmp, max_bytes=max_bytes)
|
|
27
|
+
right_path = storage.resolve(args.right, tmp, max_bytes=max_bytes)
|
|
28
|
+
except storage.StorageError as e:
|
|
29
|
+
print(f"error: {e}", file=sys.stderr)
|
|
30
|
+
raise SystemExit(1)
|
|
31
|
+
|
|
32
|
+
if args.all_sheets:
|
|
33
|
+
from .core import compare_workbooks
|
|
34
|
+
wb = compare_workbooks(left_path, right_path, key=args.key)
|
|
35
|
+
print(format_workbook(wb))
|
|
36
|
+
else:
|
|
37
|
+
changes = diff_sheets(left_path, right_path, key=args.key, sheet_name=args.sheet)
|
|
38
|
+
print(format_unified(changes))
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
if __name__ == "__main__":
|
|
42
|
+
main()
|
sheetdiff/config.py
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""Central, environment-driven configuration for sheetdiff.
|
|
2
|
+
|
|
3
|
+
Every knob here can be overridden with an environment variable so a
|
|
4
|
+
deployment can tune limits and remote-storage credentials without touching
|
|
5
|
+
code. Nothing here reads a hardcoded secret — credentials only ever come
|
|
6
|
+
from the environment (or a `.env` loaded by the process before start-up),
|
|
7
|
+
never from user-submitted form fields, so a web UI cannot be used to smuggle
|
|
8
|
+
credentials for a backend the operator didn't configure.
|
|
9
|
+
"""
|
|
10
|
+
import os
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Dict, Optional
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _env_int(name: str, default: int) -> int:
|
|
16
|
+
raw = os.environ.get(name)
|
|
17
|
+
if raw is None or raw.strip() == "":
|
|
18
|
+
return default
|
|
19
|
+
try:
|
|
20
|
+
value = int(raw)
|
|
21
|
+
except ValueError:
|
|
22
|
+
raise ValueError(f"{name}={raw!r} is not a valid integer") from None
|
|
23
|
+
if value <= 0:
|
|
24
|
+
raise ValueError(f"{name}={raw!r} must be a positive integer")
|
|
25
|
+
return value
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _env_list(name: str, default: str) -> list:
|
|
29
|
+
raw = os.environ.get(name, default)
|
|
30
|
+
return [x.strip().lower() for x in raw.split(",") if x.strip()]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# --- Upload / download size limits -----------------------------------------
|
|
34
|
+
# All expressed in MB for readability at the env-var layer; converted to
|
|
35
|
+
# bytes for use. Change per-deployment via env vars, e.g.:
|
|
36
|
+
# SHEETDIFF_MAX_UPLOAD_MB=250
|
|
37
|
+
# SHEETDIFF_MAX_REMOTE_MB=500
|
|
38
|
+
# SHEETDIFF_MAX_FILES=6
|
|
39
|
+
MAX_UPLOAD_MB = _env_int("SHEETDIFF_MAX_UPLOAD_MB", 100)
|
|
40
|
+
MAX_REMOTE_MB = _env_int("SHEETDIFF_MAX_REMOTE_MB", MAX_UPLOAD_MB)
|
|
41
|
+
MAX_UPLOAD_BYTES = MAX_UPLOAD_MB * 1024 * 1024
|
|
42
|
+
MAX_REMOTE_BYTES = MAX_REMOTE_MB * 1024 * 1024
|
|
43
|
+
|
|
44
|
+
# Hard ceiling on how many files can be compared in one request (web UI and
|
|
45
|
+
# CLI --all-sheets multi-file mode), to bound memory/CPU on shared servers.
|
|
46
|
+
MAX_FILES = _env_int("SHEETDIFF_MAX_FILES", 8)
|
|
47
|
+
|
|
48
|
+
# Network timeouts for remote storage reads (seconds).
|
|
49
|
+
REMOTE_CONNECT_TIMEOUT = _env_int("SHEETDIFF_REMOTE_CONNECT_TIMEOUT", 10)
|
|
50
|
+
REMOTE_READ_TIMEOUT = _env_int("SHEETDIFF_REMOTE_READ_TIMEOUT", 60)
|
|
51
|
+
|
|
52
|
+
# Which remote URI schemes are enabled at all. Empty by default in the sense
|
|
53
|
+
# that only schemes listed here are ever dispatched to fsspec; anything else
|
|
54
|
+
# (including http/https, to avoid turning the diff endpoint into an SSRF
|
|
55
|
+
# proxy) is rejected outright.
|
|
56
|
+
ALLOWED_REMOTE_SCHEMES = set(
|
|
57
|
+
_env_list("SHEETDIFF_ALLOWED_SCHEMES", "s3,minio,r2,garage,az,azure,gs")
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass
|
|
62
|
+
class BackendConfig:
|
|
63
|
+
protocol: str # fsspec protocol, e.g. "s3" or "az"
|
|
64
|
+
storage_options: Dict[str, str] = field(default_factory=dict)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _s3_backend(scheme: str) -> BackendConfig:
|
|
68
|
+
prefix = f"SHEETDIFF_STORAGE_{scheme.upper()}_"
|
|
69
|
+
opts: Dict[str, str] = {}
|
|
70
|
+
endpoint = os.environ.get(prefix + "ENDPOINT_URL")
|
|
71
|
+
if endpoint:
|
|
72
|
+
opts["client_kwargs"] = {"endpoint_url": endpoint}
|
|
73
|
+
key = os.environ.get(prefix + "KEY") or os.environ.get("AWS_ACCESS_KEY_ID")
|
|
74
|
+
secret = os.environ.get(prefix + "SECRET") or os.environ.get("AWS_SECRET_ACCESS_KEY")
|
|
75
|
+
region = os.environ.get(prefix + "REGION") or os.environ.get("AWS_DEFAULT_REGION")
|
|
76
|
+
if key:
|
|
77
|
+
opts["key"] = key
|
|
78
|
+
if secret:
|
|
79
|
+
opts["secret"] = secret
|
|
80
|
+
if region:
|
|
81
|
+
opts.setdefault("client_kwargs", {})["region_name"] = region
|
|
82
|
+
return BackendConfig(protocol="s3", storage_options=opts)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _azure_backend() -> BackendConfig:
|
|
86
|
+
prefix = "SHEETDIFF_STORAGE_AZ_"
|
|
87
|
+
opts: Dict[str, str] = {}
|
|
88
|
+
conn_str = os.environ.get(prefix + "CONNECTION_STRING")
|
|
89
|
+
if conn_str:
|
|
90
|
+
opts["connection_string"] = conn_str
|
|
91
|
+
else:
|
|
92
|
+
account = os.environ.get(prefix + "ACCOUNT_NAME")
|
|
93
|
+
account_key = os.environ.get(prefix + "ACCOUNT_KEY")
|
|
94
|
+
sas_token = os.environ.get(prefix + "SAS_TOKEN")
|
|
95
|
+
if account:
|
|
96
|
+
opts["account_name"] = account
|
|
97
|
+
if account_key:
|
|
98
|
+
opts["account_key"] = account_key
|
|
99
|
+
if sas_token:
|
|
100
|
+
opts["sas_token"] = sas_token
|
|
101
|
+
return BackendConfig(protocol="az", storage_options=opts)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _gcs_backend() -> BackendConfig:
|
|
105
|
+
opts: Dict[str, str] = {}
|
|
106
|
+
creds = os.environ.get("SHEETDIFF_STORAGE_GS_TOKEN")
|
|
107
|
+
if creds:
|
|
108
|
+
opts["token"] = creds
|
|
109
|
+
return BackendConfig(protocol="gcs", storage_options=opts)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def backend_for_scheme(scheme: str) -> Optional[BackendConfig]:
|
|
113
|
+
"""Map a URI scheme (e.g. 'minio', 'r2', 'az') to an fsspec backend.
|
|
114
|
+
|
|
115
|
+
Returns None if the scheme isn't recognized/enabled, so callers can
|
|
116
|
+
reject it instead of silently falling through to some default.
|
|
117
|
+
"""
|
|
118
|
+
scheme = scheme.lower()
|
|
119
|
+
if scheme not in ALLOWED_REMOTE_SCHEMES:
|
|
120
|
+
return None
|
|
121
|
+
if scheme in ("s3", "minio", "r2", "garage"):
|
|
122
|
+
return _s3_backend(scheme)
|
|
123
|
+
if scheme in ("az", "azure"):
|
|
124
|
+
return _azure_backend()
|
|
125
|
+
if scheme == "gs":
|
|
126
|
+
return _gcs_backend()
|
|
127
|
+
return None
|
sheetdiff/core.py
ADDED
|
@@ -0,0 +1,356 @@
|
|
|
1
|
+
from typing import List, Tuple, Optional, Any, Dict
|
|
2
|
+
import os
|
|
3
|
+
import tempfile
|
|
4
|
+
import pandas as pd
|
|
5
|
+
import json
|
|
6
|
+
|
|
7
|
+
# Optional Rust backend (xl_diff) if built and installed
|
|
8
|
+
_RUST_AVAILABLE = False
|
|
9
|
+
_rust = None
|
|
10
|
+
try:
|
|
11
|
+
import xl_diff as _rust # rust extension built via maturin/pyo3
|
|
12
|
+
_RUST_AVAILABLE = True
|
|
13
|
+
except Exception:
|
|
14
|
+
_RUST_AVAILABLE = False
|
|
15
|
+
|
|
16
|
+
def _load(path: str, sheet_name: Optional[str] = None) -> pd.DataFrame:
|
|
17
|
+
if str(path).lower().endswith(('.xls', '.xlsx', '.xlsm')):
|
|
18
|
+
return pd.read_excel(path, sheet_name=sheet_name if sheet_name is not None else 0, dtype=object)
|
|
19
|
+
return pd.read_csv(path, dtype=object)
|
|
20
|
+
|
|
21
|
+
def _col_letter(n: int) -> str:
|
|
22
|
+
"""0 -> A, 25 -> Z, 26 -> AA ... (Excel-style column naming)."""
|
|
23
|
+
s = ""
|
|
24
|
+
n += 1
|
|
25
|
+
while n:
|
|
26
|
+
n, rem = divmod(n - 1, 26)
|
|
27
|
+
s = chr(65 + rem) + s
|
|
28
|
+
return s
|
|
29
|
+
|
|
30
|
+
def _trim_grid(df: pd.DataFrame) -> pd.DataFrame:
|
|
31
|
+
"""Drop trailing all-empty columns/rows.
|
|
32
|
+
|
|
33
|
+
Excel files often advertise a "used range" far larger than the actual data
|
|
34
|
+
(leftover formatting on empty cells), which otherwise shows up as an
|
|
35
|
+
unbounded number of phantom empty columns/rows in the diff.
|
|
36
|
+
"""
|
|
37
|
+
if df.empty:
|
|
38
|
+
return df
|
|
39
|
+
mask = df.notna() & df.astype(str).apply(lambda s: s.str.strip() != "")
|
|
40
|
+
col_has = mask.any(axis=0)
|
|
41
|
+
row_has = mask.any(axis=1)
|
|
42
|
+
if not col_has.any() or not row_has.any():
|
|
43
|
+
return df.iloc[0:0, 0:0]
|
|
44
|
+
last_col = col_has[col_has].index.max()
|
|
45
|
+
last_row = row_has[row_has].index.max()
|
|
46
|
+
return df.loc[:last_row, :last_col]
|
|
47
|
+
|
|
48
|
+
def _load_grid(path: str, sheet_name: Optional[str] = None, header: bool = False) -> pd.DataFrame:
|
|
49
|
+
"""Load a sheet as a raw cell grid (no assumed header row) unless header=True.
|
|
50
|
+
|
|
51
|
+
Every spreadsheet row becomes a data row, and columns are labelled A, B, C...
|
|
52
|
+
matching real Excel column letters, so line numbers line up with actual rows.
|
|
53
|
+
Trailing empty columns/rows (a common Excel "used range" artifact) are trimmed.
|
|
54
|
+
"""
|
|
55
|
+
is_excel = str(path).lower().endswith(('.xls', '.xlsx', '.xlsm'))
|
|
56
|
+
read_header = 0 if header else None
|
|
57
|
+
if is_excel:
|
|
58
|
+
df = pd.read_excel(path, sheet_name=sheet_name if sheet_name is not None else 0, dtype=object, header=read_header)
|
|
59
|
+
else:
|
|
60
|
+
df = pd.read_csv(path, dtype=object, header=read_header)
|
|
61
|
+
df = _trim_grid(df)
|
|
62
|
+
if not header:
|
|
63
|
+
df.columns = [_col_letter(i) for i in range(len(df.columns))]
|
|
64
|
+
return df
|
|
65
|
+
|
|
66
|
+
def _norm(df: pd.DataFrame) -> pd.DataFrame:
|
|
67
|
+
return df.fillna("").astype(str)
|
|
68
|
+
|
|
69
|
+
Change = Tuple[str, Any, Optional[str], Optional[str], Optional[str]]
|
|
70
|
+
|
|
71
|
+
def diff_sheets(left: str, right: str, key: Optional[str] = None, sheet_name: Optional[str] = None) -> List[Change]:
|
|
72
|
+
"""Return list of changes: (type, row_key_or_index, column, old, new).
|
|
73
|
+
Types: 'added_row','removed_row','modified_cell'.
|
|
74
|
+
"""
|
|
75
|
+
L = _norm(_load(left, sheet_name))
|
|
76
|
+
R = _norm(_load(right, sheet_name))
|
|
77
|
+
|
|
78
|
+
changes: List[Change] = []
|
|
79
|
+
if key:
|
|
80
|
+
L2 = L.set_index(key)
|
|
81
|
+
R2 = R.set_index(key)
|
|
82
|
+
all_keys = list(sorted(set(L2.index).union(R2.index)))
|
|
83
|
+
for k in all_keys:
|
|
84
|
+
inL = k in L2.index
|
|
85
|
+
inR = k in R2.index
|
|
86
|
+
if not inL:
|
|
87
|
+
changes.append(("added_row", k, None, None, None))
|
|
88
|
+
continue
|
|
89
|
+
if not inR:
|
|
90
|
+
changes.append(("removed_row", k, None, None, None))
|
|
91
|
+
continue
|
|
92
|
+
lrow = L2.loc[k]
|
|
93
|
+
rrow = R2.loc[k]
|
|
94
|
+
cols = sorted(set(lrow.index).union(rrow.index))
|
|
95
|
+
for c in cols:
|
|
96
|
+
a = str(lrow.get(c, ""))
|
|
97
|
+
b = str(rrow.get(c, ""))
|
|
98
|
+
if a != b:
|
|
99
|
+
changes.append(("modified_cell", k, c, a, b))
|
|
100
|
+
else:
|
|
101
|
+
maxr = max(len(L), len(R))
|
|
102
|
+
for i in range(maxr):
|
|
103
|
+
if i >= len(L):
|
|
104
|
+
changes.append(("added_row", i, None, None, None)); continue
|
|
105
|
+
if i >= len(R):
|
|
106
|
+
changes.append(("removed_row", i, None, None, None)); continue
|
|
107
|
+
lrow = L.iloc[i]
|
|
108
|
+
rrow = R.iloc[i]
|
|
109
|
+
cols = sorted(set(lrow.index).union(rrow.index))
|
|
110
|
+
for c in cols:
|
|
111
|
+
a = str(lrow.get(c, ""))
|
|
112
|
+
b = str(rrow.get(c, ""))
|
|
113
|
+
if a != b:
|
|
114
|
+
changes.append(("modified_cell", i, c, a, b))
|
|
115
|
+
return changes
|
|
116
|
+
|
|
117
|
+
def _is_excel(path: str) -> bool:
|
|
118
|
+
return str(path).lower().endswith(('.xls', '.xlsx', '.xlsm'))
|
|
119
|
+
|
|
120
|
+
def diff_workbook(left: str, right: str, key: Optional[str] = None) -> Dict[str, List[Change]]:
|
|
121
|
+
"""Compare workbooks across all sheets. Returns dict: sheet_name -> changes."""
|
|
122
|
+
# If neither is excel, fallback to single-sheet comparison named 'sheet'
|
|
123
|
+
if not (_is_excel(left) or _is_excel(right)):
|
|
124
|
+
return {"sheet": diff_sheets(left, right, key=key)}
|
|
125
|
+
|
|
126
|
+
left_sheets = []
|
|
127
|
+
right_sheets = []
|
|
128
|
+
if _is_excel(left):
|
|
129
|
+
with pd.ExcelFile(left) as xl:
|
|
130
|
+
left_sheets = xl.sheet_names
|
|
131
|
+
else:
|
|
132
|
+
left_sheets = ["sheet"]
|
|
133
|
+
if _is_excel(right):
|
|
134
|
+
with pd.ExcelFile(right) as xl:
|
|
135
|
+
right_sheets = xl.sheet_names
|
|
136
|
+
else:
|
|
137
|
+
right_sheets = ["sheet"]
|
|
138
|
+
|
|
139
|
+
all_sheets = list(dict.fromkeys(list(left_sheets) + list(right_sheets)))
|
|
140
|
+
result: Dict[str, List[Change]] = {}
|
|
141
|
+
with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as tmp:
|
|
142
|
+
empty_path = None
|
|
143
|
+
|
|
144
|
+
def _empty_csv() -> str:
|
|
145
|
+
nonlocal empty_path
|
|
146
|
+
if empty_path is None:
|
|
147
|
+
empty_path = os.path.join(tmp, "empty.csv")
|
|
148
|
+
pd.DataFrame().to_csv(empty_path, index=False)
|
|
149
|
+
return empty_path
|
|
150
|
+
|
|
151
|
+
for s in all_sheets:
|
|
152
|
+
# If a sheet is missing in one workbook, represent as added/removed row marker
|
|
153
|
+
Lpath = left if s in left_sheets else _empty_csv()
|
|
154
|
+
Rpath = right if s in right_sheets else _empty_csv()
|
|
155
|
+
result[s] = diff_sheets(Lpath, Rpath, key=key, sheet_name=s)
|
|
156
|
+
return result
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def compare_workbooks(left: str, right: str, key: Optional[str] = None) -> Dict[str, List[Change]]:
|
|
160
|
+
"""High-level comparator that prefers a Rust backend when available.
|
|
161
|
+
|
|
162
|
+
Returns a mapping of sheet_name -> list of Change tuples.
|
|
163
|
+
"""
|
|
164
|
+
if _RUST_AVAILABLE:
|
|
165
|
+
try:
|
|
166
|
+
# rust function returns JSON string report
|
|
167
|
+
raw = _rust.compare_workbooks_json(left, right, None, None)
|
|
168
|
+
report = json.loads(raw)
|
|
169
|
+
out: Dict[str, List[Change]] = {}
|
|
170
|
+
for sheet in report.get("sheets", []):
|
|
171
|
+
name = sheet.get("sheet_name")
|
|
172
|
+
deltas = []
|
|
173
|
+
for d in sheet.get("cell_deltas", []):
|
|
174
|
+
status = d.get("status")
|
|
175
|
+
row_old = d.get("row_idx_old")
|
|
176
|
+
row_new = d.get("row_idx_new")
|
|
177
|
+
col = d.get("col_idx")
|
|
178
|
+
oldv = d.get("old_value")
|
|
179
|
+
newv = d.get("new_value")
|
|
180
|
+
if status == "Added":
|
|
181
|
+
deltas.append(("added_row", row_new, None, None, None))
|
|
182
|
+
elif status == "Deleted":
|
|
183
|
+
deltas.append(("removed_row", row_old, None, None, None))
|
|
184
|
+
else:
|
|
185
|
+
deltas.append(("modified_cell", row_new if row_new is not None else row_old, col, oldv, newv))
|
|
186
|
+
out[name] = deltas
|
|
187
|
+
return out
|
|
188
|
+
except Exception:
|
|
189
|
+
# On any rust failure, fallback to Python implementation
|
|
190
|
+
pass
|
|
191
|
+
|
|
192
|
+
return diff_workbook(left, right, key=key)
|
|
193
|
+
|
|
194
|
+
def diff_rows_full(
|
|
195
|
+
left: str,
|
|
196
|
+
right: str,
|
|
197
|
+
key: Optional[str] = None,
|
|
198
|
+
sheet_name: Optional[str] = None,
|
|
199
|
+
header: bool = False,
|
|
200
|
+
left_exists: bool = True,
|
|
201
|
+
right_exists: bool = True,
|
|
202
|
+
) -> Dict[str, Any]:
|
|
203
|
+
"""Full row-level diff for GitHub-style rendering.
|
|
204
|
+
|
|
205
|
+
By default every spreadsheet row is treated as data (no assumed header row),
|
|
206
|
+
so line numbers match real Excel row numbers and columns are labelled A, B, C...
|
|
207
|
+
Pass header=True to treat row 1 as column names instead (then `key` is a column name).
|
|
208
|
+
|
|
209
|
+
left_exists/right_exists: pass False when the sheet is entirely absent from
|
|
210
|
+
that workbook (e.g. a sheet that was added or deleted) - every row of the
|
|
211
|
+
other side is then reported as added/removed rather than raising.
|
|
212
|
+
|
|
213
|
+
Returns {"columns": [...], "rows": [{"status": "unchanged"|"added"|"removed"|"modified",
|
|
214
|
+
"left": {col: val}, "right": {col: val}, "changed": [col, ...]}, ...]}.
|
|
215
|
+
"""
|
|
216
|
+
L = _norm(_load_grid(left, sheet_name, header=header)) if left_exists else pd.DataFrame()
|
|
217
|
+
R = _norm(_load_grid(right, sheet_name, header=header)) if right_exists else pd.DataFrame()
|
|
218
|
+
columns = list(dict.fromkeys(list(L.columns) + list(R.columns)))
|
|
219
|
+
rows: List[Dict[str, Any]] = []
|
|
220
|
+
|
|
221
|
+
def row_dict(row, key_val=None) -> Dict[str, str]:
|
|
222
|
+
d = {c: str(row.get(c, "")) for c in columns}
|
|
223
|
+
if key_val is not None:
|
|
224
|
+
d[key] = str(key_val)
|
|
225
|
+
return d
|
|
226
|
+
|
|
227
|
+
if key and key in L.columns and key in R.columns:
|
|
228
|
+
L2, R2 = L.set_index(key), R.set_index(key)
|
|
229
|
+
for k in sorted(set(L2.index) | set(R2.index)):
|
|
230
|
+
inL, inR = k in L2.index, k in R2.index
|
|
231
|
+
if inL and not inR:
|
|
232
|
+
rows.append({"status": "removed", "left": row_dict(L2.loc[k], k), "right": None, "changed": []})
|
|
233
|
+
elif inR and not inL:
|
|
234
|
+
rows.append({"status": "added", "left": None, "right": row_dict(R2.loc[k], k), "changed": []})
|
|
235
|
+
else:
|
|
236
|
+
lrow, rrow = row_dict(L2.loc[k], k), row_dict(R2.loc[k], k)
|
|
237
|
+
changed = [c for c in columns if lrow.get(c) != rrow.get(c)]
|
|
238
|
+
rows.append({"status": "modified" if changed else "unchanged", "left": lrow, "right": rrow, "changed": changed})
|
|
239
|
+
else:
|
|
240
|
+
maxr = max(len(L), len(R))
|
|
241
|
+
for i in range(maxr):
|
|
242
|
+
if i >= len(L):
|
|
243
|
+
rows.append({"status": "added", "left": None, "right": row_dict(R.iloc[i]), "changed": []}); continue
|
|
244
|
+
if i >= len(R):
|
|
245
|
+
rows.append({"status": "removed", "left": row_dict(L.iloc[i]), "right": None, "changed": []}); continue
|
|
246
|
+
lrow, rrow = row_dict(L.iloc[i]), row_dict(R.iloc[i])
|
|
247
|
+
changed = [c for c in columns if lrow.get(c) != rrow.get(c)]
|
|
248
|
+
rows.append({"status": "modified" if changed else "unchanged", "left": lrow, "right": rrow, "changed": changed})
|
|
249
|
+
|
|
250
|
+
old_ln = new_ln = 0
|
|
251
|
+
for r in rows:
|
|
252
|
+
if r["status"] in ("removed", "modified", "unchanged"):
|
|
253
|
+
old_ln += 1
|
|
254
|
+
r["old_line"] = old_ln
|
|
255
|
+
else:
|
|
256
|
+
r["old_line"] = None
|
|
257
|
+
if r["status"] in ("added", "modified", "unchanged"):
|
|
258
|
+
new_ln += 1
|
|
259
|
+
r["new_line"] = new_ln
|
|
260
|
+
else:
|
|
261
|
+
r["new_line"] = None
|
|
262
|
+
|
|
263
|
+
return {"columns": columns, "rows": rows}
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def diff_multi_grid(
|
|
267
|
+
paths: List[str],
|
|
268
|
+
present: List[bool],
|
|
269
|
+
key: Optional[str] = None,
|
|
270
|
+
sheet_name: Optional[str] = None,
|
|
271
|
+
header: bool = False,
|
|
272
|
+
) -> Dict[str, Any]:
|
|
273
|
+
"""N-way row-aligned comparison across several files, for a side-by-side split view.
|
|
274
|
+
|
|
275
|
+
`present[i]` is False when this sheet doesn't exist in paths[i] at all.
|
|
276
|
+
The first file with the sheet present is the baseline; any other file's
|
|
277
|
+
cell that differs from the baseline value (when both have the row) is flagged.
|
|
278
|
+
|
|
279
|
+
Returns {"columns": [...], "rows": [{"line": n, "status": "same"|"diff",
|
|
280
|
+
"values": [dict|None, ...], "changed": [set of col per file index], "missing": [bool,...]}]}.
|
|
281
|
+
"""
|
|
282
|
+
frames = []
|
|
283
|
+
for p, ok in zip(paths, present):
|
|
284
|
+
frames.append(_norm(_load_grid(p, sheet_name, header=header)) if ok else pd.DataFrame())
|
|
285
|
+
|
|
286
|
+
columns: List[str] = []
|
|
287
|
+
for f in frames:
|
|
288
|
+
for c in f.columns:
|
|
289
|
+
if c not in columns:
|
|
290
|
+
columns.append(c)
|
|
291
|
+
|
|
292
|
+
def row_dict(row, key_val=None) -> Dict[str, str]:
|
|
293
|
+
d = {c: str(row.get(c, "")) for c in columns}
|
|
294
|
+
if key_val is not None and key:
|
|
295
|
+
d[key] = str(key_val)
|
|
296
|
+
return d
|
|
297
|
+
|
|
298
|
+
rows: List[Dict[str, Any]] = []
|
|
299
|
+
use_key = bool(key) and any(key in f.columns for f in frames if not f.empty)
|
|
300
|
+
|
|
301
|
+
if use_key:
|
|
302
|
+
indexed = [f.set_index(key) if (not f.empty and key in f.columns) else f for f in frames]
|
|
303
|
+
all_keys = sorted({k for f in indexed for k in (f.index if not f.empty else [])})
|
|
304
|
+
for k in all_keys:
|
|
305
|
+
values = []
|
|
306
|
+
for f in indexed:
|
|
307
|
+
has = (not f.empty) and (k in getattr(f, "index", []))
|
|
308
|
+
values.append(row_dict(f.loc[k], k) if has else None)
|
|
309
|
+
rows.append(_build_multi_row(k, values, columns))
|
|
310
|
+
else:
|
|
311
|
+
maxr = max((len(f) for f in frames), default=0)
|
|
312
|
+
for i in range(maxr):
|
|
313
|
+
values = [row_dict(f.iloc[i]) if i < len(f) else None for f in frames]
|
|
314
|
+
rows.append(_build_multi_row(i, values, columns))
|
|
315
|
+
|
|
316
|
+
return {"columns": columns, "rows": rows, "file_count": len(paths)}
|
|
317
|
+
|
|
318
|
+
def _build_multi_row(idx, values: List[Optional[Dict[str, str]]], columns: List[str]) -> Dict[str, Any]:
|
|
319
|
+
baseline = next((v for v in values if v is not None), None)
|
|
320
|
+
changed = set()
|
|
321
|
+
if baseline is not None:
|
|
322
|
+
for v in values:
|
|
323
|
+
if v is None:
|
|
324
|
+
continue
|
|
325
|
+
for c in columns:
|
|
326
|
+
if v.get(c, "") != baseline.get(c, ""):
|
|
327
|
+
changed.add(c)
|
|
328
|
+
missing = [v is None for v in values]
|
|
329
|
+
if missing[0] and not all(missing):
|
|
330
|
+
status = "added"
|
|
331
|
+
elif not missing[0] and len(missing) > 1 and all(missing[1:]):
|
|
332
|
+
status = "removed"
|
|
333
|
+
elif changed or any(missing):
|
|
334
|
+
status = "diff"
|
|
335
|
+
else:
|
|
336
|
+
status = "same"
|
|
337
|
+
return {"line": idx, "status": status, "cells": values, "changed": sorted(changed), "missing": missing, "baseline": baseline}
|
|
338
|
+
|
|
339
|
+
def format_unified(changes: List[Change]) -> str:
|
|
340
|
+
out: List[str] = []
|
|
341
|
+
for t, r, col, a, b in changes:
|
|
342
|
+
if t == "added_row":
|
|
343
|
+
out.append(f"+ ROW {r}")
|
|
344
|
+
elif t == "removed_row":
|
|
345
|
+
out.append(f"- ROW {r}")
|
|
346
|
+
else:
|
|
347
|
+
out.append(f"- {r} | {col} = {a}")
|
|
348
|
+
out.append(f"+ {r} | {col} = {b}")
|
|
349
|
+
return "\n".join(out)
|
|
350
|
+
|
|
351
|
+
def format_workbook(changes_map: Dict[str, List[Change]]) -> str:
|
|
352
|
+
parts: List[str] = []
|
|
353
|
+
for sheet, changes in changes_map.items():
|
|
354
|
+
parts.append(f"--- Sheet: {sheet} ---")
|
|
355
|
+
parts.append(format_unified(changes) or "(no changes)")
|
|
356
|
+
return "\n\n".join(parts)
|