extended_stats 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- extended_stats/__about__.py +2 -0
- extended_stats/__init__.py +14 -0
- extended_stats/cleaning/__init__.py +4 -0
- extended_stats/cleaning/core.py +40 -0
- extended_stats/diffing/__init__.py +5 -0
- extended_stats/diffing/core.py +176 -0
- extended_stats/diffing/narrative.py +47 -0
- extended_stats/profiling/__init__.py +4 -0
- extended_stats/profiling/core.py +133 -0
- extended_stats/py.typed +1 -0
- extended_stats/utils.py +2 -0
- extended_stats-0.1.0.dist-info/METADATA +103 -0
- extended_stats-0.1.0.dist-info/RECORD +15 -0
- extended_stats-0.1.0.dist-info/WHEEL +4 -0
- extended_stats-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
from .diffing import diff_datasets, summarize_diff, DiffReport
|
|
2
|
+
from .profiling import profile_dataset, DatasetProfile
|
|
3
|
+
from .cleaning import drop_duplicate_rows, standardize_column_names
|
|
4
|
+
from .__about__ import __version__
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"diff_datasets",
|
|
8
|
+
"summarize_diff",
|
|
9
|
+
"DiffReport",
|
|
10
|
+
"profile_dataset",
|
|
11
|
+
"DatasetProfile",
|
|
12
|
+
"drop_duplicate_rows",
|
|
13
|
+
"standardize_column_names",
|
|
14
|
+
]
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Deterministic DataFrame cleaning transformations."""
|
|
2
|
+
|
|
3
|
+
import pandas as pd
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def drop_duplicate_rows(df: pd.DataFrame) -> pd.DataFrame:
|
|
7
|
+
"""Return a copy of df with fully duplicate rows removed.
|
|
8
|
+
|
|
9
|
+
Parameters
|
|
10
|
+
----------
|
|
11
|
+
df : pd.DataFrame
|
|
12
|
+
|
|
13
|
+
Returns
|
|
14
|
+
-------
|
|
15
|
+
pd.DataFrame
|
|
16
|
+
A new DataFrame with duplicate rows dropped (keeps first
|
|
17
|
+
occurrence). The index is reset.
|
|
18
|
+
"""
|
|
19
|
+
return df.drop_duplicates().reset_index(drop=True)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def standardize_column_names(df: pd.DataFrame) -> pd.DataFrame:
|
|
23
|
+
"""Return a copy of df with column names lowercased, stripped,
|
|
24
|
+
and spaces replaced with underscores.
|
|
25
|
+
|
|
26
|
+
Parameters
|
|
27
|
+
----------
|
|
28
|
+
df : pd.DataFrame
|
|
29
|
+
|
|
30
|
+
Returns
|
|
31
|
+
-------
|
|
32
|
+
pd.DataFrame
|
|
33
|
+
A new DataFrame with cleaned column names. Original df is
|
|
34
|
+
not modified.
|
|
35
|
+
"""
|
|
36
|
+
new_df = df.copy()
|
|
37
|
+
new_df.columns = (
|
|
38
|
+
new_df.columns.str.strip().str.lower().str.replace(r"\s+", "_", regex=True)
|
|
39
|
+
)
|
|
40
|
+
return new_df
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""Core dataset comparison logic."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class DiffReport:
|
|
10
|
+
"""Holds the structured result of comparing two DataFrames.
|
|
11
|
+
|
|
12
|
+
Attributes
|
|
13
|
+
----------
|
|
14
|
+
rows_added : int
|
|
15
|
+
rows_removed : int
|
|
16
|
+
columns_added : list[str]
|
|
17
|
+
columns_removed : list[str]
|
|
18
|
+
dtype_changes : dict[str, tuple]
|
|
19
|
+
column -> (old_dtype, new_dtype)
|
|
20
|
+
value_diffs : pd.DataFrame
|
|
21
|
+
Tidy frame of (key, column, old_value, new_value) for matched
|
|
22
|
+
rows whose values changed. Empty if no key was provided.
|
|
23
|
+
stat_shifts : pd.DataFrame
|
|
24
|
+
Per numeric column, mean/median/null-count shift between old
|
|
25
|
+
and new.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
rows_added: int
|
|
29
|
+
rows_removed: int
|
|
30
|
+
columns_added: list = field(default_factory=list)
|
|
31
|
+
columns_removed: list = field(default_factory=list)
|
|
32
|
+
dtype_changes: dict = field(default_factory=dict)
|
|
33
|
+
value_diffs: pd.DataFrame = field(default_factory=pd.DataFrame, repr=False)
|
|
34
|
+
stat_shifts: pd.DataFrame = field(default_factory=pd.DataFrame, repr=False)
|
|
35
|
+
|
|
36
|
+
def __repr__(self) -> str:
|
|
37
|
+
return (
|
|
38
|
+
f"DiffReport(rows_added={self.rows_added}, "
|
|
39
|
+
f"rows_removed={self.rows_removed}, "
|
|
40
|
+
f"columns_added={self.columns_added}, "
|
|
41
|
+
f"columns_removed={self.columns_removed}, "
|
|
42
|
+
f"dtype_changes={self.dtype_changes}, "
|
|
43
|
+
f"value_diffs={len(self.value_diffs)} cell(s), "
|
|
44
|
+
f"stat_shifts={len(self.stat_shifts)} column(s))"
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
def to_dict(self) -> dict:
|
|
48
|
+
return {
|
|
49
|
+
"rows_added": self.rows_added,
|
|
50
|
+
"rows_removed": self.rows_removed,
|
|
51
|
+
"columns_added": self.columns_added,
|
|
52
|
+
"columns_removed": self.columns_removed,
|
|
53
|
+
"dtype_changes": self.dtype_changes,
|
|
54
|
+
"value_diffs": self.value_diffs.to_dict(orient="records"),
|
|
55
|
+
"stat_shifts": self.stat_shifts.to_dict(orient="index"),
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _compare_row_counts(df_old: pd.DataFrame, df_new: pd.DataFrame, key):
|
|
60
|
+
if key is None:
|
|
61
|
+
old_n, new_n = len(df_old), len(df_new)
|
|
62
|
+
added = max(new_n - old_n, 0)
|
|
63
|
+
removed = max(old_n - new_n, 0)
|
|
64
|
+
return {"added": added, "removed": removed}
|
|
65
|
+
|
|
66
|
+
keys = [key] if isinstance(key, str) else list(key)
|
|
67
|
+
old_keys = (
|
|
68
|
+
set(map(tuple, df_old[keys].values)) if len(keys) > 1 else set(df_old[keys[0]])
|
|
69
|
+
)
|
|
70
|
+
new_keys = (
|
|
71
|
+
set(map(tuple, df_new[keys].values)) if len(keys) > 1 else set(df_new[keys[0]])
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
return {
|
|
75
|
+
"added": len(new_keys - old_keys),
|
|
76
|
+
"removed": len(old_keys - new_keys),
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _compare_columns(df_old: pd.DataFrame, df_new: pd.DataFrame):
|
|
81
|
+
old_cols, new_cols = set(df_old.columns), set(df_new.columns)
|
|
82
|
+
added = sorted(new_cols - old_cols)
|
|
83
|
+
removed = sorted(old_cols - new_cols)
|
|
84
|
+
|
|
85
|
+
dtype_changes = {}
|
|
86
|
+
for col in old_cols & new_cols:
|
|
87
|
+
old_dtype, new_dtype = str(df_old[col].dtype), str(df_new[col].dtype)
|
|
88
|
+
if old_dtype != new_dtype:
|
|
89
|
+
dtype_changes[col] = (old_dtype, new_dtype)
|
|
90
|
+
|
|
91
|
+
return {"added": added, "removed": removed, "dtype_changes": dtype_changes}
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _compare_values(df_old: pd.DataFrame, df_new: pd.DataFrame, key):
|
|
95
|
+
if key is None:
|
|
96
|
+
return pd.DataFrame(columns=["key", "column", "old_value", "new_value"])
|
|
97
|
+
|
|
98
|
+
keys = [key] if isinstance(key, str) else list(key)
|
|
99
|
+
merged = df_old.merge(df_new, on=keys, suffixes=("_old", "_new"), how="inner")
|
|
100
|
+
|
|
101
|
+
shared_cols = (set(df_old.columns) & set(df_new.columns)) - set(keys)
|
|
102
|
+
records = []
|
|
103
|
+
for col in shared_cols:
|
|
104
|
+
old_col, new_col = f"{col}_old", f"{col}_new"
|
|
105
|
+
changed = merged[merged[old_col] != merged[new_col]]
|
|
106
|
+
for _, row in changed.iterrows():
|
|
107
|
+
key_val = tuple(row[k] for k in keys) if len(keys) > 1 else row[keys[0]]
|
|
108
|
+
records.append(
|
|
109
|
+
{
|
|
110
|
+
"key": key_val,
|
|
111
|
+
"column": col,
|
|
112
|
+
"old_value": row[old_col],
|
|
113
|
+
"new_value": row[new_col],
|
|
114
|
+
}
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
return pd.DataFrame(records, columns=["key", "column", "old_value", "new_value"])
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _compare_summary_stats(df_old: pd.DataFrame, df_new: pd.DataFrame):
|
|
121
|
+
shared_numeric = [
|
|
122
|
+
col
|
|
123
|
+
for col in set(df_old.columns) & set(df_new.columns)
|
|
124
|
+
if pd.api.types.is_numeric_dtype(df_old[col])
|
|
125
|
+
and pd.api.types.is_numeric_dtype(df_new[col])
|
|
126
|
+
]
|
|
127
|
+
|
|
128
|
+
rows = []
|
|
129
|
+
for col in shared_numeric:
|
|
130
|
+
rows.append(
|
|
131
|
+
{
|
|
132
|
+
"column": col,
|
|
133
|
+
"mean_old": df_old[col].mean(),
|
|
134
|
+
"mean_new": df_new[col].mean(),
|
|
135
|
+
"median_old": df_old[col].median(),
|
|
136
|
+
"median_new": df_new[col].median(),
|
|
137
|
+
"nulls_old": int(df_old[col].isnull().sum()),
|
|
138
|
+
"nulls_new": int(df_new[col].isnull().sum()),
|
|
139
|
+
}
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
return pd.DataFrame(rows).set_index("column") if rows else pd.DataFrame()
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def diff_datasets(df_old: pd.DataFrame, df_new: pd.DataFrame, key=None) -> DiffReport:
|
|
146
|
+
"""Compare two DataFrames and return a structured DiffReport.
|
|
147
|
+
|
|
148
|
+
Parameters
|
|
149
|
+
----------
|
|
150
|
+
df_old : pd.DataFrame
|
|
151
|
+
The "before" version of the dataset.
|
|
152
|
+
df_new : pd.DataFrame
|
|
153
|
+
The "after" version of the dataset.
|
|
154
|
+
key : str or list of str, optional
|
|
155
|
+
Column(s) to match rows on. If omitted, row-level value diffs
|
|
156
|
+
are skipped and only row counts, column changes, and summary
|
|
157
|
+
stat shifts are computed.
|
|
158
|
+
|
|
159
|
+
Returns
|
|
160
|
+
-------
|
|
161
|
+
DiffReport
|
|
162
|
+
"""
|
|
163
|
+
row_diff = _compare_row_counts(df_old, df_new, key)
|
|
164
|
+
col_diff = _compare_columns(df_old, df_new)
|
|
165
|
+
value_diffs = _compare_values(df_old, df_new, key)
|
|
166
|
+
stat_shifts = _compare_summary_stats(df_old, df_new)
|
|
167
|
+
|
|
168
|
+
return DiffReport(
|
|
169
|
+
rows_added=row_diff["added"],
|
|
170
|
+
rows_removed=row_diff["removed"],
|
|
171
|
+
columns_added=col_diff["added"],
|
|
172
|
+
columns_removed=col_diff["removed"],
|
|
173
|
+
dtype_changes=col_diff["dtype_changes"],
|
|
174
|
+
value_diffs=value_diffs,
|
|
175
|
+
stat_shifts=stat_shifts,
|
|
176
|
+
)
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Human-readable summaries of a DiffReport."""
|
|
2
|
+
|
|
3
|
+
from .core import DiffReport
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def summarize_diff(report: DiffReport) -> str:
|
|
7
|
+
"""Return a plain-language summary of a DiffReport.
|
|
8
|
+
|
|
9
|
+
Parameters
|
|
10
|
+
----------
|
|
11
|
+
report : DiffReport
|
|
12
|
+
|
|
13
|
+
Returns
|
|
14
|
+
-------
|
|
15
|
+
str
|
|
16
|
+
"""
|
|
17
|
+
lines = []
|
|
18
|
+
|
|
19
|
+
if report.rows_added or report.rows_removed:
|
|
20
|
+
lines.append(
|
|
21
|
+
f"{report.rows_added} row(s) added, {report.rows_removed} row(s) removed."
|
|
22
|
+
)
|
|
23
|
+
else:
|
|
24
|
+
lines.append("No row count changes.")
|
|
25
|
+
|
|
26
|
+
if report.columns_added:
|
|
27
|
+
lines.append(f"Columns added: {', '.join(report.columns_added)}.")
|
|
28
|
+
if report.columns_removed:
|
|
29
|
+
lines.append(f"Columns removed: {', '.join(report.columns_removed)}.")
|
|
30
|
+
if report.dtype_changes:
|
|
31
|
+
changes = [
|
|
32
|
+
f"{col} ({old} -> {new})"
|
|
33
|
+
for col, (old, new) in report.dtype_changes.items()
|
|
34
|
+
]
|
|
35
|
+
lines.append(f"Dtype changes: {', '.join(changes)}.")
|
|
36
|
+
|
|
37
|
+
if not report.value_diffs.empty:
|
|
38
|
+
changed_cols = report.value_diffs["column"].value_counts()
|
|
39
|
+
parts = [f"'{col}' changed in {n} row(s)" for col, n in changed_cols.items()]
|
|
40
|
+
lines.append("Value changes: " + "; ".join(parts) + ".")
|
|
41
|
+
|
|
42
|
+
if not report.stat_shifts.empty:
|
|
43
|
+
for col, row in report.stat_shifts.iterrows():
|
|
44
|
+
mean_delta = row["mean_new"] - row["mean_old"]
|
|
45
|
+
lines.append(f"'{col}' mean shifted by {mean_delta:+.2f}.")
|
|
46
|
+
|
|
47
|
+
return " ".join(lines)
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""Combined describe()/info()-style dataset profiling."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
|
|
5
|
+
import pandas as pd
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass
|
|
9
|
+
class DatasetProfile:
|
|
10
|
+
"""Holds a structured profile of a DataFrame.
|
|
11
|
+
|
|
12
|
+
Attributes
|
|
13
|
+
----------
|
|
14
|
+
shape : tuple
|
|
15
|
+
(n_rows, n_columns)
|
|
16
|
+
memory_usage_bytes : int
|
|
17
|
+
Total memory usage of the DataFrame in bytes.
|
|
18
|
+
duplicated_rows : int
|
|
19
|
+
Count of fully duplicated rows.
|
|
20
|
+
column_summary : pd.DataFrame
|
|
21
|
+
One row per column with dtype, null count, null %, unique
|
|
22
|
+
count, and (for numeric columns) mean/median/std/min/max, or
|
|
23
|
+
(for non-numeric columns) the most frequent value and its
|
|
24
|
+
frequency.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
shape: tuple
|
|
28
|
+
memory_usage_bytes: int
|
|
29
|
+
duplicated_rows: int
|
|
30
|
+
column_summary: pd.DataFrame = field(repr=False)
|
|
31
|
+
|
|
32
|
+
def __repr__(self) -> str:
|
|
33
|
+
lines = [
|
|
34
|
+
f"DatasetProfile(rows={self.shape[0]}, columns={self.shape[1]}, "
|
|
35
|
+
f"memory={self.memory_usage_bytes / 1024:.1f} KB, "
|
|
36
|
+
f"duplicated_rows={self.duplicated_rows})",
|
|
37
|
+
"",
|
|
38
|
+
self.column_summary.to_string(),
|
|
39
|
+
]
|
|
40
|
+
return "\n".join(lines)
|
|
41
|
+
|
|
42
|
+
def to_dict(self) -> dict:
|
|
43
|
+
return {
|
|
44
|
+
"shape": self.shape,
|
|
45
|
+
"memory_usage_bytes": self.memory_usage_bytes,
|
|
46
|
+
"duplicated_rows": self.duplicated_rows,
|
|
47
|
+
"column_summary": self.column_summary.to_dict(orient="index"),
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def profile_dataset(df: pd.DataFrame) -> DatasetProfile:
|
|
52
|
+
"""Produce a combined describe()/info()-style profile of a DataFrame.
|
|
53
|
+
|
|
54
|
+
Parameters
|
|
55
|
+
----------
|
|
56
|
+
df : pd.DataFrame
|
|
57
|
+
|
|
58
|
+
Returns
|
|
59
|
+
-------
|
|
60
|
+
DatasetProfile
|
|
61
|
+
"""
|
|
62
|
+
rows = []
|
|
63
|
+
for col in df.columns:
|
|
64
|
+
series = df[col]
|
|
65
|
+
null_count = series.isnull().sum()
|
|
66
|
+
null_pct = round(100 * null_count / len(df), 2) if len(df) else 0.0
|
|
67
|
+
|
|
68
|
+
row = {
|
|
69
|
+
"column": col,
|
|
70
|
+
"dtype": str(series.dtype),
|
|
71
|
+
"non_null_count": series.notnull().sum(),
|
|
72
|
+
"null_count": null_count,
|
|
73
|
+
"null_pct": null_pct,
|
|
74
|
+
"unique_count": series.nunique(),
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
if pd.api.types.is_numeric_dtype(series):
|
|
78
|
+
row.update(
|
|
79
|
+
{
|
|
80
|
+
"mean": series.mean(),
|
|
81
|
+
"median": series.median(),
|
|
82
|
+
"std": series.std(),
|
|
83
|
+
"min": series.min(),
|
|
84
|
+
"max": series.max(),
|
|
85
|
+
"top_value": None,
|
|
86
|
+
"top_freq": None,
|
|
87
|
+
}
|
|
88
|
+
)
|
|
89
|
+
else:
|
|
90
|
+
value_counts = series.value_counts()
|
|
91
|
+
top_value = value_counts.index[0] if not value_counts.empty else None
|
|
92
|
+
top_freq = value_counts.iloc[0] if not value_counts.empty else None
|
|
93
|
+
row.update(
|
|
94
|
+
{
|
|
95
|
+
"mean": None,
|
|
96
|
+
"median": None,
|
|
97
|
+
"std": None,
|
|
98
|
+
"min": None,
|
|
99
|
+
"max": None,
|
|
100
|
+
"top_value": top_value,
|
|
101
|
+
"top_freq": top_freq,
|
|
102
|
+
}
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
rows.append(row)
|
|
106
|
+
|
|
107
|
+
if rows:
|
|
108
|
+
column_summary = pd.DataFrame(rows).set_index("column")
|
|
109
|
+
else:
|
|
110
|
+
column_summary = pd.DataFrame(
|
|
111
|
+
columns=[
|
|
112
|
+
"dtype",
|
|
113
|
+
"non_null_count",
|
|
114
|
+
"null_count",
|
|
115
|
+
"null_pct",
|
|
116
|
+
"unique_count",
|
|
117
|
+
"mean",
|
|
118
|
+
"median",
|
|
119
|
+
"std",
|
|
120
|
+
"min",
|
|
121
|
+
"max",
|
|
122
|
+
"top_value",
|
|
123
|
+
"top_freq",
|
|
124
|
+
]
|
|
125
|
+
)
|
|
126
|
+
column_summary.index.name = "column"
|
|
127
|
+
|
|
128
|
+
return DatasetProfile(
|
|
129
|
+
shape=df.shape,
|
|
130
|
+
memory_usage_bytes=int(df.memory_usage(deep=True).sum()),
|
|
131
|
+
duplicated_rows=int(df.duplicated().sum()),
|
|
132
|
+
column_summary=column_summary,
|
|
133
|
+
)
|
extended_stats/py.typed
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# Marker file for PEP 561
|
extended_stats/utils.py
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: extended_stats
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Python Boilerplate contains all the boilerplate you need to create a Python package.
|
|
5
|
+
Project-URL: bugs, https://github.com/dreireyez/extended_stats/issues
|
|
6
|
+
Project-URL: changelog, https://github.com/dreireyez/extended_stats/releases
|
|
7
|
+
Project-URL: documentation, https://dreireyez.github.io/extended_stats/
|
|
8
|
+
Project-URL: homepage, https://github.com/dreireyez/extended_stats
|
|
9
|
+
Author-email: Djem Andreif Reyes <contact.dafreyes@gmail.com>
|
|
10
|
+
Maintainer-email: Djem Andreif Reyes <contact.dafreyes@gmail.com>
|
|
11
|
+
License: MIT
|
|
12
|
+
License-File: LICENSE
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Typing :: Typed
|
|
16
|
+
Requires-Python: >=3.9
|
|
17
|
+
Requires-Dist: numpy>=1.24.0
|
|
18
|
+
Requires-Dist: pandas>=2.0.0
|
|
19
|
+
Requires-Dist: rich
|
|
20
|
+
Requires-Dist: typer
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
|
|
23
|
+
# Extended Statistics
|
|
24
|
+
|
|
25
|
+
[](https://pypi.org/project/extended_stats/)
|
|
26
|
+
[](https://pepy.tech/projects/extended_stats)
|
|
27
|
+
|
|
28
|
+
**Extended Statistics** is a lightweight toolkit for comparing, profiling, and cleaning pandas DataFrames. It combines dataset diffing, an extended summary, and a handful of common cleaning transformations into a single, subpackage-organized API — built as a proof of concept for practicing Python package development fundamentals (structure, subpackages, testing, and packaging).
|
|
29
|
+
|
|
30
|
+
* [GitHub](https://github.com/dreireyez/extended_stats/) | [PyPI](https://pypi.org/project/extended_stats/) | [Documentation](https://dreireyez.github.io/extended_stats/)
|
|
31
|
+
* Created by **Djem Andreif Reyes**
|
|
32
|
+
* MIT License
|
|
33
|
+
|
|
34
|
+
## Features
|
|
35
|
+
|
|
36
|
+
* **Dataset diffing:** Compare two versions of a DataFrame and get a structured report of what changed: rows added/removed, columns added/removed, dtype changes, per-cell value changes (when a key column is provided), and shifts in summary statistics (mean/median/nulls) for numeric columns.
|
|
37
|
+
* **Human-readable diff summaries:** Turn a diff report into a plain-language sentence summary, no manual interpretation required.
|
|
38
|
+
* **Dataset profiling:** A single function that merges the useful parts of .describe() and .info() into one report: shape, memory usage, duplicate row count, and per-column stats (nulls, uniques, mean/median/std for numeric columns, top value/frequency for categorical columns).
|
|
39
|
+
* **Basic cleaning utilities:** Drop fully duplicate rows and standardize column names (lowercase, trimmed, underscore-separated) in one call each.
|
|
40
|
+
* **Subpackage-organized API:** Diffing, profiling, and cleaning each live in their own subpackage, with a top-level re-export layer so functions are accessible.
|
|
41
|
+
|
|
42
|
+
## Installation
|
|
43
|
+
|
|
44
|
+
### Stable Release
|
|
45
|
+
|
|
46
|
+
To install **Extended Statistics**, run this command in your terminal:
|
|
47
|
+
|
|
48
|
+
```sh
|
|
49
|
+
uv add extended_stats
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Or if you prefer to use `pip`:
|
|
53
|
+
|
|
54
|
+
```sh
|
|
55
|
+
pip install extended_stats
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## From Source
|
|
59
|
+
|
|
60
|
+
The source files for Extended Statistics can be downloaded from the [Github repo](https://github.com/dreireyez/extended_stats).
|
|
61
|
+
|
|
62
|
+
You can either clone the public repository:
|
|
63
|
+
|
|
64
|
+
```sh
|
|
65
|
+
git clone https://github.com/dreireyez/extended_stats
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Or download the [tarball](https://github.com/dreireyez/extended_stats/tarball/main):
|
|
69
|
+
|
|
70
|
+
```sh
|
|
71
|
+
curl -OJL https://github.com/dreireyez/extended_stats/tarball/main
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Once you have a copy of the source, you can install it with:
|
|
75
|
+
|
|
76
|
+
```sh
|
|
77
|
+
cd extended_stats
|
|
78
|
+
uv sync
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Usage
|
|
82
|
+
|
|
83
|
+
To use **Extended Statistics** in a project:
|
|
84
|
+
|
|
85
|
+
```python
|
|
86
|
+
import extended_stats
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## Documentation
|
|
90
|
+
|
|
91
|
+
Full documentation is available on
|
|
92
|
+
[GitHub Pages](https://dreireyez.github.io/extended_stats/).
|
|
93
|
+
|
|
94
|
+
## Contributing
|
|
95
|
+
|
|
96
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md) for development setup, testing, and
|
|
97
|
+
documentation instructions.
|
|
98
|
+
|
|
99
|
+
## Author
|
|
100
|
+
|
|
101
|
+
**Extended Statistics** was created in **2026** by **Djem Andreif Reyes**.
|
|
102
|
+
|
|
103
|
+
GitHub [@dreireyez](https://github.com/dreireyez) | PyPI [@dreireyez](https://pypi.org/user/dreireyez/)
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
extended_stats/__about__.py,sha256=iEaqPDduXz5v3Haiav21xA_XJhpc3_g4d5ooEglsvmY,37
|
|
2
|
+
extended_stats/__init__.py,sha256=2f9V2FuobQhFnguogvCLkrVvFQJxjk7pu5eYZoO7bLQ,415
|
|
3
|
+
extended_stats/py.typed,sha256=TBDV9m9vjfnp9vsgTeJmpoNseoJEfHwvZaBLvLMfzz8,27
|
|
4
|
+
extended_stats/utils.py,sha256=xnQZ-yxjfHFE5dIwg6oa73ocytJ0DV69K4gFB5gfcQE,87
|
|
5
|
+
extended_stats/cleaning/__init__.py,sha256=DUdlf6ZFnCveiVqHL7sq56Ctb6bdQ-fR5tYOTbK0E1o,154
|
|
6
|
+
extended_stats/cleaning/core.py,sha256=z6HIDTcbKxJqPDN2yFr1x6g1lzS_wUrqEX4bLpm2fI8,1024
|
|
7
|
+
extended_stats/diffing/__init__.py,sha256=tVUYgjLB2KEJ_QbfQ_o2YSu2CM4_6m9DTrZ6tjQQGeY,170
|
|
8
|
+
extended_stats/diffing/core.py,sha256=5bxql5Mj5MSdGAuujmT6R61GWhFM8jT5sNF6a2tmQd4,6263
|
|
9
|
+
extended_stats/diffing/narrative.py,sha256=exahyuwUBDZR17WaV4KVBUV0GpdjUKlbCZMjtRH90NU,1512
|
|
10
|
+
extended_stats/profiling/__init__.py,sha256=NN-cWBzt9OpzgeSxp9EMQZtZYHvOZRDphaqvYKsOikc,127
|
|
11
|
+
extended_stats/profiling/core.py,sha256=3fQrEwl5u7G3jv2vv97mc46SkjmmKPDIkGYkMxI58zQ,4023
|
|
12
|
+
extended_stats-0.1.0.dist-info/METADATA,sha256=kxPEi2bqJdX9fLi9hAmNVJ7jGUFkF5tKHPAeOH6-98Q,4023
|
|
13
|
+
extended_stats-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
14
|
+
extended_stats-0.1.0.dist-info/licenses/LICENSE,sha256=-SqQD97Q9UKTz8t92Lhlwj-iQB7aWSzGkseXb2giKAc,1097
|
|
15
|
+
extended_stats-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026, Djem Andreif Reyes
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|