lessPython 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lessPy/ANOVA.py +680 -0
- lessPy/Chart.py +1055 -0
- lessPy/Correlation.py +236 -0
- lessPy/Flows.py +116 -0
- lessPy/Logit.py +615 -0
- lessPy/Prop_test.py +267 -0
- lessPy/Regression.py +1491 -0
- lessPy/VariableLabels.py +119 -0
- lessPy/X.py +426 -0
- lessPy/XY.py +2007 -0
- lessPy/__init__.py +60 -0
- lessPy/anova_rmd.py +227 -0
- lessPy/bc_plotly.py +575 -0
- lessPy/bubble_plotly.py +470 -0
- lessPy/corCFA.py +316 -0
- lessPy/corEFA.py +220 -0
- lessPy/corPrint.py +45 -0
- lessPy/corProp.py +73 -0
- lessPy/corRead.py +48 -0
- lessPy/corReflect.py +72 -0
- lessPy/corReorder.py +161 -0
- lessPy/corScree.py +87 -0
- lessPy/data/Anova_1way.csv +25 -0
- lessPy/data/Anova_2way.csv +49 -0
- lessPy/data/Anova_rb.csv +8 -0
- lessPy/data/Anova_rbf.csv +49 -0
- lessPy/data/Anova_sp.csv +57 -0
- lessPy/data/BodyMeas.csv +341 -0
- lessPy/data/Cars93.csv +94 -0
- lessPy/data/Employee.csv +38 -0
- lessPy/data/Employee_lbl.csv +9 -0
- lessPy/data/FreqTable99.csv +5 -0
- lessPy/data/Jackets.csv +1026 -0
- lessPy/data/Learn.csv +35 -0
- lessPy/data/Mach4.csv +352 -0
- lessPy/data/Mach4_lbl.csv +21 -0
- lessPy/data/Reading.csv +101 -0
- lessPy/data/StockPrice.csv +1489 -0
- lessPy/data/WeightLoss.csv +11 -0
- lessPy/datasets.py +46 -0
- lessPy/date_infer.py +112 -0
- lessPy/details.py +314 -0
- lessPy/dn_plotly.py +495 -0
- lessPy/dot_plotly.py +385 -0
- lessPy/freq_poly_plotly.py +324 -0
- lessPy/getColors.py +399 -0
- lessPy/hier_plotly.py +352 -0
- lessPy/hs_plotly.py +395 -0
- lessPy/logit_rmd.py +410 -0
- lessPy/order_by.py +94 -0
- lessPy/pie_plotly.py +292 -0
- lessPy/pivot.py +158 -0
- lessPy/plotly_utils.py +787 -0
- lessPy/plt_add.py +129 -0
- lessPy/plt_contour.py +192 -0
- lessPy/plt_contour_facet.py +194 -0
- lessPy/plt_forecast.py +677 -0
- lessPy/plt_mat_plotly.py +201 -0
- lessPy/plt_plotly.py +216 -0
- lessPy/plt_smooth.py +170 -0
- lessPy/plt_time.py +143 -0
- lessPy/prob_norm.py +111 -0
- lessPy/prob_tcut.py +131 -0
- lessPy/prob_znorm.py +110 -0
- lessPy/radar_plotly.py +201 -0
- lessPy/reg_rmd.py +754 -0
- lessPy/rename.py +33 -0
- lessPy/reshape.py +95 -0
- lessPy/showColors.py +130 -0
- lessPy/simCImean.py +165 -0
- lessPy/simCLT.py +265 -0
- lessPy/simFlips.py +104 -0
- lessPy/simMeans.py +146 -0
- lessPy/stats_out.py +189 -0
- lessPy/ttest.py +641 -0
- lessPy/utils.py +235 -0
- lessPy/vbs_plotly.py +545 -0
- lesspython-0.1.0.dist-info/METADATA +93 -0
- lesspython-0.1.0.dist-info/RECORD +82 -0
- lesspython-0.1.0.dist-info/WHEEL +5 -0
- lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
- lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/datasets.py
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# datasets.py — the bundled example datasets, the Python analog
|
|
2
|
+
# of lessR's data/*.rda files loaded via Read(name,
|
|
3
|
+
# format="lessR"). The .rda frames were exported to CSV (no R
|
|
4
|
+
# factors exist in them, so CSV round-trips faithfully); read_data
|
|
5
|
+
# loads one into a pandas DataFrame.
|
|
6
|
+
|
|
7
|
+
from importlib import resources
|
|
8
|
+
|
|
9
|
+
import pandas as pd
|
|
10
|
+
|
|
11
|
+
# datasets whose first CSV column is a meaningful row label (the
|
|
12
|
+
# R row names) rather than data: used as the DataFrame index
|
|
13
|
+
_INDEXED = {"Employee", "Cars93", "WeightLoss", "Employee_lbl",
|
|
14
|
+
"Mach4_lbl"}
|
|
15
|
+
# datasets with a date column to parse
|
|
16
|
+
_DATE_COL = {"StockPrice": "Month"}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _data_dir():
|
|
20
|
+
return resources.files(__package__).joinpath("data")
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def datasets():
|
|
24
|
+
"""Sorted names of the bundled example datasets, each usable
|
|
25
|
+
with read_data(). R analog: the lessR data/*.rda files."""
|
|
26
|
+
return sorted(p.name[:-4] for p in _data_dir().iterdir()
|
|
27
|
+
if p.name.endswith(".csv"))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def read_data(name):
|
|
31
|
+
"""Load a bundled example dataset as a pandas DataFrame, the
|
|
32
|
+
analog of lessR's Read("<name>", format="lessR"). The
|
|
33
|
+
_lbl datasets (Employee_lbl, Mach4_lbl) are the variable-label
|
|
34
|
+
tables. See datasets() for the available names."""
|
|
35
|
+
res = _data_dir().joinpath(f"{name}.csv")
|
|
36
|
+
if not res.is_file():
|
|
37
|
+
raise ValueError(
|
|
38
|
+
f"no bundled dataset '{name}'. Available: "
|
|
39
|
+
f"{', '.join(datasets())}")
|
|
40
|
+
idx = 0 if name in _INDEXED else None
|
|
41
|
+
parse = [_DATE_COL[name]] if name in _DATE_COL else False
|
|
42
|
+
with resources.as_file(res) as path:
|
|
43
|
+
df = pd.read_csv(path, index_col=idx, parse_dates=parse)
|
|
44
|
+
if idx == 0:
|
|
45
|
+
df.index.name = None
|
|
46
|
+
return df
|
lessPy/date_infer.py
ADDED
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# date_infer.py — analog of date.infer.R / .charToDate.
|
|
2
|
+
#
|
|
3
|
+
# date_infer(): infer the format of a column of date strings and
|
|
4
|
+
# convert to pandas datetimes. Handles the numeric delimited
|
|
5
|
+
# forms (Y/m/d, d/m/Y, m/d/Y with "/", "-", or "."), the
|
|
6
|
+
# month-name form "2024Jan", and the quarter form "2024 Q3" —
|
|
7
|
+
# the last two beyond pandas' own inference. The month/day/year
|
|
8
|
+
# order of a numeric date is inferred from the component values,
|
|
9
|
+
# as R's .charToDate.
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
|
|
13
|
+
import numpy as np
|
|
14
|
+
import pandas as pd
|
|
15
|
+
|
|
16
|
+
_MONTHS = ("Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def date_infer(x, quiet=True):
|
|
20
|
+
"""Infer the date format of the string column x (a list or
|
|
21
|
+
Series) and return a datetime Series. Recognizes numeric
|
|
22
|
+
dates delimited by / - or . (order inferred), "2024Jan"
|
|
23
|
+
month-name dates, and "2024 Q3" quarter dates. Returns the
|
|
24
|
+
input unchanged if no date format is recognized.
|
|
25
|
+
R analog: date.infer()"""
|
|
26
|
+
s = pd.Series(list(x) if not isinstance(x, pd.Series) else x)
|
|
27
|
+
vals = s.dropna().astype(str)
|
|
28
|
+
if vals.empty:
|
|
29
|
+
return s
|
|
30
|
+
first = vals.iloc[0]
|
|
31
|
+
n_ch = len(first)
|
|
32
|
+
if not 6 <= n_ch <= 10:
|
|
33
|
+
return s
|
|
34
|
+
|
|
35
|
+
if re.search(_MONTHS, first): # "2024Jan"
|
|
36
|
+
t = s.astype(str).str.replace(" ", "", regex=False)
|
|
37
|
+
return pd.to_datetime(t.str[:4] + "-" + t.str[4:7]
|
|
38
|
+
+ "-01", format="%Y-%b-%d")
|
|
39
|
+
if re.search(r"Q[1-4]", first): # "2024 Q3"
|
|
40
|
+
t = s.astype(str).str.replace(r"\s+", "", regex=True)
|
|
41
|
+
yr = t.str.split("Q").str[0].astype(int)
|
|
42
|
+
q = t.str.split("Q").str[1].astype(int)
|
|
43
|
+
month = 1 + (q - 1) * 3
|
|
44
|
+
return pd.to_datetime(
|
|
45
|
+
{"year": yr, "month": month, "day": 1})
|
|
46
|
+
|
|
47
|
+
# numeric date: the delimiter is the punctuation that appears
|
|
48
|
+
# exactly twice in the first value
|
|
49
|
+
punct = next((p for p in ("/", "-", ".")
|
|
50
|
+
if first.count(p) == 2), None)
|
|
51
|
+
if punct is None:
|
|
52
|
+
return s
|
|
53
|
+
return _char_to_date(s.astype(str), punct, quiet)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _char_to_date(char, punct, quiet):
|
|
57
|
+
parts = char.str.split(re.escape(punct), expand=True)
|
|
58
|
+
c1 = pd.to_numeric(parts[0], errors="coerce")
|
|
59
|
+
c2 = pd.to_numeric(parts[1], errors="coerce")
|
|
60
|
+
c3 = pd.to_numeric(parts[2], errors="coerce")
|
|
61
|
+
mx1, mx2, mx3 = c1.max(), c2.max(), c3.max()
|
|
62
|
+
unq1, unq2 = c1.nunique(), c2.nunique()
|
|
63
|
+
if np.isnan(mx1) or np.isnan(mx2):
|
|
64
|
+
raise ValueError("at least one date has non-numeric "
|
|
65
|
+
"characters where a number is expected")
|
|
66
|
+
|
|
67
|
+
fmt = None
|
|
68
|
+
if mx1 > 31:
|
|
69
|
+
fmt = f"%Y{punct}%m{punct}%d"
|
|
70
|
+
elif mx1 > 12:
|
|
71
|
+
fmt = f"%d{punct}%m{punct}%Y"
|
|
72
|
+
elif mx2 > 12:
|
|
73
|
+
fmt = f"%m{punct}%d{punct}%Y"
|
|
74
|
+
elif unq2 <= 12 and unq2 in (2, 4, 12):
|
|
75
|
+
fmt = f"%d{punct}%m{punct}%Y"
|
|
76
|
+
elif unq1 <= 12 and unq1 in (2, 4, 12):
|
|
77
|
+
fmt = f"%m{punct}%d{punct}%Y"
|
|
78
|
+
if fmt is None:
|
|
79
|
+
raise ValueError(
|
|
80
|
+
"the date format could not be inferred; supply "
|
|
81
|
+
"dates as one of Y-m-d, d-m-Y, or m-d-Y")
|
|
82
|
+
|
|
83
|
+
# a 2-digit year in the 3rd position takes %y
|
|
84
|
+
if fmt[1] != "Y" and mx3 <= 99:
|
|
85
|
+
fmt = fmt.replace("Y", "y")
|
|
86
|
+
if not quiet:
|
|
87
|
+
print(f"Best guess for the date format: {fmt}")
|
|
88
|
+
return pd.to_datetime(char, format=fmt)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def format_date_labels(dates, ts_unit):
|
|
92
|
+
"""Format a column of dates as axis labels for a time unit:
|
|
93
|
+
years "2024", quarters "2024 Q3", months "Aug 2024", and
|
|
94
|
+
weeks / days "18 Aug 2024". Returns a Series of strings.
|
|
95
|
+
R analog: .format_date_labels()"""
|
|
96
|
+
if ts_unit not in ("years", "quarters", "months", "weeks",
|
|
97
|
+
"days", "days7", "unknown"):
|
|
98
|
+
raise ValueError(
|
|
99
|
+
'ts_unit: "years", "quarters", "months", "weeks", '
|
|
100
|
+
'"days", "days7"')
|
|
101
|
+
d = pd.to_datetime(pd.Series(list(dates)
|
|
102
|
+
if not isinstance(dates,
|
|
103
|
+
pd.Series)
|
|
104
|
+
else dates))
|
|
105
|
+
if ts_unit == "years":
|
|
106
|
+
return d.dt.strftime("%Y")
|
|
107
|
+
if ts_unit == "quarters":
|
|
108
|
+
return (d.dt.strftime("%Y") + " Q"
|
|
109
|
+
+ d.dt.quarter.astype(str))
|
|
110
|
+
if ts_unit == "months":
|
|
111
|
+
return d.dt.strftime("%b %Y")
|
|
112
|
+
return d.dt.strftime("%d %b %Y") # weeks, days, days7
|
lessPy/details.py
ADDED
|
@@ -0,0 +1,314 @@
|
|
|
1
|
+
# details.py — analog of details.R.
|
|
2
|
+
#
|
|
3
|
+
# details(): a diagnostic report on a data frame. Prints the
|
|
4
|
+
# dimensions and row names, a legend of the data types present, a
|
|
5
|
+
# per-variable table (type, non-missing / missing / unique counts,
|
|
6
|
+
# and the first and last data values), a suggestion when a text
|
|
7
|
+
# column is a unique per-row ID, a note on numeric variables with
|
|
8
|
+
# few unique values, and a missing-data analysis. Returns a
|
|
9
|
+
# DetailsResults with the per-variable summary as a DataFrame.
|
|
10
|
+
#
|
|
11
|
+
# R's storage types map onto pandas dtypes: an unordered/ordered
|
|
12
|
+
# pandas Categorical is a factor/ordfactor, object/string is
|
|
13
|
+
# "character", and integer/float/datetime/bool map to integer/
|
|
14
|
+
# double/Date/logical. R reads variable labels and units from the
|
|
15
|
+
# data-frame attributes attr(data, "variable.labels"/"units"); the
|
|
16
|
+
# pandas analog is data.attrs["variable_labels"/"variable_units"]
|
|
17
|
+
# (each a name -> text mapping).
|
|
18
|
+
|
|
19
|
+
import numpy as np
|
|
20
|
+
import pandas as pd
|
|
21
|
+
from pandas.api import types as pdt
|
|
22
|
+
|
|
23
|
+
from .utils import get_option
|
|
24
|
+
|
|
25
|
+
_TYPE_ORDER = ("factor", "ordfactor", "character", "integer",
|
|
26
|
+
"Date", "double", "logical")
|
|
27
|
+
_TYPE_DESC = {
|
|
28
|
+
"factor": "Non-numeric categories, read as unordered "
|
|
29
|
+
"categories",
|
|
30
|
+
"ordfactor": "Ordered, non-numeric categories",
|
|
31
|
+
"character": "Non-numeric data values",
|
|
32
|
+
"integer": "Numeric data values, integers only",
|
|
33
|
+
"Date": "Date with year, month and day",
|
|
34
|
+
"double": "Numeric data values with decimal digits",
|
|
35
|
+
"logical": "Boolean True/False values",
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class DetailsResults:
|
|
40
|
+
"""Result of details(): the dimensions, the total/proportion of
|
|
41
|
+
missing values, the per-variable summary DataFrame, a detected
|
|
42
|
+
ID column (or None), and the rows with missing data (or None)."""
|
|
43
|
+
|
|
44
|
+
def __init__(self, **kw):
|
|
45
|
+
self.__dict__.update(kw)
|
|
46
|
+
|
|
47
|
+
def __repr__(self):
|
|
48
|
+
return (f"<lessPy details: {self.n_var} variables, "
|
|
49
|
+
f"{self.n_obs} rows>")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _dash(n):
|
|
53
|
+
print("-" * n)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _type_label(s):
|
|
57
|
+
dt = s.dtype
|
|
58
|
+
if isinstance(dt, pd.CategoricalDtype):
|
|
59
|
+
return "ordfactor" if dt.ordered else "factor"
|
|
60
|
+
if pdt.is_bool_dtype(dt):
|
|
61
|
+
return "logical"
|
|
62
|
+
if pdt.is_datetime64_any_dtype(dt):
|
|
63
|
+
return "Date"
|
|
64
|
+
if pdt.is_integer_dtype(dt):
|
|
65
|
+
return "integer"
|
|
66
|
+
if pdt.is_float_dtype(dt):
|
|
67
|
+
# pandas stores an integer column that has any missing value
|
|
68
|
+
# as float64; if every non-missing value is whole, report it
|
|
69
|
+
# as integer to match R's storage-based view (else double).
|
|
70
|
+
v = s.dropna()
|
|
71
|
+
v = v[pd.Series(np.isfinite(v), index=v.index)]
|
|
72
|
+
if len(v) and (v % 1 == 0).all():
|
|
73
|
+
return "integer"
|
|
74
|
+
return "double"
|
|
75
|
+
return "character"
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _cell(x):
|
|
79
|
+
if pd.isna(x):
|
|
80
|
+
return "NA"
|
|
81
|
+
if isinstance(x, float) and np.isfinite(x) and x == int(x):
|
|
82
|
+
return str(int(x))
|
|
83
|
+
return str(x)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _first_last(col, n_obs):
|
|
87
|
+
"""First up to three and last up to three values, joined with
|
|
88
|
+
' ... ', trimmed toward the ends as the line grows (the width
|
|
89
|
+
rule of details.R)."""
|
|
90
|
+
v = col.tolist()
|
|
91
|
+
n1 = _cell(v[0])
|
|
92
|
+
n2 = _cell(v[1]) if n_obs >= 2 else ""
|
|
93
|
+
n3 = _cell(v[2]) if n_obs >= 3 else ""
|
|
94
|
+
e3 = _cell(v[n_obs - 3]) if n_obs >= 5 else ""
|
|
95
|
+
e2 = _cell(v[n_obs - 2]) if n_obs >= 4 else ""
|
|
96
|
+
e1 = _cell(v[n_obs - 1]) if n_obs >= 2 else ""
|
|
97
|
+
tot = len(" ".join(x for x in (n1, n2, n3, e3, e2, e1) if x))
|
|
98
|
+
if tot > 34:
|
|
99
|
+
n3 = e3 = ""
|
|
100
|
+
if tot > 58:
|
|
101
|
+
n2 = e2 = ""
|
|
102
|
+
fp = " ".join(x for x in (n1, n2, n3) if x)
|
|
103
|
+
lp = " ".join(x for x in (e3, e2, e1) if x)
|
|
104
|
+
return f"{fp} ... {lp}" if lp else fp
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def details(data=None, n_mcut=1, max_lines=30, miss_show=30,
|
|
108
|
+
miss_zero=False, miss_matrix=False, var_labels=False,
|
|
109
|
+
brief=None, n_cat=None):
|
|
110
|
+
"""Diagnostic report on a data frame: dimensions, data-type
|
|
111
|
+
legend, a per-variable table of type / non-missing / missing /
|
|
112
|
+
unique counts with first and last values, ID-column and
|
|
113
|
+
numeric-category notes, and a missing-data analysis. Returns a
|
|
114
|
+
DetailsResults with the per-variable summary. R analog:
|
|
115
|
+
details()"""
|
|
116
|
+
if data is None:
|
|
117
|
+
raise ValueError("specify a data frame: data=")
|
|
118
|
+
if brief is None:
|
|
119
|
+
brief = get_option("brief", False)
|
|
120
|
+
if n_cat is None:
|
|
121
|
+
n_cat = get_option("n_cat", 0)
|
|
122
|
+
|
|
123
|
+
n_var = data.shape[1]
|
|
124
|
+
n_obs = data.shape[0]
|
|
125
|
+
if n_obs == 0 or n_var == 0:
|
|
126
|
+
raise ValueError("data has no rows or no columns")
|
|
127
|
+
max_lines = min(max_lines, n_obs)
|
|
128
|
+
n_miss_tot = int(data.isna().sum().sum())
|
|
129
|
+
|
|
130
|
+
names = list(data.columns)
|
|
131
|
+
types = [_type_label(data[c]) for c in names]
|
|
132
|
+
n_val = [int(data[c].notna().sum()) for c in names]
|
|
133
|
+
n_miss = [int(data[c].isna().sum()) for c in names]
|
|
134
|
+
n_uniq = [int(data[c].dropna().nunique()) for c in names]
|
|
135
|
+
|
|
136
|
+
_print_header(data, n_var, n_obs, brief)
|
|
137
|
+
_print_types(types)
|
|
138
|
+
first_last = [_first_last(data[c], n_obs) for c in names]
|
|
139
|
+
maybe_id, id_col = _print_table(names, types, n_val, n_miss,
|
|
140
|
+
n_uniq, first_last, n_var)
|
|
141
|
+
|
|
142
|
+
if maybe_id is not None and not var_labels:
|
|
143
|
+
_print_id_note(maybe_id, id_col)
|
|
144
|
+
|
|
145
|
+
if not brief:
|
|
146
|
+
_print_num_cat(names, types, n_uniq, n_cat)
|
|
147
|
+
|
|
148
|
+
missing_rows = None
|
|
149
|
+
if not brief:
|
|
150
|
+
if n_miss_tot > 0:
|
|
151
|
+
missing_rows = _print_missing(
|
|
152
|
+
data, n_var, n_obs, n_miss_tot, n_mcut, miss_show,
|
|
153
|
+
miss_zero, miss_matrix)
|
|
154
|
+
else:
|
|
155
|
+
print("No missing data\n")
|
|
156
|
+
|
|
157
|
+
_print_meta(data, max_lines)
|
|
158
|
+
|
|
159
|
+
summary = pd.DataFrame(
|
|
160
|
+
{"Type": types, "Values": n_val, "Missing": n_miss,
|
|
161
|
+
"Unique": n_uniq}, index=pd.Index(names, name="Variable"))
|
|
162
|
+
prop_miss = round(n_miss_tot / (n_var * n_obs), 3)
|
|
163
|
+
return DetailsResults(
|
|
164
|
+
n_var=n_var, n_obs=n_obs, n_miss=n_miss_tot,
|
|
165
|
+
prop_miss=prop_miss, summary=summary, maybe_id=maybe_id,
|
|
166
|
+
missing_rows=missing_rows)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
# --- report sections -----------------------------------------------
|
|
170
|
+
|
|
171
|
+
def _print_header(data, n_var, n_obs, brief):
|
|
172
|
+
if brief:
|
|
173
|
+
return
|
|
174
|
+
print()
|
|
175
|
+
_dash(58)
|
|
176
|
+
print(f"Dimensions: {n_var} variables over {n_obs} rows of "
|
|
177
|
+
"data")
|
|
178
|
+
print()
|
|
179
|
+
idx = data.index
|
|
180
|
+
if n_obs >= 2:
|
|
181
|
+
print(f"First two row names: {idx[0]} {idx[1]}")
|
|
182
|
+
print(f"Last two row names: {idx[-2]} {idx[-1]}")
|
|
183
|
+
else:
|
|
184
|
+
print(f"Only one row, row name: {idx[0]}")
|
|
185
|
+
_dash(58)
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _print_types(types):
|
|
189
|
+
present = [t for t in _TYPE_ORDER if t in types]
|
|
190
|
+
print("Data Types")
|
|
191
|
+
_dash(60)
|
|
192
|
+
for t in present:
|
|
193
|
+
print(f"{t}: {_TYPE_DESC[t]}")
|
|
194
|
+
_dash(60)
|
|
195
|
+
print()
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _print_table(names, types, n_val, n_miss, n_uniq, first_last,
|
|
199
|
+
n_var):
|
|
200
|
+
w_num = max(2, len(str(n_var)))
|
|
201
|
+
w_nam = max(len("Variable"), *(len(x) for x in names))
|
|
202
|
+
w_typ = max(len("Type"), *(len(t) for t in types))
|
|
203
|
+
w_val = max(len("Values"), *(len(str(x)) for x in n_val))
|
|
204
|
+
w_mis = max(len("Missing"), *(len(str(x)) for x in n_miss))
|
|
205
|
+
w_uni = max(len("Unique"), *(len(str(x)) for x in n_uniq))
|
|
206
|
+
|
|
207
|
+
hdr = (f"{'':>{w_num}} {'Variable':<{w_nam}} "
|
|
208
|
+
f"{'Type':<{w_typ}} {'Values':>{w_val}} "
|
|
209
|
+
f"{'Missing':>{w_mis}} {'Unique':>{w_uni}} "
|
|
210
|
+
"First and last values")
|
|
211
|
+
_dash(len(hdr))
|
|
212
|
+
print(hdr)
|
|
213
|
+
_dash(len(hdr))
|
|
214
|
+
|
|
215
|
+
maybe_id, id_col = None, 0
|
|
216
|
+
for i in range(n_var):
|
|
217
|
+
print(f"{i + 1:>{w_num}} {names[i]:<{w_nam}} "
|
|
218
|
+
f"{types[i]:<{w_typ}} {n_val[i]:>{w_val}} "
|
|
219
|
+
f"{n_miss[i]:>{w_mis}} {n_uniq[i]:>{w_uni}} "
|
|
220
|
+
f"{first_last[i]}")
|
|
221
|
+
if (n_uniq[i] == n_val[i]
|
|
222
|
+
and types[i] in ("factor", "ordfactor",
|
|
223
|
+
"character")):
|
|
224
|
+
maybe_id, id_col = names[i], i + 1
|
|
225
|
+
_dash(len(hdr))
|
|
226
|
+
return maybe_id, id_col
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _print_id_note(maybe_id, id_col):
|
|
230
|
+
print("\n")
|
|
231
|
+
print(f"For the column {maybe_id}, each row of data is unique. "
|
|
232
|
+
"Are these values")
|
|
233
|
+
print("a unique ID for each row? To define as a row name, "
|
|
234
|
+
"re-read the data")
|
|
235
|
+
print(f"file with index_col={id_col} in your read_data() / "
|
|
236
|
+
"pandas call.")
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _print_num_cat(names, types, n_uniq, n_cat):
|
|
240
|
+
if n_cat <= 0:
|
|
241
|
+
return
|
|
242
|
+
flagged = [names[j] for j in range(len(names))
|
|
243
|
+
if types[j] == "double" and n_uniq[j] <= n_cat]
|
|
244
|
+
if not flagged:
|
|
245
|
+
return
|
|
246
|
+
print("\n")
|
|
247
|
+
print(f"Each of these variables is numeric but has fewer than "
|
|
248
|
+
f"{n_cat}")
|
|
249
|
+
print("unique values. Perhaps they are categorical. Consider "
|
|
250
|
+
"converting")
|
|
251
|
+
print("each to a category, or set n_cat, e.g. "
|
|
252
|
+
"set_option('n_cat', 4).")
|
|
253
|
+
_dash(63)
|
|
254
|
+
for nm in flagged:
|
|
255
|
+
print(nm)
|
|
256
|
+
_dash(63)
|
|
257
|
+
print()
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def _print_missing(data, n_var, n_obs, n_miss_tot, n_mcut,
|
|
261
|
+
miss_show, miss_zero, miss_matrix):
|
|
262
|
+
print("Missing Data Analysis")
|
|
263
|
+
_dash(60)
|
|
264
|
+
mcut = 0 if miss_zero else n_mcut
|
|
265
|
+
per_row = data.isna().sum(axis=1)
|
|
266
|
+
bad = []
|
|
267
|
+
for i in range(n_obs):
|
|
268
|
+
if int(per_row.iloc[i]) >= mcut and len(bad) < miss_show:
|
|
269
|
+
bad.append(i)
|
|
270
|
+
if len(bad) == miss_show:
|
|
271
|
+
break
|
|
272
|
+
n_lines = len(bad)
|
|
273
|
+
|
|
274
|
+
print(f"Number of cells in the data table: {n_var * n_obs}")
|
|
275
|
+
print(f"Number of missing data values: {n_miss_tot}")
|
|
276
|
+
print("Proportion of missing data values: "
|
|
277
|
+
f"{round(n_miss_tot / (n_var * n_obs), 3)}")
|
|
278
|
+
label = ("Number of rows of data listed: " if miss_zero
|
|
279
|
+
else "Number of rows of data with missing values: ")
|
|
280
|
+
print(f"{label}{n_lines}")
|
|
281
|
+
_dash(60)
|
|
282
|
+
print()
|
|
283
|
+
rows = data.iloc[bad]
|
|
284
|
+
print(rows.to_string())
|
|
285
|
+
_dash(60)
|
|
286
|
+
print()
|
|
287
|
+
|
|
288
|
+
if miss_matrix:
|
|
289
|
+
print("\nTable of Missing Values, 1 means missing")
|
|
290
|
+
print(data.isna().astype(int).to_string())
|
|
291
|
+
return rows
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _print_meta(data, max_lines):
|
|
295
|
+
labels = data.attrs.get("variable_labels")
|
|
296
|
+
units = data.attrs.get("variable_units")
|
|
297
|
+
if labels:
|
|
298
|
+
_print_named(labels, "Variable Labels", max_lines)
|
|
299
|
+
if units:
|
|
300
|
+
_print_named(units, "Variable Units", max_lines)
|
|
301
|
+
print()
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _print_named(mapping, header, max_lines):
|
|
305
|
+
items = list(mapping.items())[:max_lines]
|
|
306
|
+
w = max(len("Variable Names"),
|
|
307
|
+
*(len(str(k)) for k, _ in items)) + 1
|
|
308
|
+
print(f"\n{'Variable Names':<{w}} {header}")
|
|
309
|
+
width = max(len(f"{'Variable Names':<{w}} {header}"),
|
|
310
|
+
*(len(f"{k:<{w}} {v}") for k, v in items))
|
|
311
|
+
_dash(min(width + 1, 80))
|
|
312
|
+
for k, v in items:
|
|
313
|
+
print(f"{str(k):<{w}} {'' if pd.isna(v) else v}")
|
|
314
|
+
_dash(min(width + 1, 80))
|