lessPython 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. lessPy/ANOVA.py +680 -0
  2. lessPy/Chart.py +1055 -0
  3. lessPy/Correlation.py +236 -0
  4. lessPy/Flows.py +116 -0
  5. lessPy/Logit.py +615 -0
  6. lessPy/Prop_test.py +267 -0
  7. lessPy/Regression.py +1491 -0
  8. lessPy/VariableLabels.py +119 -0
  9. lessPy/X.py +426 -0
  10. lessPy/XY.py +2007 -0
  11. lessPy/__init__.py +60 -0
  12. lessPy/anova_rmd.py +227 -0
  13. lessPy/bc_plotly.py +575 -0
  14. lessPy/bubble_plotly.py +470 -0
  15. lessPy/corCFA.py +316 -0
  16. lessPy/corEFA.py +220 -0
  17. lessPy/corPrint.py +45 -0
  18. lessPy/corProp.py +73 -0
  19. lessPy/corRead.py +48 -0
  20. lessPy/corReflect.py +72 -0
  21. lessPy/corReorder.py +161 -0
  22. lessPy/corScree.py +87 -0
  23. lessPy/data/Anova_1way.csv +25 -0
  24. lessPy/data/Anova_2way.csv +49 -0
  25. lessPy/data/Anova_rb.csv +8 -0
  26. lessPy/data/Anova_rbf.csv +49 -0
  27. lessPy/data/Anova_sp.csv +57 -0
  28. lessPy/data/BodyMeas.csv +341 -0
  29. lessPy/data/Cars93.csv +94 -0
  30. lessPy/data/Employee.csv +38 -0
  31. lessPy/data/Employee_lbl.csv +9 -0
  32. lessPy/data/FreqTable99.csv +5 -0
  33. lessPy/data/Jackets.csv +1026 -0
  34. lessPy/data/Learn.csv +35 -0
  35. lessPy/data/Mach4.csv +352 -0
  36. lessPy/data/Mach4_lbl.csv +21 -0
  37. lessPy/data/Reading.csv +101 -0
  38. lessPy/data/StockPrice.csv +1489 -0
  39. lessPy/data/WeightLoss.csv +11 -0
  40. lessPy/datasets.py +46 -0
  41. lessPy/date_infer.py +112 -0
  42. lessPy/details.py +314 -0
  43. lessPy/dn_plotly.py +495 -0
  44. lessPy/dot_plotly.py +385 -0
  45. lessPy/freq_poly_plotly.py +324 -0
  46. lessPy/getColors.py +399 -0
  47. lessPy/hier_plotly.py +352 -0
  48. lessPy/hs_plotly.py +395 -0
  49. lessPy/logit_rmd.py +410 -0
  50. lessPy/order_by.py +94 -0
  51. lessPy/pie_plotly.py +292 -0
  52. lessPy/pivot.py +158 -0
  53. lessPy/plotly_utils.py +787 -0
  54. lessPy/plt_add.py +129 -0
  55. lessPy/plt_contour.py +192 -0
  56. lessPy/plt_contour_facet.py +194 -0
  57. lessPy/plt_forecast.py +677 -0
  58. lessPy/plt_mat_plotly.py +201 -0
  59. lessPy/plt_plotly.py +216 -0
  60. lessPy/plt_smooth.py +170 -0
  61. lessPy/plt_time.py +143 -0
  62. lessPy/prob_norm.py +111 -0
  63. lessPy/prob_tcut.py +131 -0
  64. lessPy/prob_znorm.py +110 -0
  65. lessPy/radar_plotly.py +201 -0
  66. lessPy/reg_rmd.py +754 -0
  67. lessPy/rename.py +33 -0
  68. lessPy/reshape.py +95 -0
  69. lessPy/showColors.py +130 -0
  70. lessPy/simCImean.py +165 -0
  71. lessPy/simCLT.py +265 -0
  72. lessPy/simFlips.py +104 -0
  73. lessPy/simMeans.py +146 -0
  74. lessPy/stats_out.py +189 -0
  75. lessPy/ttest.py +641 -0
  76. lessPy/utils.py +235 -0
  77. lessPy/vbs_plotly.py +545 -0
  78. lesspython-0.1.0.dist-info/METADATA +93 -0
  79. lesspython-0.1.0.dist-info/RECORD +82 -0
  80. lesspython-0.1.0.dist-info/WHEEL +5 -0
  81. lesspython-0.1.0.dist-info/licenses/LICENSE +338 -0
  82. lesspython-0.1.0.dist-info/top_level.txt +1 -0
lessPy/Correlation.py ADDED
@@ -0,0 +1,236 @@
1
+ # Correlation.py — analog of Correlation.R (cr.main, cr.data.frame).
2
+ #
3
+ # Correlation(): with two variables, the correlation of a pair
4
+ # with its significance test (Pearson t-test and Fisher-z 95% CI,
5
+ # or Spearman/Kendall); with a data frame or a set of variables,
6
+ # the correlation matrix. miss= controls pairwise/listwise/
7
+ # everything deletion; show="missing" reports the pairwise
8
+ # complete-case counts. Prints as R; the two-variable form
9
+ # returns a CorrelationResults, the matrix form the correlation
10
+ # DataFrame (its .attrs["plots"] holds the heat map).
11
+
12
+ import numpy as np
13
+ import pandas as pd
14
+
15
+ from .utils import fmt, get_column, get_option
16
+
17
+
18
+ class CorrelationResults:
19
+ """Results of a two-variable Correlation(): the coefficient,
20
+ t-test, and (Pearson) confidence interval."""
21
+
22
+ def __init__(self, **kw):
23
+ self.__dict__.update(kw)
24
+
25
+ def __repr__(self):
26
+ return (f"<lessPy Correlation: {self.method}, "
27
+ f"r={self.r}, p={self.pvalue}>")
28
+
29
+
30
+ def Correlation(x=None, y=None, data=None, miss="pairwise",
31
+ show="cor", method="pearson", brief=False,
32
+ digits_d=None, heat_map=True, main=None):
33
+ """Correlation of two variables (with a significance test) or
34
+ the correlation matrix of several. x and y are column names
35
+ (or arrays); with y omitted, x is a DataFrame or a list of
36
+ column names (or None for all numeric columns of data).
37
+ method: "pearson" (default), "spearman", "kendall".
38
+ R analog: Correlation()"""
39
+ if miss not in ("pairwise", "listwise", "everything"):
40
+ raise ValueError('miss: "pairwise", "listwise", '
41
+ '"everything"')
42
+ if method not in ("pearson", "spearman", "kendall"):
43
+ raise ValueError('method: "pearson", "spearman", '
44
+ '"kendall"')
45
+ if y is not None:
46
+ xv = _resolve1(x, data, "x")
47
+ yv = _resolve1(y, data, "y")
48
+ return _two_var(xv, yv, _name(x, "x"), _name(y, "y"),
49
+ method, brief)
50
+ return _matrix(_frame(x, data), miss, show, method, digits_d,
51
+ heat_map, main)
52
+
53
+
54
+ def _name(v, default):
55
+ return v if isinstance(v, str) else default
56
+
57
+
58
+ def _resolve1(v, data, arg):
59
+ if isinstance(v, str):
60
+ if data is None:
61
+ raise ValueError(f"{arg}='{v}' is a column name, so "
62
+ "data= is required")
63
+ return get_column(data, v, arg).to_numpy(dtype=float)
64
+ return np.asarray(v, dtype=float)
65
+
66
+
67
+ def _frame(x, data):
68
+ if isinstance(x, pd.DataFrame):
69
+ df = x
70
+ elif isinstance(x, (list, tuple)):
71
+ if data is None:
72
+ raise ValueError("data= is required with a list of "
73
+ "variable names")
74
+ df = data[list(x)]
75
+ elif x is None:
76
+ if data is None:
77
+ raise ValueError("supply a DataFrame, a list of "
78
+ "variables with data=, or data=")
79
+ df = data
80
+ else:
81
+ raise TypeError("for a correlation matrix, x is a "
82
+ "DataFrame or a list of variable names")
83
+ return df
84
+
85
+
86
+ # ------------------------------------------------------------
87
+ # two variables
88
+ # ------------------------------------------------------------
89
+
90
+ def _two_var(x, y, x_name, y_name, method, brief):
91
+ from scipy import stats as sps
92
+ keep = ~(np.isnan(x) | np.isnan(y))
93
+ n = int(keep.sum())
94
+ n_del = int((~keep).sum())
95
+ xk, yk = x[keep], y[keep]
96
+ lb = ub = np.nan
97
+
98
+ if method == "pearson":
99
+ r, p = sps.pearsonr(xk, yk)
100
+ df = n - 2
101
+ t = r * np.sqrt(df / (1 - r ** 2))
102
+ p = 2 * sps.t.sf(abs(t), df)
103
+ z = np.arctanh(r)
104
+ sig = 1 / np.sqrt(n - 3)
105
+ zc = sps.norm.ppf(0.975)
106
+ lb, ub = np.tanh(z - zc * sig), np.tanh(z + zc * sig)
107
+ method_txt = "Pearson's product-moment correlation"
108
+ sym, sym_pop = "r", "Correlation"
109
+ cov = float(np.cov(xk, yk)[0, 1])
110
+ elif method == "spearman":
111
+ res = sps.spearmanr(xk, yk)
112
+ r, p = float(res.statistic), float(res.pvalue)
113
+ t, df = res.statistic * np.sqrt(
114
+ (n - 2) / (1 - r ** 2)), n - 2
115
+ method_txt = "Spearman's rank correlation rho"
116
+ sym = sym_pop = "rho"
117
+ else:
118
+ res = sps.kendalltau(xk, yk)
119
+ r, p = float(res.statistic), float(res.pvalue)
120
+ t = res.statistic
121
+ df = None
122
+ method_txt = "Kendall's rank correlation tau"
123
+ sym = sym_pop = "tau"
124
+
125
+ L = []
126
+ if not brief:
127
+ L += [f"Correlation Analysis for Variables {x_name} "
128
+ f"and {y_name}", ""]
129
+ L += [f"\n>>> {method_txt}", "",
130
+ "Number of paired values with neither missing, "
131
+ f"n = {n}"]
132
+ if not brief:
133
+ L.append(f"Number of cases (rows of data) deleted: "
134
+ f"{n_del}")
135
+ L.append("")
136
+ if method == "pearson" and not brief:
137
+ L += [f"Sample Covariance: s = {fmt(cov, 3)}", ""]
138
+ if brief:
139
+ L.append(f"Sample Correlation of {x_name} and {y_name}: "
140
+ f"{sym} = {fmt(r, 3)}")
141
+ else:
142
+ L.append(f"Sample Correlation: {sym} = {fmt(r, 3)}")
143
+ L.append("")
144
+ dfs = "NA" if df is None else str(df)
145
+ L.append(f"Hypothesis Test of 0 {sym_pop}: t = {fmt(t, 3)}, "
146
+ f" df = {dfs}, p-value = {fmt(p, 3)}")
147
+ if method == "pearson":
148
+ L.append("95% Confidence Interval for Correlation: "
149
+ f"{fmt(lb, 3)} to {fmt(ub, 3)}")
150
+ print("\n".join(L))
151
+
152
+ return CorrelationResults(
153
+ method=method, r=round(float(r), 3),
154
+ tvalue=round(float(t), 3),
155
+ df=(None if df is None else int(df)),
156
+ pvalue=round(float(p), 3),
157
+ lb=(round(lb, 3) if np.isfinite(lb) else None),
158
+ ub=(round(ub, 3) if np.isfinite(ub) else None), n=n)
159
+
160
+
161
+ # ------------------------------------------------------------
162
+ # correlation matrix
163
+ # ------------------------------------------------------------
164
+
165
+ def _matrix(df, miss, show, method, digits_d, heat_map, main):
166
+ num = df.select_dtypes("number")
167
+ dropped = [c for c in df.columns if c not in num.columns]
168
+ if num.shape[1] < 2:
169
+ raise ValueError(
170
+ "a correlation matrix needs at least 2 numeric "
171
+ "variables")
172
+ d = digits_d if digits_d is not None else 2
173
+
174
+ if miss == "listwise":
175
+ crs = num.dropna().corr(method=method)
176
+ else:
177
+ crs = num.corr(method=method)
178
+ if miss == "everything":
179
+ for c in num.columns[num.isna().any()]:
180
+ crs.loc[c, :] = np.nan
181
+ crs.loc[:, c] = np.nan
182
+ crs = crs.round(d)
183
+
184
+ if dropped:
185
+ print("The following non-numeric variables are deleted "
186
+ "from the analysis")
187
+ for i, c in enumerate(dropped, 1):
188
+ print(f"{i}. {c}")
189
+ tot_miss = int(num.isna().sum().sum())
190
+ if tot_miss == 0:
191
+ print("\n>>> No missing data\n")
192
+ else:
193
+ print(f"\nMissing data deletion: {miss}")
194
+
195
+ if show == "missing":
196
+ n = pd.DataFrame(
197
+ np.zeros((num.shape[1], num.shape[1]), dtype=int),
198
+ index=num.columns, columns=num.columns)
199
+ for i in num.columns:
200
+ for j in num.columns:
201
+ n.loc[i, j] = int((~(num[i].isna()
202
+ | num[j].isna())).sum())
203
+ print("\nPairwise complete-case counts\n")
204
+ print(n.to_string())
205
+ crs.attrs["missing_n"] = n
206
+ return crs
207
+
208
+ print("\nCorrelation Matrix")
209
+ print(crs.round(2).to_string())
210
+ plots = {}
211
+ if heat_map:
212
+ plots["heatmap"] = _heatmap(crs, main)
213
+ crs.attrs["plots"] = plots
214
+ return crs
215
+
216
+
217
+ def _heatmap(df, main):
218
+ import plotly.graph_objects as go
219
+
220
+ from .plotly_utils import plotly_style, to_hex
221
+ labels = list(df.columns)
222
+ Z = df.to_numpy(dtype=float).copy()
223
+ np.fill_diagonal(Z, np.nan)
224
+ style = plotly_style()
225
+ fig = go.Figure(go.Heatmap(
226
+ z=Z, x=labels, y=labels, zmin=-1, zmax=1,
227
+ colorscale="RdBu", reversescale=True,
228
+ colorbar=dict(title="r")))
229
+ fig.update_yaxes(autorange="reversed")
230
+ fig.update_layout(
231
+ template=None, paper_bgcolor=to_hex(style["window_fill"]),
232
+ title=dict(text=main or "Correlations", x=0.5,
233
+ xanchor="center",
234
+ font=dict(size=round(
235
+ 16 * get_option("main_size", 1)))))
236
+ return fig
lessPy/Flows.py ADDED
@@ -0,0 +1,116 @@
1
+ # Flows.py — analog of Flows.R.
2
+ #
3
+ # Flows(): a Sankey flow diagram, stage1 -> stage2 -> optional
4
+ # stage3, with ribbon widths weighted by value. Each source's
5
+ # flow keeps its color through every stage. lessR's Flows is
6
+ # already plotly-based, so this is a direct go.Sankey port.
7
+ # Returns the plotly figure.
8
+
9
+ import numpy as np
10
+ import pandas as pd
11
+
12
+ from .plotly_utils import BASE_COLORS, make_trans, to_hex
13
+ from .utils import get_column, get_option
14
+
15
+
16
+ def Flows(value, stage1, stage2, stage3=None, data=None,
17
+ title=None, fill=None, link_alpha=0.55,
18
+ nodes_gray=False, neutral_gray="gray80",
19
+ border_gray="gray60", lift_y=0, labels_size=1.0,
20
+ digits_d=0):
21
+ """Sankey flow diagram of value across the stages stage1 ->
22
+ stage2 (-> stage3). value, stage1, stage2, stage3 are column
23
+ names in data. Each source's flow keeps its color. Returns the
24
+ plotly figure. R analog: Flows()"""
25
+ import plotly.graph_objects as go
26
+ if data is None:
27
+ raise ValueError("data= is required: a DataFrame with "
28
+ "the value and stage columns")
29
+ v = get_column(data, value, "value").to_numpy(dtype=float)
30
+ s1 = get_column(data, stage1, "stage1").astype(str)
31
+ s2 = get_column(data, stage2, "stage2").astype(str)
32
+ s3 = (get_column(data, stage3, "stage3").astype(str)
33
+ if stage3 is not None else None)
34
+
35
+ sources = list(pd.unique(s1)) # first-seen order
36
+ mids = list(pd.unique(s2))
37
+ dests = list(pd.unique(s3)) if s3 is not None else []
38
+ nA, nB, nC = len(sources), len(mids), len(dests)
39
+ nodes = sources + mids + dests
40
+ i_src = {lv: i for i, lv in enumerate(sources)}
41
+ i_mid = {lv: nA + i for i, lv in enumerate(mids)}
42
+ i_dest = {lv: nA + nB + i for i, lv in enumerate(dests)}
43
+
44
+ # source palette (one color per stage1 level); the flow color
45
+ df = pd.DataFrame({"s1": s1, "s2": s2, "v": v})
46
+ pal = {lv: to_hex(fill[i % len(fill)]) if fill else
47
+ to_hex(BASE_COLORS[i % len(BASE_COLORS)])
48
+ for i, lv in enumerate(sources)}
49
+
50
+ # links: stage1->stage2, then (colored by stage1) stage2->stage3
51
+ src, tgt, val, lcol = [], [], [], []
52
+ for (a, b), g in df.groupby(["s1", "s2"], sort=False):
53
+ src.append(i_src[a])
54
+ tgt.append(i_mid[b])
55
+ val.append(float(g["v"].sum()))
56
+ lcol.append(make_trans(pal[a], link_alpha))
57
+ if dests:
58
+ df3 = pd.DataFrame({"s1": s1, "s2": s2, "s3": s3, "v": v})
59
+ for (a, b, c), g in df3.groupby(["s1", "s2", "s3"],
60
+ sort=False):
61
+ src.append(i_mid[b])
62
+ tgt.append(i_dest[c])
63
+ val.append(float(g["v"].sum()))
64
+ lcol.append(make_trans(pal[a], link_alpha))
65
+
66
+ # node colors and positions
67
+ if nodes_gray:
68
+ node_color = [to_hex(neutral_gray)] * len(nodes)
69
+ else:
70
+ node_color = ([pal[lv] for lv in sources]
71
+ + [to_hex(neutral_gray)] * (nB + nC))
72
+ x_mid = 0.5 if nC else 0.98
73
+ node_x = ([0.02] * nA + [x_mid] * nB + [0.98] * nC)
74
+
75
+ def y_even(n):
76
+ return ([0.5] if n <= 1
77
+ else list(np.linspace(0.1, 0.9, n)))
78
+ node_y = y_even(nA) + y_even(nB) + (y_even(nC) if nC else [])
79
+
80
+ arrow = "→"
81
+ if title:
82
+ ttl = f"<b>{title}</b>"
83
+ elif dests:
84
+ ttl = (f"<b>{stage1} {arrow} {stage2} {arrow} "
85
+ f"{stage3}</b>")
86
+ else:
87
+ ttl = f"<b>{stage1} {arrow} {stage2}</b>"
88
+
89
+ h = max(0.80, min(0.94, 0.94 - 0.50 * abs(lift_y)))
90
+ y1 = max(0.0, min(0.02 + lift_y, 1 - h))
91
+ y_dom = [y1, min(1.0, y1 + h)]
92
+
93
+ fig = go.Figure(go.Sankey(
94
+ arrangement="fixed",
95
+ domain=dict(x=[0, 1], y=y_dom),
96
+ node=dict(
97
+ label=nodes, color=node_color, x=node_x, y=node_y,
98
+ pad=12, thickness=16,
99
+ line=dict(color=to_hex(border_gray), width=1),
100
+ hovertemplate="%{label}<extra></extra>"),
101
+ link=dict(
102
+ source=src, target=tgt, value=val, color=lcol,
103
+ hovertemplate=(f"%{{source.label}} {arrow} "
104
+ "%{target.label}<br>"
105
+ f"{value}: %{{value:,.{digits_d}f}}"
106
+ "<extra></extra>"))))
107
+ fig.update_layout(
108
+ title=dict(text=ttl, x=0.5,
109
+ y=min(0.98, y_dom[1] + 0.03),
110
+ xanchor="center", yanchor="top",
111
+ font=dict(size=round(18 * labels_size))),
112
+ margin=dict(t=90, r=30, b=30, l=30),
113
+ font=dict(size=round(15 * labels_size)),
114
+ template=None,
115
+ paper_bgcolor=to_hex(get_option("window_fill", "white")))
116
+ return fig