hds 0.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hds/__init__.py +4 -0
- hds/plot.py +964 -0
- hds/stat.py +234 -0
- hds-0.1.1.dist-info/METADATA +40 -0
- hds-0.1.1.dist-info/RECORD +8 -0
- hds-0.1.1.dist-info/WHEEL +5 -0
- hds-0.1.1.dist-info/licenses/LICENSE +21 -0
- hds-0.1.1.dist-info/top_level.txt +3 -0
hds/__init__.py
ADDED
hds/plot.py
ADDED
|
@@ -0,0 +1,964 @@
|
|
|
1
|
+
# 관련 라이브러리 호출
|
|
2
|
+
import requests
|
|
3
|
+
from bs4 import BeautifulSoup as bts
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
import platform
|
|
8
|
+
import shutil
|
|
9
|
+
import subprocess
|
|
10
|
+
import matplotlib
|
|
11
|
+
import glob
|
|
12
|
+
import numpy as np
|
|
13
|
+
import pandas as pd
|
|
14
|
+
from scipy import stats
|
|
15
|
+
import seaborn as sns
|
|
16
|
+
import matplotlib.pyplot as plt
|
|
17
|
+
import inspect
|
|
18
|
+
from sklearn.tree import DecisionTreeRegressor
|
|
19
|
+
from sklearn.tree import DecisionTreeClassifier
|
|
20
|
+
from sklearn.tree import export_graphviz
|
|
21
|
+
import graphviz
|
|
22
|
+
from sklearn.cluster import KMeans
|
|
23
|
+
from sklearn import metrics
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# 구글 폰트 파일 목록을 반환하는 함수
|
|
27
|
+
def search_google_font_file(font_name: str) -> list:
|
|
28
|
+
'''
|
|
29
|
+
이 함수는 구글 폰트(https://fonts.google.com)에 등록된 폰트명을 지정하면
|
|
30
|
+
해당 폰트명의 ttf 파일 목록을 반환합니다.
|
|
31
|
+
|
|
32
|
+
매개변수:
|
|
33
|
+
font_name: 구글 폰트명을 문자열로 지정합니다.
|
|
34
|
+
|
|
35
|
+
반환값:
|
|
36
|
+
구글 폰트 ttf 파일명을 리스트로 반환합니다.
|
|
37
|
+
'''
|
|
38
|
+
# 구글 폰트명에서 공백 제거
|
|
39
|
+
font_name_removed = font_name.replace(' ', '')
|
|
40
|
+
|
|
41
|
+
# 구글 폰트명 URL 생성
|
|
42
|
+
url = f'https://github.com/google/fonts/tree/main/ofl/{font_name_removed.lower()}'
|
|
43
|
+
|
|
44
|
+
# 구글 폰트 파일 목록 내려받기
|
|
45
|
+
res = requests.get(url)
|
|
46
|
+
if res.status_code == 200:
|
|
47
|
+
soup = bts(markup = res.text, features = 'html.parser')
|
|
48
|
+
items = soup.select('script[type="application/json"][data-target="react-app.embeddedData"]')
|
|
49
|
+
dat = json.loads(s = items[0].text)
|
|
50
|
+
files = dat['payload']['tree']['items']
|
|
51
|
+
return [file['name'] for file in files if '.ttf' in file['name']]
|
|
52
|
+
else:
|
|
53
|
+
raise FileNotFoundError(f'Font not found with {font_name}')
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# 구글 폰트 파일을 다운로드 폴더에 내려받는 함수
|
|
57
|
+
def download_google_font_file(font_file: str) -> str:
|
|
58
|
+
'''
|
|
59
|
+
이 함수는 구글 폰트 ttf 파일명을 지정하면 다운로드 폴더에 내려받습니다.
|
|
60
|
+
|
|
61
|
+
매개변수:
|
|
62
|
+
font_file: 구글 폰트 ttf 파일명을 문자열로 지정합니다.
|
|
63
|
+
|
|
64
|
+
반환값:
|
|
65
|
+
다운로드 폴더에 내려받은 구글 폰트 ttf 파일명을 문자열로 반환합니다.
|
|
66
|
+
'''
|
|
67
|
+
# 다운로드 폴더 경로 지정
|
|
68
|
+
download_path = os.path.join(os.path.expanduser('~'), 'Downloads')
|
|
69
|
+
os.makedirs(name = download_path, exist_ok = True)
|
|
70
|
+
font_path = os.path.join(download_path, font_file)
|
|
71
|
+
|
|
72
|
+
# 구글 폰트명 생성
|
|
73
|
+
font_name = re.split(pattern = r'(-)|(\[)|(\.ttf)', string = font_file)[0].lower()
|
|
74
|
+
|
|
75
|
+
# 구글 폰트 파일 다운로드 URL 생성
|
|
76
|
+
domain = 'https://raw.githubusercontent.com/google/fonts/refs/heads/main/ofl/'
|
|
77
|
+
url = os.path.join(domain, font_name, font_file)
|
|
78
|
+
|
|
79
|
+
# 구글 폰트 파일 내려받기
|
|
80
|
+
res = requests.get(url)
|
|
81
|
+
if res.status_code == 200:
|
|
82
|
+
with open(file = font_path, mode = 'wb') as file:
|
|
83
|
+
file.write(res.content)
|
|
84
|
+
print(f'Downloaded to {font_path}')
|
|
85
|
+
return font_path
|
|
86
|
+
else:
|
|
87
|
+
raise FileNotFoundError(f'Font not found at {url}')
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# 구글 폰트를 설치하고 다운로드 폴더에서 삭제하는 함수
|
|
91
|
+
def install_google_font_path(font_path: str) -> None:
|
|
92
|
+
'''
|
|
93
|
+
이 함수는 다운로드 폴더에 내려받은 구글 폰트 ttf 파일명을 운영체제에 맞게 설치하고
|
|
94
|
+
다운로드 폴더에 있는 ttf 파일명을 삭제합니다.
|
|
95
|
+
|
|
96
|
+
매개변수:
|
|
97
|
+
font_path: 다운로드 폴더에 내려받은 구글 폰트 ttf 파일명을 문자열로 지정합니다.
|
|
98
|
+
|
|
99
|
+
반환값:
|
|
100
|
+
없습니다.
|
|
101
|
+
'''
|
|
102
|
+
# 운영체제별 구글 폰트 설치 경로 지정
|
|
103
|
+
system = platform.system()
|
|
104
|
+
if system == 'Windows':
|
|
105
|
+
fonts_dir = os.path.join(os.getenv(key = 'WINDIR'), 'Fonts')
|
|
106
|
+
shutil.copy(src = font_path, dst = fonts_dir)
|
|
107
|
+
elif system == 'Darwin':
|
|
108
|
+
fonts_dir = os.path.expanduser('~/Library/Fonts')
|
|
109
|
+
shutil.copy(src = font_path, dst = fonts_dir)
|
|
110
|
+
elif system == 'Linux':
|
|
111
|
+
fonts_dir = os.path.expanduser('~/.fonts')
|
|
112
|
+
os.makedirs(name = fonts_dir, exist_ok = True)
|
|
113
|
+
shutil.copy(src = font_path, dst = fonts_dir)
|
|
114
|
+
subprocess.run(['fc-cache', '-f', '-v'])
|
|
115
|
+
else:
|
|
116
|
+
raise OSError('Unsupported operating system')
|
|
117
|
+
|
|
118
|
+
# 실행 완료 문구 출력
|
|
119
|
+
print(f'Installed font at {fonts_dir}')
|
|
120
|
+
|
|
121
|
+
# 구글 폰트 파일 삭제
|
|
122
|
+
os.remove(font_path)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
# 구글 폰트를 설치하고 matplotlib 임시 폴더에 있는 json 파일을 삭제하는 함수
|
|
126
|
+
def add_google_font(font_name: str) -> None:
|
|
127
|
+
'''
|
|
128
|
+
이 함수는 구글 폰트명을 지정하면 해당 폰트의 ttf 파일명을 다운로드 폴더에
|
|
129
|
+
내려받은 다음 운영체제에 맞게 설치하고 다운로드 폴더에서 삭제합니다.
|
|
130
|
+
|
|
131
|
+
매개변수:
|
|
132
|
+
font_name: 구글 폰트명을 문자열로 지정합니다.
|
|
133
|
+
|
|
134
|
+
반환값:
|
|
135
|
+
없습니다.
|
|
136
|
+
'''
|
|
137
|
+
# 구글 폰트 파일 목록 생성
|
|
138
|
+
font_files = search_google_font_file(font_name)
|
|
139
|
+
|
|
140
|
+
# 반복문 실행
|
|
141
|
+
for font_file in font_files:
|
|
142
|
+
try:
|
|
143
|
+
# 구글 폰트 파일을 다운로드 폴더에 내려받기
|
|
144
|
+
font_path = download_google_font_file(font_file)
|
|
145
|
+
|
|
146
|
+
# 구글 폰트를 설치하고 다운로드 폴더에서 삭제
|
|
147
|
+
install_google_font_path(font_path)
|
|
148
|
+
|
|
149
|
+
except Exception as e:
|
|
150
|
+
print(f'Error: {e}')
|
|
151
|
+
|
|
152
|
+
# matplotlib 임시 폴더에 있는 json 파일 삭제
|
|
153
|
+
path = matplotlib.get_cachedir()
|
|
154
|
+
file = glob.glob(f'{path}/fontlist-*.json')[0]
|
|
155
|
+
os.remove(path = file)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# 범례가 있을 때만 제거하도록 변경
|
|
159
|
+
def remove_legend():
|
|
160
|
+
legend = plt.gca().get_legend()
|
|
161
|
+
if legend is not None:
|
|
162
|
+
legend.remove()
|
|
163
|
+
|
|
164
|
+
# 집단별 상자 수염 그림을 그리는 함수
|
|
165
|
+
def box_group(data: pd.DataFrame, x: str, y: str, palette: list = None, legend: bool = False) -> None:
|
|
166
|
+
'''
|
|
167
|
+
이 함수는 범주형 변수(x축)에 따라 연속형 변수(y축)의 상자 수염 그림을 그립니다.
|
|
168
|
+
상자에 빨간 점은 해당 범주의 평균이며, 가로 직선은 전체 평균입니다.
|
|
169
|
+
|
|
170
|
+
매개변수:
|
|
171
|
+
data: 데이터프레임을 지정합니다.
|
|
172
|
+
x: 범주형 변수명을 문자열로 지정합니다.
|
|
173
|
+
y: 연속형 변수명을 문자열로 지정합니다.
|
|
174
|
+
palette: 팔레트를 리스트로 지정합니다.
|
|
175
|
+
legend: 범례 추가 여부를 True 또는 False로 지정합니다.(기본값: False)
|
|
176
|
+
|
|
177
|
+
반환값:
|
|
178
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
179
|
+
'''
|
|
180
|
+
avg = data.groupby(by = x)[y].mean()
|
|
181
|
+
|
|
182
|
+
sns.boxplot(
|
|
183
|
+
data = data,
|
|
184
|
+
x = x,
|
|
185
|
+
y = y,
|
|
186
|
+
hue = x,
|
|
187
|
+
order = avg.index,
|
|
188
|
+
palette = palette,
|
|
189
|
+
flierprops = {
|
|
190
|
+
'marker': 'o',
|
|
191
|
+
'markersize': 3,
|
|
192
|
+
'markerfacecolor': 'pink',
|
|
193
|
+
'markeredgecolor': 'red',
|
|
194
|
+
'markeredgewidth': 0.2
|
|
195
|
+
},
|
|
196
|
+
linecolor = '0.5',
|
|
197
|
+
linewidth = 0.5
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
plt.axhline(
|
|
201
|
+
y = data[y].mean(),
|
|
202
|
+
color = 'red',
|
|
203
|
+
linewidth = 0.5,
|
|
204
|
+
linestyle = '--'
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
for i, v in enumerate(avg):
|
|
208
|
+
plt.text(
|
|
209
|
+
x = i,
|
|
210
|
+
y = v,
|
|
211
|
+
s = f'{v:,.2f}',
|
|
212
|
+
ha = 'center',
|
|
213
|
+
va = 'center',
|
|
214
|
+
fontsize = 6,
|
|
215
|
+
fontweight = 'bold'
|
|
216
|
+
)
|
|
217
|
+
|
|
218
|
+
if legend is True:
|
|
219
|
+
plt.legend(loc = 'center left', bbox_to_anchor = (1, 0.5), title = x)
|
|
220
|
+
else:
|
|
221
|
+
remove_legend()
|
|
222
|
+
|
|
223
|
+
plt.title(label = f'{x} 범주별 {y}의 평균 비교', fontdict = {'fontweight': 'bold'});
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
# 두 연속형 변수로 산점도를 그리는 함수
|
|
227
|
+
def scatter(data: pd.DataFrame, x: str, y: str, color: str = '0.3') -> None:
|
|
228
|
+
'''
|
|
229
|
+
이 함수는 두 연속형 변수의 산점도를 그립니다.
|
|
230
|
+
|
|
231
|
+
매개변수:
|
|
232
|
+
data: 데이터프레임을 지정합니다.
|
|
233
|
+
x: 원인이 되는 연속형 변수명을 문자열로 지정합니다.
|
|
234
|
+
y: 결과가 되는 연속형 변수명을 문자열로 지정합니다.
|
|
235
|
+
color: 점의 채우기 색을 문자열로 지정합니다.(기본값: '0.3')
|
|
236
|
+
|
|
237
|
+
반환값:
|
|
238
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
239
|
+
'''
|
|
240
|
+
sns.scatterplot(
|
|
241
|
+
data = data,
|
|
242
|
+
x = x,
|
|
243
|
+
y = y,
|
|
244
|
+
color = color
|
|
245
|
+
)
|
|
246
|
+
|
|
247
|
+
plt.title(label = f'{x}와(과) {y}의 관계', fontdict = {'fontweight': 'bold'});
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
# 두 연속형 변수로 산점도와 회귀직선을 그리는 함수
|
|
251
|
+
def regline(data: pd.DataFrame, x: str, y: str, color: str = '0.3', size: int = 15) -> None:
|
|
252
|
+
'''
|
|
253
|
+
이 함수는 두 연속형 변수의 산점도에 회귀직선을 그립니다.
|
|
254
|
+
|
|
255
|
+
매개변수:
|
|
256
|
+
data: 데이터프레임을 지정합니다.
|
|
257
|
+
x: 원인이 되는 연속형 변수명을 문자열로 지정합니다.
|
|
258
|
+
y: 결과가 되는 연속형 변수명을 문자열로 지정합니다.
|
|
259
|
+
color: 점의 채우기 색을 문자열로 지정합니다.(기본값: '0.3')
|
|
260
|
+
size: 점의 크기를 정수로 지정합니다.(기본값: 15)
|
|
261
|
+
|
|
262
|
+
반환값:
|
|
263
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
264
|
+
'''
|
|
265
|
+
sns.regplot(
|
|
266
|
+
data = data,
|
|
267
|
+
x = x,
|
|
268
|
+
y = y,
|
|
269
|
+
ci = None,
|
|
270
|
+
scatter_kws = {
|
|
271
|
+
'facecolor': color,
|
|
272
|
+
'edgecolor': '1',
|
|
273
|
+
's': size,
|
|
274
|
+
'alpha': 0.2
|
|
275
|
+
},
|
|
276
|
+
line_kws = {
|
|
277
|
+
'color': 'red',
|
|
278
|
+
'linewidth': 1.5
|
|
279
|
+
}
|
|
280
|
+
)
|
|
281
|
+
|
|
282
|
+
plt.title(label = f'{x}와(과) {y}의 관계', fontdict = {'fontweight': 'bold'});
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
# 범주형 변수의 도수로 막대 그래프를 그리는 함수
|
|
286
|
+
def bar_freq(data: pd.DataFrame, x: str, color: str = None, palette: list = None, legend: bool = False) -> None:
|
|
287
|
+
'''
|
|
288
|
+
이 함수는 범주형 변수의 도수를 내림차순 정렬한 막대 그래프를 그립니다.
|
|
289
|
+
|
|
290
|
+
매개변수:
|
|
291
|
+
data: 데이터프레임을 지정합니다.
|
|
292
|
+
x: 범주형 변수명을 문자열로 지정합니다.
|
|
293
|
+
color: 점의 채우기 색을 문자열로 지정합니다.
|
|
294
|
+
palette: 팔레트를 리스트로 지정합니다.
|
|
295
|
+
legend: 범례 추가 여부를 True 또는 False로 지정합니다.(기본값: False)
|
|
296
|
+
|
|
297
|
+
반환값:
|
|
298
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
299
|
+
'''
|
|
300
|
+
grp = data[x].value_counts()
|
|
301
|
+
v_max = grp.values.max()
|
|
302
|
+
space = np.ceil(v_max * 0.01)
|
|
303
|
+
|
|
304
|
+
sns.countplot(
|
|
305
|
+
data = data,
|
|
306
|
+
x = x,
|
|
307
|
+
hue = x,
|
|
308
|
+
order = grp.index,
|
|
309
|
+
color = color,
|
|
310
|
+
palette = palette,
|
|
311
|
+
)
|
|
312
|
+
|
|
313
|
+
for i, v in enumerate(grp):
|
|
314
|
+
plt.text(
|
|
315
|
+
x = i,
|
|
316
|
+
y = v + space,
|
|
317
|
+
s = v,
|
|
318
|
+
ha = 'center',
|
|
319
|
+
va = 'bottom',
|
|
320
|
+
c = 'black',
|
|
321
|
+
fontsize = 8,
|
|
322
|
+
fontweight = 'bold'
|
|
323
|
+
)
|
|
324
|
+
|
|
325
|
+
if legend is True:
|
|
326
|
+
plt.legend(loc = 'center left', bbox_to_anchor = (1, 0.5), title = x)
|
|
327
|
+
else:
|
|
328
|
+
remove_legend()
|
|
329
|
+
|
|
330
|
+
plt.ylim(0, v_max * 1.2)
|
|
331
|
+
plt.title(label = '목표변수의 범주별 도수 비교', fontdict = {'fontweight': 'bold'});
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
# 범주형 변수를 소그룹으로 나누고 도수로 펼친 막대 그래프를 그리는 함수
|
|
335
|
+
def bar_dodge_freq(data: pd.DataFrame, x: str, g: str, palette: list = None) -> None:
|
|
336
|
+
'''
|
|
337
|
+
이 함수는 범주형 변수를 소그룹으로 나누고 도수로 펼친 막대 그래프를 그립니다.
|
|
338
|
+
|
|
339
|
+
매개변수:
|
|
340
|
+
data: 데이터프레임을 지정합니다.
|
|
341
|
+
x: 범주형 변수명을 문자열로 지정합니다.
|
|
342
|
+
g: x를 소그룹으로 나눌 범주형 변수명을 문자열로 지정합니다.
|
|
343
|
+
palette: 팔레트를 리스트로 지정합니다.
|
|
344
|
+
|
|
345
|
+
반환값:
|
|
346
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
347
|
+
'''
|
|
348
|
+
grp = data.groupby(by = [x, g]).count().iloc[:, 0]
|
|
349
|
+
v_max = grp.values.max()
|
|
350
|
+
space = np.ceil(v_max * 0.01)
|
|
351
|
+
|
|
352
|
+
sns.countplot(
|
|
353
|
+
data = data,
|
|
354
|
+
x = x,
|
|
355
|
+
hue = g,
|
|
356
|
+
order = grp.index.levels[0],
|
|
357
|
+
hue_order = grp.index.levels[1],
|
|
358
|
+
palette = palette,
|
|
359
|
+
)
|
|
360
|
+
|
|
361
|
+
for i, v in enumerate(grp):
|
|
362
|
+
if i % 2 == 0:
|
|
363
|
+
i = i/2 - 0.2
|
|
364
|
+
else:
|
|
365
|
+
i = (i-1)/2 + 0.2
|
|
366
|
+
plt.text(
|
|
367
|
+
x = i,
|
|
368
|
+
y = v + space,
|
|
369
|
+
s = v,
|
|
370
|
+
ha = 'center',
|
|
371
|
+
va = 'bottom',
|
|
372
|
+
fontsize = 8,
|
|
373
|
+
fontweight = 'bold'
|
|
374
|
+
)
|
|
375
|
+
|
|
376
|
+
plt.ylim(0, v_max * 1.2)
|
|
377
|
+
plt.title(label = f'{x}의 범주별 {g}의 도수 비교', fontdict = {'fontweight': 'bold'})
|
|
378
|
+
plt.legend(loc = 'center left', bbox_to_anchor = (1, 0.5), fontsize = 8);
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
# 범주형 변수를 소그룹으로 나누고 도수로 쌓은 막대 그래프를 그리는 함수
|
|
382
|
+
def bar_stack_freq(data: pd.DataFrame, x: str, g: str, kind: str = 'bar', palette: list = None) -> None:
|
|
383
|
+
'''
|
|
384
|
+
이 함수는 범주형 변수를 소그룹으로 나누고 도수로 쌓은 막대 그래프를 그립니다.
|
|
385
|
+
|
|
386
|
+
매개변수:
|
|
387
|
+
data: 데이터프레임을 지정합니다.
|
|
388
|
+
x: 범주형 변수명을 문자열로 지정합니다.
|
|
389
|
+
g: x를 소그룹으로 나눌 범주형 변수명을 문자열로 지정합니다.
|
|
390
|
+
kind: 막대 그래프의 종류를 'bar' 또는 'barh'로 지정합니다.(기본값: 'bar')
|
|
391
|
+
palette: 팔레트를 리스트로 지정합니다.
|
|
392
|
+
|
|
393
|
+
반환값:
|
|
394
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
395
|
+
'''
|
|
396
|
+
p = data[g].unique().size
|
|
397
|
+
|
|
398
|
+
pv = pd.pivot_table(
|
|
399
|
+
data = data,
|
|
400
|
+
index = x,
|
|
401
|
+
columns = g,
|
|
402
|
+
aggfunc = 'count'
|
|
403
|
+
)
|
|
404
|
+
|
|
405
|
+
pv = pv.iloc[:, 0:p].sort_index()
|
|
406
|
+
pv.columns = pv.columns.droplevel(level = 0)
|
|
407
|
+
pv.columns.name = None
|
|
408
|
+
pv = pv.reset_index()
|
|
409
|
+
cols = pv.columns[1:]
|
|
410
|
+
cumsum = pv[cols].cumsum(axis = 1)
|
|
411
|
+
|
|
412
|
+
if type(palette) == list:
|
|
413
|
+
palette = sns.set_palette(sns.color_palette(palette))
|
|
414
|
+
|
|
415
|
+
pv.plot(
|
|
416
|
+
x = x,
|
|
417
|
+
kind = kind,
|
|
418
|
+
stacked = True,
|
|
419
|
+
rot = 0,
|
|
420
|
+
legend = 'reverse',
|
|
421
|
+
colormap = palette
|
|
422
|
+
)
|
|
423
|
+
|
|
424
|
+
if kind == 'bar':
|
|
425
|
+
for col in cols:
|
|
426
|
+
for i, (v1, v2) in enumerate(zip(cumsum[col], pv[col])):
|
|
427
|
+
plt.text(
|
|
428
|
+
x = i,
|
|
429
|
+
y = v1 - v2/2,
|
|
430
|
+
s = v2,
|
|
431
|
+
ha = 'center',
|
|
432
|
+
va = 'center',
|
|
433
|
+
c = 'black',
|
|
434
|
+
fontsize = 8,
|
|
435
|
+
fontweight = 'bold'
|
|
436
|
+
)
|
|
437
|
+
elif kind == 'barh':
|
|
438
|
+
for col in cols:
|
|
439
|
+
for i, (v1, v2) in enumerate(zip(cumsum[col], pv[col])):
|
|
440
|
+
plt.text(
|
|
441
|
+
x = v1 - v2/2,
|
|
442
|
+
y = i,
|
|
443
|
+
s = v2,
|
|
444
|
+
ha = 'center',
|
|
445
|
+
va = 'center',
|
|
446
|
+
c = 'black',
|
|
447
|
+
fontsize = 8,
|
|
448
|
+
fontweight = 'bold'
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
plt.title(label = f'{x}의 범주별 {g}의 도수 비교', fontweight = 'bold')
|
|
452
|
+
plt.legend(loc = 'center left', bbox_to_anchor = (1, 0.5), fontsize = 8);
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
# 범주형 변수를 소그룹으로 나누고 상대도수로 쌓은 막대 그래프를 그리는 함수
|
|
456
|
+
def bar_stack_prop(data: pd.DataFrame, x: str, g: str, kind: str = 'bar', palette: list = None) -> None:
|
|
457
|
+
'''
|
|
458
|
+
이 함수는 범주형 변수를 소그룹으로 나누고 상대도수로 쌓은 막대 그래프를 그립니다.
|
|
459
|
+
|
|
460
|
+
매개변수:
|
|
461
|
+
data: 데이터프레임을 지정합니다.
|
|
462
|
+
x: 범주형 변수명을 문자열로 지정합니다.
|
|
463
|
+
g: x를 소그룹으로 나눌 범주형 변수명을 문자열로 지정합니다.
|
|
464
|
+
kind: 막대 그래프의 종류를 'bar' 또는 'barh'로 지정합니다.(기본값: 'bar')
|
|
465
|
+
palette: 팔레트를 리스트로 지정합니다.(기본값: None)
|
|
466
|
+
|
|
467
|
+
반환값:
|
|
468
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
469
|
+
'''
|
|
470
|
+
p = data[g].unique().size
|
|
471
|
+
|
|
472
|
+
pv = pd.pivot_table(
|
|
473
|
+
data = data,
|
|
474
|
+
index = x,
|
|
475
|
+
columns = g,
|
|
476
|
+
aggfunc = 'count'
|
|
477
|
+
)
|
|
478
|
+
|
|
479
|
+
pv = pv.iloc[:, 0:p].sort_index()
|
|
480
|
+
pv.columns = pv.columns.droplevel(level = 0)
|
|
481
|
+
pv.columns.name = None
|
|
482
|
+
pv = pv.reset_index()
|
|
483
|
+
cols = pv.columns[1:]
|
|
484
|
+
rowsum = pv[cols].apply(func = sum, axis = 1)
|
|
485
|
+
pv[cols] = pv[cols].div(rowsum, 0) * 100
|
|
486
|
+
cumsum = pv[cols].cumsum(axis = 1)
|
|
487
|
+
|
|
488
|
+
if type(palette) == list:
|
|
489
|
+
palette = sns.set_palette(sns.color_palette(palette))
|
|
490
|
+
|
|
491
|
+
pv.plot(
|
|
492
|
+
x = x,
|
|
493
|
+
kind = kind,
|
|
494
|
+
stacked = True,
|
|
495
|
+
rot = 0,
|
|
496
|
+
legend = 'reverse',
|
|
497
|
+
colormap = palette,
|
|
498
|
+
mark_right = True
|
|
499
|
+
)
|
|
500
|
+
|
|
501
|
+
if kind == 'bar':
|
|
502
|
+
for col in cols:
|
|
503
|
+
for i, (v1, v2) in enumerate(zip(cumsum[col], pv[col])):
|
|
504
|
+
v3 = f'{np.round(v2, 1)}%'
|
|
505
|
+
plt.text(
|
|
506
|
+
x = i,
|
|
507
|
+
y = v1 - v2/2,
|
|
508
|
+
s = v3,
|
|
509
|
+
ha = 'center',
|
|
510
|
+
va = 'center',
|
|
511
|
+
c = 'black',
|
|
512
|
+
fontsize = 8,
|
|
513
|
+
fontweight = 'bold'
|
|
514
|
+
)
|
|
515
|
+
elif kind == 'barh':
|
|
516
|
+
for col in cols:
|
|
517
|
+
for i, (v1, v2) in enumerate(zip(cumsum[col], pv[col])):
|
|
518
|
+
v3 = f'{np.round(v2, 1)}%'
|
|
519
|
+
plt.text(
|
|
520
|
+
x = v1 - v2/2,
|
|
521
|
+
y = i,
|
|
522
|
+
s = v3,
|
|
523
|
+
ha = 'center',
|
|
524
|
+
va = 'center',
|
|
525
|
+
c = 'black',
|
|
526
|
+
fontsize = 8,
|
|
527
|
+
fontweight = 'bold'
|
|
528
|
+
)
|
|
529
|
+
|
|
530
|
+
plt.title(label = f'{x}의 범주별 {g}의 상대도수 비교', fontweight = 'bold')
|
|
531
|
+
plt.legend(loc = 'center left', bbox_to_anchor = (1, 0.5), fontsize = 8);
|
|
532
|
+
|
|
533
|
+
|
|
534
|
+
# 의사결정나무 모델 시각화
|
|
535
|
+
def tree_model(model, fileName: str = None, className: str = None) -> None:
|
|
536
|
+
'''
|
|
537
|
+
이 함수는 의사결정나무 모델을 시각화하여 png 파일로 저장합니다.
|
|
538
|
+
|
|
539
|
+
매개변수:
|
|
540
|
+
model: 사이킷런으로 적합한 의사결정나무 모델을 지정합니다.
|
|
541
|
+
fileName: 입력변수명을 문자열로 지정합니다.(기본값: None)
|
|
542
|
+
className: 분류 모델은 목표변수의 범주를 문자열로 지정합니다.(기본값: None)
|
|
543
|
+
|
|
544
|
+
반환값:
|
|
545
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
546
|
+
'''
|
|
547
|
+
if fileName == None:
|
|
548
|
+
global_objs = inspect.currentframe().f_back.f_globals.items()
|
|
549
|
+
result = [name for name, value in global_objs if value is model]
|
|
550
|
+
fileName = result[0]
|
|
551
|
+
|
|
552
|
+
if type(model) == DecisionTreeRegressor:
|
|
553
|
+
export_graphviz(
|
|
554
|
+
decision_tree = model,
|
|
555
|
+
out_file = f'{fileName}.dot',
|
|
556
|
+
feature_names = model.feature_names_in_,
|
|
557
|
+
filled = True,
|
|
558
|
+
leaves_parallel = False,
|
|
559
|
+
impurity = True
|
|
560
|
+
)
|
|
561
|
+
elif type(model) == DecisionTreeClassifier:
|
|
562
|
+
if className == None:
|
|
563
|
+
className = model.classes_
|
|
564
|
+
export_graphviz(
|
|
565
|
+
decision_tree = model,
|
|
566
|
+
out_file = f'{fileName}.dot',
|
|
567
|
+
class_names = className,
|
|
568
|
+
feature_names = model.feature_names_in_,
|
|
569
|
+
filled = True,
|
|
570
|
+
leaves_parallel = False,
|
|
571
|
+
impurity = True
|
|
572
|
+
)
|
|
573
|
+
|
|
574
|
+
with open(file = f'{fileName}.dot', mode = 'rt') as file:
|
|
575
|
+
graph = file.read()
|
|
576
|
+
graph = graphviz.Source(source = graph, format = 'png')
|
|
577
|
+
graph.render(filename = fileName)
|
|
578
|
+
|
|
579
|
+
os.remove(f'{fileName}')
|
|
580
|
+
os.remove(f'{fileName}.dot')
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
# 입력변수별 중요도 시각화
|
|
584
|
+
def feature_importance(model, palette: str = 'Spectral') -> None:
|
|
585
|
+
'''
|
|
586
|
+
이 함수는 입력변수별 중요도를 막대 그래프로 시각화합니다.
|
|
587
|
+
|
|
588
|
+
매개변수:
|
|
589
|
+
model: 사이킷런으로 적합한 분류 모델을 지정합니다.
|
|
590
|
+
palette: 팔레트를 문자열로 지정합니다.(기본값: Spectral)
|
|
591
|
+
|
|
592
|
+
반환값:
|
|
593
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
594
|
+
'''
|
|
595
|
+
if 'LGBM' in str(type(model)):
|
|
596
|
+
fi = pd.DataFrame(
|
|
597
|
+
data = model.feature_importances_,
|
|
598
|
+
index = model.feature_name_,
|
|
599
|
+
columns = ['importance']
|
|
600
|
+
)
|
|
601
|
+
else:
|
|
602
|
+
fi = pd.DataFrame(
|
|
603
|
+
data = model.feature_importances_,
|
|
604
|
+
index = model.feature_names_in_,
|
|
605
|
+
columns = ['importance']
|
|
606
|
+
) \
|
|
607
|
+
.sort_values(by = 'importance', ascending = False) \
|
|
608
|
+
.reset_index()
|
|
609
|
+
|
|
610
|
+
sns.barplot(
|
|
611
|
+
data = fi,
|
|
612
|
+
x = 'importance',
|
|
613
|
+
y = 'index',
|
|
614
|
+
hue = 'index',
|
|
615
|
+
palette = palette,
|
|
616
|
+
# legend = True
|
|
617
|
+
)
|
|
618
|
+
|
|
619
|
+
for i, r in fi.iterrows():
|
|
620
|
+
plt.text(
|
|
621
|
+
x = r['importance'] + 0.01,
|
|
622
|
+
y = i,
|
|
623
|
+
s = f"{r['importance']:.3f}",
|
|
624
|
+
ha = 'left',
|
|
625
|
+
va = 'center',
|
|
626
|
+
fontsize = 8,
|
|
627
|
+
fontweight = 'bold'
|
|
628
|
+
)
|
|
629
|
+
|
|
630
|
+
plt.xlim(0, fi['importance'].max() * 1.2)
|
|
631
|
+
plt.title(label = 'Feature Importances', fontdict = {'fontweight': 'bold'})
|
|
632
|
+
plt.xlabel(xlabel = 'Feature Importances')
|
|
633
|
+
plt.ylabel(ylabel = 'Feature');
|
|
634
|
+
|
|
635
|
+
|
|
636
|
+
# 의사결정나무 모델 가지치기 단계 그래프 시각화
|
|
637
|
+
def step(data: pd.DataFrame, x: str = 'alpha', y: str = 'impurity', color: str = 'blue', title: str = None, xangle: int = None) -> None:
|
|
638
|
+
'''
|
|
639
|
+
이 함수는 의사결정나무 모델의 사후 가지치기 결과를 단계 그래프로 시각화합니다.
|
|
640
|
+
|
|
641
|
+
매개변수:
|
|
642
|
+
data: 의사결정나무 모델의 가지치기 단계별 비용 복잡도 파라미터를 데이터프레임으로 지정합니다.
|
|
643
|
+
x: x축에 지정할 변수명을 문자열로 지정합니다.(기본값: 'alpha')
|
|
644
|
+
y: y축에 지정할 변수명을 문자열로 지정합니다.(기본값: 'impurity')
|
|
645
|
+
color: 선과 점의 색을 문자열로 지정합니다.(기본값: 'blue')
|
|
646
|
+
title: 그래프의 제목을 문자열로 지정합니다.(기본값: None)
|
|
647
|
+
xangle: x축 회전 각도를 정수로 지정합니다.(기본값: None)
|
|
648
|
+
|
|
649
|
+
반환값:
|
|
650
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
651
|
+
'''
|
|
652
|
+
sns.lineplot(
|
|
653
|
+
data = data,
|
|
654
|
+
x = x,
|
|
655
|
+
y = y,
|
|
656
|
+
color = color,
|
|
657
|
+
drawstyle = 'steps-pre',
|
|
658
|
+
label = y
|
|
659
|
+
)
|
|
660
|
+
|
|
661
|
+
sns.scatterplot(
|
|
662
|
+
data = data,
|
|
663
|
+
x = x,
|
|
664
|
+
y = y,
|
|
665
|
+
color = color,
|
|
666
|
+
s = 15
|
|
667
|
+
)
|
|
668
|
+
|
|
669
|
+
if title != None:
|
|
670
|
+
plt.title(label = title, fontweight = 'bold')
|
|
671
|
+
|
|
672
|
+
plt.xticks(rotation = xangle);
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
# 분류 모델의 ROC 곡선 시각화 및 AUC 계산 함수
|
|
676
|
+
def roc_curve(y_true: np.ndarray, y_prob: np.array, pos: str = None, color: str = None) -> None:
|
|
677
|
+
'''
|
|
678
|
+
이 함수는 분류 모델의 ROC 곡선을 그리고 AUC를 계산합니다.
|
|
679
|
+
|
|
680
|
+
매개변수:
|
|
681
|
+
y_true: 목표변수의 실제값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
682
|
+
y_prob: 목표변수의 예측 확률을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
683
|
+
pos: Positive 범주를 문자열로 지정합니다.
|
|
684
|
+
color: 곡선의 색을 문자열로 지정합니다.
|
|
685
|
+
|
|
686
|
+
반환:
|
|
687
|
+
ROC 곡선 그래프 외에 반환하는 객체는 없습니다.
|
|
688
|
+
'''
|
|
689
|
+
if isinstance(y_true, np.ndarray):
|
|
690
|
+
y_class = pd.Series(data = y_true).value_counts().sort_index()
|
|
691
|
+
else:
|
|
692
|
+
y_class = y_true.value_counts().sort_index()
|
|
693
|
+
|
|
694
|
+
if pos == None:
|
|
695
|
+
pos = y_class.loc[y_class == y_class.min()].index[0]
|
|
696
|
+
|
|
697
|
+
idx = np.where(y_class.index == pos)[0][0]
|
|
698
|
+
|
|
699
|
+
if y_prob.ndim == 2:
|
|
700
|
+
y_prob = y_prob[:, idx]
|
|
701
|
+
|
|
702
|
+
fpr, tpr, _ = metrics.roc_curve(
|
|
703
|
+
y_true = y_true,
|
|
704
|
+
y_score = y_prob,
|
|
705
|
+
pos_label = pos
|
|
706
|
+
)
|
|
707
|
+
|
|
708
|
+
auc_ = metrics.auc(x = fpr, y = tpr)
|
|
709
|
+
|
|
710
|
+
plt.plot(
|
|
711
|
+
fpr,
|
|
712
|
+
tpr,
|
|
713
|
+
color = color,
|
|
714
|
+
label = f'AUC = {auc_:.4f}',
|
|
715
|
+
linewidth = 1.0
|
|
716
|
+
)
|
|
717
|
+
|
|
718
|
+
plt.plot(
|
|
719
|
+
[0, 1],
|
|
720
|
+
[0, 1],
|
|
721
|
+
color = 'k',
|
|
722
|
+
linestyle = '--',
|
|
723
|
+
linewidth = 0.5
|
|
724
|
+
)
|
|
725
|
+
|
|
726
|
+
plt.title(label = 'ROC Curve', fontdict = {'fontweight': 'bold'})
|
|
727
|
+
plt.xlabel(xlabel = 'FPR')
|
|
728
|
+
plt.ylabel(ylabel = 'TPR')
|
|
729
|
+
plt.legend(loc = 'lower right', fontsize = 8);
|
|
730
|
+
|
|
731
|
+
|
|
732
|
+
# 분류 모델의 PR 곡선 시각화 및 AP 계산 함수
|
|
733
|
+
def pr_curve(y_true: np.ndarray, y_prob: np.array, pos: str = None, color: str = None) -> None:
|
|
734
|
+
'''
|
|
735
|
+
이 함수는 분류 모델의 PR 곡선을 그리고 AP를 계산합니다.
|
|
736
|
+
|
|
737
|
+
매개변수:
|
|
738
|
+
y_true: 목표변수의 실제값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
739
|
+
y_prob: 목표변수의 예측 확률을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
740
|
+
pos: Positive 범주를 문자열로 지정합니다.
|
|
741
|
+
color: 곡선의 색을 문자열로 지정합니다.
|
|
742
|
+
|
|
743
|
+
반환:
|
|
744
|
+
PR 곡선 그래프 외에 반환하는 객체는 없습니다.
|
|
745
|
+
'''
|
|
746
|
+
if isinstance(y_true, np.ndarray):
|
|
747
|
+
y_class = pd.Series(data = y_true).value_counts().sort_index()
|
|
748
|
+
else:
|
|
749
|
+
y_class = y_true.value_counts().sort_index()
|
|
750
|
+
|
|
751
|
+
if pos is None:
|
|
752
|
+
pos = y_class.loc[y_class == y_class.min()].index[0]
|
|
753
|
+
|
|
754
|
+
idx = np.where(y_class.index == pos)[0][0]
|
|
755
|
+
|
|
756
|
+
if y_prob.ndim == 2:
|
|
757
|
+
y_prob = y_prob[:, idx]
|
|
758
|
+
|
|
759
|
+
precision, recall, _ = metrics.precision_recall_curve(
|
|
760
|
+
y_true = y_true,
|
|
761
|
+
y_score = y_prob,
|
|
762
|
+
pos_label = pos
|
|
763
|
+
)
|
|
764
|
+
|
|
765
|
+
ap = metrics.average_precision_score(
|
|
766
|
+
y_true = y_true,
|
|
767
|
+
y_score = y_prob,
|
|
768
|
+
pos_label = pos
|
|
769
|
+
)
|
|
770
|
+
|
|
771
|
+
plt.plot(
|
|
772
|
+
recall,
|
|
773
|
+
precision,
|
|
774
|
+
color = color,
|
|
775
|
+
label = f'AP = {ap:.4f}',
|
|
776
|
+
linewidth = 1.0
|
|
777
|
+
)
|
|
778
|
+
|
|
779
|
+
plt.title(label = 'Precision-Recall Curve', fontdict = {'fontweight': 'bold'})
|
|
780
|
+
plt.xlabel(xlabel = 'Recall')
|
|
781
|
+
plt.ylabel(ylabel = 'Precision')
|
|
782
|
+
plt.legend(loc = 'lower left', fontsize = 8);
|
|
783
|
+
|
|
784
|
+
|
|
785
|
+
# 주성분 분석 스크리 도표 시각화
|
|
786
|
+
def screeplot(X: pd.DataFrame) -> None:
|
|
787
|
+
'''
|
|
788
|
+
이 함수는 주성분 점수 행렬을 스크리 도표로 시각화합니다.
|
|
789
|
+
|
|
790
|
+
매개변수:
|
|
791
|
+
X: 주성분 점수 행렬을 데이터프레임으로 지정합니다.
|
|
792
|
+
|
|
793
|
+
반환값:
|
|
794
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
795
|
+
'''
|
|
796
|
+
X = X.var()
|
|
797
|
+
n = len(X)
|
|
798
|
+
xticks = range(1, n + 1)
|
|
799
|
+
|
|
800
|
+
sns.lineplot(
|
|
801
|
+
x = xticks,
|
|
802
|
+
y = X,
|
|
803
|
+
color = 'blue',
|
|
804
|
+
linestyle = '-',
|
|
805
|
+
linewidth = 1,
|
|
806
|
+
marker = 'o'
|
|
807
|
+
)
|
|
808
|
+
|
|
809
|
+
plt.axhline(
|
|
810
|
+
y = 1,
|
|
811
|
+
color = 'red',
|
|
812
|
+
linestyle = '--',
|
|
813
|
+
linewidth = 0.5
|
|
814
|
+
)
|
|
815
|
+
|
|
816
|
+
plt.xticks(ticks = xticks)
|
|
817
|
+
plt.title(label = 'Scree Plot', fontdict = {'fontweight': 'bold'})
|
|
818
|
+
plt.xlabel(xlabel = 'Number of PC')
|
|
819
|
+
plt.ylabel(ylabel = 'Variance');
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
# 주성분 분석 행렬도 시각화
|
|
823
|
+
def biplot(score: pd.DataFrame, coefs: pd.DataFrame, x: int = 1, y: int = 2, zoom: float = 1.0) -> None:
|
|
824
|
+
'''
|
|
825
|
+
이 함수는 주성분 분석 결과를 스크리 도표로 시각화합니다.
|
|
826
|
+
|
|
827
|
+
매개변수:
|
|
828
|
+
score: 주성분 점수 행렬을 데이터프레임으로 지정합니다.
|
|
829
|
+
coefs: 고유벡터 행렬을 데이터프레임으로 지정합니다.
|
|
830
|
+
x: x축에 지정할 주성분의 인덱스를 정수로 지정합니다.(기본값: 1)
|
|
831
|
+
y: y축에 지정할 주성분의 인덱스를 정수로 지정합니다.(기본값: 2)
|
|
832
|
+
zoom: 변수의 벡터 크기를 조절하는 값을 실수로 지정합니다. (기본값: 1.0)
|
|
833
|
+
|
|
834
|
+
반환값:
|
|
835
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
836
|
+
'''
|
|
837
|
+
xs = score.iloc[:, x-1]
|
|
838
|
+
ys = score.iloc[:, y-1]
|
|
839
|
+
|
|
840
|
+
sns.scatterplot(
|
|
841
|
+
x = xs,
|
|
842
|
+
y = ys,
|
|
843
|
+
fc = 'silver',
|
|
844
|
+
ec = 'black',
|
|
845
|
+
s = 15,
|
|
846
|
+
alpha = 0.2
|
|
847
|
+
)
|
|
848
|
+
|
|
849
|
+
plt.axvline(
|
|
850
|
+
x = 0,
|
|
851
|
+
color = '0.5',
|
|
852
|
+
linestyle = '--',
|
|
853
|
+
linewidth = 0.5
|
|
854
|
+
)
|
|
855
|
+
|
|
856
|
+
plt.axhline(
|
|
857
|
+
y = 0,
|
|
858
|
+
color = '0.5',
|
|
859
|
+
linestyle = '--',
|
|
860
|
+
linewidth = 0.5
|
|
861
|
+
)
|
|
862
|
+
|
|
863
|
+
n = score.shape[1]
|
|
864
|
+
|
|
865
|
+
for i in range(n):
|
|
866
|
+
plt.arrow(
|
|
867
|
+
x = 0,
|
|
868
|
+
y = 0,
|
|
869
|
+
dx = coefs.iloc[i, x-1] * zoom,
|
|
870
|
+
dy = coefs.iloc[i, y-1] * zoom,
|
|
871
|
+
color = 'red',
|
|
872
|
+
linewidth = 0.5,
|
|
873
|
+
alpha = 0.5
|
|
874
|
+
)
|
|
875
|
+
|
|
876
|
+
plt.text(
|
|
877
|
+
x = coefs.iloc[i, x-1] * (zoom + 0.5),
|
|
878
|
+
y = coefs.iloc[i, y-1] * (zoom + 0.5),
|
|
879
|
+
s = coefs.index[i],
|
|
880
|
+
color = 'darkred',
|
|
881
|
+
ha = 'center',
|
|
882
|
+
va = 'center',
|
|
883
|
+
fontsize = 8,
|
|
884
|
+
fontweight = 'bold'
|
|
885
|
+
)
|
|
886
|
+
|
|
887
|
+
plt.title(label = 'Biplot with PC1 and PC2', fontdict = {'fontweight': 'bold'})
|
|
888
|
+
plt.xlabel(xlabel = 'PC{}'.format(x))
|
|
889
|
+
plt.ylabel(ylabel = 'PC{}'.format(y));
|
|
890
|
+
|
|
891
|
+
|
|
892
|
+
# k-평균 군집분석 WSS 단계 그래프 시각화
|
|
893
|
+
def wcss(X: pd.DataFrame, k: int = 3) -> None:
|
|
894
|
+
'''
|
|
895
|
+
이 함수는 군집별 편차 제곱합을 선 그래프로 시각화합니다.
|
|
896
|
+
|
|
897
|
+
매개변수:
|
|
898
|
+
X: 표준화된 데이터셋을 데이터프레임으로 지정합니다.
|
|
899
|
+
k: 군집 개수를 정수로 지정합니다.(기본값: 3)
|
|
900
|
+
|
|
901
|
+
반환값:
|
|
902
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
903
|
+
'''
|
|
904
|
+
ks = range(1, k + 1)
|
|
905
|
+
result = []
|
|
906
|
+
|
|
907
|
+
for k in ks:
|
|
908
|
+
model = KMeans(n_clusters = k, random_state = 0)
|
|
909
|
+
model.fit(X = X)
|
|
910
|
+
wcss = model.inertia_
|
|
911
|
+
result.append(wcss)
|
|
912
|
+
|
|
913
|
+
sns.lineplot(
|
|
914
|
+
x = ks,
|
|
915
|
+
y = result,
|
|
916
|
+
marker = 'o',
|
|
917
|
+
linestyle = '-',
|
|
918
|
+
linewidth = 1
|
|
919
|
+
)
|
|
920
|
+
|
|
921
|
+
plt.xticks(ticks = ks)
|
|
922
|
+
plt.title(label = 'Elbow Method', fontdict = {'fontweight': 'bold'})
|
|
923
|
+
plt.xlabel(xlabel = 'Number of clusters')
|
|
924
|
+
plt.ylabel(ylabel = 'Within Cluster Sum of Squares');
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
# k-평균 군집분석 실루엣 지수 시각화
|
|
928
|
+
def silhouette(X: pd.DataFrame, k: int = 3) -> None:
|
|
929
|
+
'''
|
|
930
|
+
이 함수는 군집별 실루엣 지수를 선 그래프로 시각화합니다.
|
|
931
|
+
|
|
932
|
+
매개변수:
|
|
933
|
+
X: 표준화된 데이터셋을 데이터프레임으로 지정합니다.
|
|
934
|
+
k: 군집 개수를 정수로 지정합니다.(기본값: 3)
|
|
935
|
+
|
|
936
|
+
반환값:
|
|
937
|
+
그래프 외에 반환하는 객체는 없습니다.
|
|
938
|
+
'''
|
|
939
|
+
ks = range(1, k + 1)
|
|
940
|
+
result = [0]
|
|
941
|
+
|
|
942
|
+
for k in ks:
|
|
943
|
+
if k == 1: continue
|
|
944
|
+
model = KMeans(n_clusters = k, random_state = 0)
|
|
945
|
+
model.fit(X = X)
|
|
946
|
+
cluster = model.predict(X = X)
|
|
947
|
+
silwidth = metrics.silhouette_score(X = X, labels = cluster)
|
|
948
|
+
result.append(silwidth)
|
|
949
|
+
|
|
950
|
+
sns.lineplot(
|
|
951
|
+
x = ks,
|
|
952
|
+
y = result,
|
|
953
|
+
marker = 'o',
|
|
954
|
+
linestyle = '-',
|
|
955
|
+
linewidth = 1
|
|
956
|
+
)
|
|
957
|
+
|
|
958
|
+
plt.xticks(ticks = ks)
|
|
959
|
+
plt.title(label = 'Silhouette Width', fontdict = {'fontweight': 'bold'})
|
|
960
|
+
plt.xlabel(xlabel = 'Number of clusters')
|
|
961
|
+
plt.ylabel(ylabel = 'Silhouette Width Average');
|
|
962
|
+
|
|
963
|
+
|
|
964
|
+
## End of Document
|
hds/stat.py
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
# 관련 라이브러리 호출
|
|
2
|
+
import pandas as pd
|
|
3
|
+
import numpy as np
|
|
4
|
+
import seaborn as sns
|
|
5
|
+
import matplotlib.pyplot as plt
|
|
6
|
+
from scipy import stats
|
|
7
|
+
from sklearn import metrics
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
# 회귀 모델의 성능 지표 반환 함수
|
|
11
|
+
def regmetrics(y_true: np.ndarray, y_pred: np.ndarray) -> pd.DataFrame:
|
|
12
|
+
'''
|
|
13
|
+
이 함수는 회귀 모델의 다양한 성능 지표를 계산합니다.
|
|
14
|
+
|
|
15
|
+
매개변수:
|
|
16
|
+
y_true: 목표변수의 실제값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
17
|
+
y_pred: 목표변수의 추정값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
18
|
+
|
|
19
|
+
반환:
|
|
20
|
+
회귀 모델의 다양한 성능 지표를 데이터프레임으로 반환합니다.
|
|
21
|
+
실제값과 추정값이 음수일 때 RMSLE는 결측값으로 채웁니다.
|
|
22
|
+
'''
|
|
23
|
+
MSE = metrics.mean_squared_error(
|
|
24
|
+
y_true = y_true,
|
|
25
|
+
y_pred = y_pred
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
RMSE = metrics.root_mean_squared_error(
|
|
29
|
+
y_true = y_true,
|
|
30
|
+
y_pred = y_pred
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
minus_count = pd.Series(data = y_pred).lt(0).sum()
|
|
34
|
+
|
|
35
|
+
if minus_count > 0:
|
|
36
|
+
MSLE = None
|
|
37
|
+
RMSLE = None
|
|
38
|
+
else:
|
|
39
|
+
MSLE = metrics.mean_squared_log_error(
|
|
40
|
+
y_true = y_true,
|
|
41
|
+
y_pred = y_pred
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
RMSLE = metrics.root_mean_squared_log_error(
|
|
45
|
+
y_true = y_true,
|
|
46
|
+
y_pred = y_pred
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
MAE = metrics.mean_absolute_error(
|
|
50
|
+
y_true = y_true,
|
|
51
|
+
y_pred = y_pred
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
MAPE = metrics.mean_absolute_percentage_error(
|
|
55
|
+
y_true = y_true,
|
|
56
|
+
y_pred = y_pred
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
result = pd.DataFrame(
|
|
60
|
+
data = [MSE, RMSE, MSLE, RMSLE, MAE, MAPE],
|
|
61
|
+
index = ['MSE', 'RMSE', 'MSLE', 'RMSLE', 'MAE', 'MAPE']
|
|
62
|
+
).T
|
|
63
|
+
|
|
64
|
+
return result
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
# 분류 모델의 성능 지표 반환 함수
|
|
68
|
+
def clfmetrics(y_true: np.ndarray, y_pred: np.ndarray) -> None:
|
|
69
|
+
'''
|
|
70
|
+
이 함수는 분류 모델의 다양한 성능 지표를 계산합니다.
|
|
71
|
+
|
|
72
|
+
매개변수:
|
|
73
|
+
y_true: 목표변수의 실제값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
74
|
+
y_pred: 목표변수의 추정값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
75
|
+
|
|
76
|
+
반환:
|
|
77
|
+
분류 모델의 다양한 성능 지표를 출력합니다.
|
|
78
|
+
'''
|
|
79
|
+
print('▶ Confusion Matrix')
|
|
80
|
+
|
|
81
|
+
cfm = pd.crosstab(
|
|
82
|
+
index = y_pred,
|
|
83
|
+
columns = y_true,
|
|
84
|
+
margins = True
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
cfm.index.name = 'Pred'
|
|
88
|
+
cfm.columns.name = 'Real'
|
|
89
|
+
display(cfm)
|
|
90
|
+
|
|
91
|
+
print()
|
|
92
|
+
print('▶ Classification Report')
|
|
93
|
+
print(
|
|
94
|
+
metrics.classification_report(
|
|
95
|
+
y_true = y_true,
|
|
96
|
+
y_pred = y_pred,
|
|
97
|
+
digits = 4
|
|
98
|
+
)
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
# 분류 모델의 분류 기준점별 성능 지표 계산(TPR, FPR, Matthew's Correlation coefficient)
|
|
103
|
+
def clfCutoffs(y_true: np.ndarray, y_pred: np.ndarray) -> pd.DataFrame:
|
|
104
|
+
'''
|
|
105
|
+
이 함수는 분류 모델에 대한 최적의 분류 기준점을 탐색합니다.
|
|
106
|
+
|
|
107
|
+
매개변수:
|
|
108
|
+
y_true: 목표변수의 실제값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
109
|
+
y_pred: 목표변수의 추정값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
110
|
+
|
|
111
|
+
반환:
|
|
112
|
+
분류 모델의 분류 기준점별로 TPR, FPR, MCC 등을 반환합니다.
|
|
113
|
+
'''
|
|
114
|
+
cutoffs = np.linspace(0, 1, 101)
|
|
115
|
+
sens = []
|
|
116
|
+
spec = []
|
|
117
|
+
prec = []
|
|
118
|
+
mccs = []
|
|
119
|
+
|
|
120
|
+
for cutoff in cutoffs:
|
|
121
|
+
pred = np.where(y_prob >= cutoff, 1, 0)
|
|
122
|
+
clfr = metrics.classification_report(
|
|
123
|
+
y_true = y_true,
|
|
124
|
+
y_pred = pred,
|
|
125
|
+
output_dict = True,
|
|
126
|
+
zero_division = True
|
|
127
|
+
)
|
|
128
|
+
sens.append(clfr['1']['recall'])
|
|
129
|
+
spec.append(clfr['0']['recall'])
|
|
130
|
+
prec.append(clfr['1']['precision'])
|
|
131
|
+
|
|
132
|
+
mcc = metrics.matthews_corrcoef(
|
|
133
|
+
y_true = y_true,
|
|
134
|
+
y_pred = pred
|
|
135
|
+
)
|
|
136
|
+
mccs.append(mcc)
|
|
137
|
+
|
|
138
|
+
result = pd.DataFrame(
|
|
139
|
+
data = {
|
|
140
|
+
'Cutoff': cutoffs,
|
|
141
|
+
'Sensitivity': sens,
|
|
142
|
+
'Specificity': spec,
|
|
143
|
+
'Precision': prec,
|
|
144
|
+
'MCC': mccs
|
|
145
|
+
}
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
# The Optimal Point is the sum of Sensitivity and Specificity.
|
|
149
|
+
result['Optimal'] = result['Sensitivity'] + result['Specificity']
|
|
150
|
+
|
|
151
|
+
# TPR and FPR for ROC Curve.
|
|
152
|
+
result['TPR'] = result['Sensitivity']
|
|
153
|
+
result['FPR'] = 1 - result['Specificity']
|
|
154
|
+
|
|
155
|
+
# Set Column name.
|
|
156
|
+
cols = ['Cutoff', 'Sensitivity', 'Specificity', 'Optimal', \
|
|
157
|
+
'Precision', 'TPR', 'FPR', 'MCC']
|
|
158
|
+
|
|
159
|
+
# Select columns
|
|
160
|
+
result = result[cols]
|
|
161
|
+
|
|
162
|
+
return result
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
# 최적의 분류 기준점 시각화 함수
|
|
166
|
+
def EpiROC(y_true: np.ndarray, y_pred: np.ndarray) -> None:
|
|
167
|
+
'''
|
|
168
|
+
이 함수는 분류 모델에 대한 최적의 분류 기준점을 ROC 곡선에 추가합니다.
|
|
169
|
+
|
|
170
|
+
매개변수:
|
|
171
|
+
y_true: 목표변수의 실제값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
172
|
+
y_pred: 목표변수의 추정값을 pd.Series 또는 1차원 np.ndarray로 지정합니다.
|
|
173
|
+
|
|
174
|
+
반환:
|
|
175
|
+
ROC 곡선 그래프 외에 반환하는 객체는 없습니다.
|
|
176
|
+
'''
|
|
177
|
+
obj = clfCutoffs(y_true, y_prob)
|
|
178
|
+
|
|
179
|
+
# Draw ROC curve
|
|
180
|
+
sns.lineplot(
|
|
181
|
+
data = obj,
|
|
182
|
+
x = 'FPR',
|
|
183
|
+
y = 'TPR',
|
|
184
|
+
color = 'black'
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
# Add title
|
|
188
|
+
plt.title(label = '최적의 분류 기준점 탐색',
|
|
189
|
+
fontdict = {'fontweight': 'bold'})
|
|
190
|
+
|
|
191
|
+
# Draw diagonal line
|
|
192
|
+
plt.plot(
|
|
193
|
+
[0, 1],
|
|
194
|
+
[0, 1],
|
|
195
|
+
color = '0.5',
|
|
196
|
+
linestyle = '--',
|
|
197
|
+
linewidth = 0.5
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
# Add the Optimal Point
|
|
201
|
+
opt = obj.iloc[[obj['Optimal'].argmax()]]
|
|
202
|
+
|
|
203
|
+
sns.scatterplot(
|
|
204
|
+
data = opt,
|
|
205
|
+
x = 'FPR',
|
|
206
|
+
y = 'TPR',
|
|
207
|
+
color = 'red'
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
# Add tangent line
|
|
211
|
+
optX = opt['FPR'].iloc[0]
|
|
212
|
+
optY = opt['TPR'].iloc[0]
|
|
213
|
+
|
|
214
|
+
b = optY - optX
|
|
215
|
+
|
|
216
|
+
plt.plot(
|
|
217
|
+
[0, 1-b],
|
|
218
|
+
[b, 1],
|
|
219
|
+
color = 'red',
|
|
220
|
+
linestyle = '-.',
|
|
221
|
+
linewidth = 0.5
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
# Add text
|
|
225
|
+
plt.text(
|
|
226
|
+
x = opt['FPR'].values[0] - 0.01,
|
|
227
|
+
y = opt['TPR'].values[0] + 0.01,
|
|
228
|
+
s = f"Cutoff = {opt['Cutoff'].round(2).values[0]}",
|
|
229
|
+
ha = 'right',
|
|
230
|
+
va = 'bottom'
|
|
231
|
+
);
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
## End of Document
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hds
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Functions for EDA, Statistics and Machine Learning
|
|
5
|
+
Home-page: https://github.com/HelloDataScience/hds
|
|
6
|
+
Author: HelloDataScience
|
|
7
|
+
Author-email: hellodatasciencekorea@gmail.com
|
|
8
|
+
License: MIT
|
|
9
|
+
Project-URL: Bug Tracker, https://github.com/HelloDataScience/hds/issues
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Operating System :: OS Independent
|
|
13
|
+
Requires-Python: >=3.11
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
License-File: LICENSE
|
|
16
|
+
Requires-Dist: numpy
|
|
17
|
+
Requires-Dist: pandas
|
|
18
|
+
Requires-Dist: scipy
|
|
19
|
+
Requires-Dist: seaborn
|
|
20
|
+
Requires-Dist: matplotlib
|
|
21
|
+
Requires-Dist: statsmodels
|
|
22
|
+
Requires-Dist: scikit-learn
|
|
23
|
+
Requires-Dist: graphviz
|
|
24
|
+
Requires-Dist: requests
|
|
25
|
+
Requires-Dist: bs4
|
|
26
|
+
Dynamic: author
|
|
27
|
+
Dynamic: author-email
|
|
28
|
+
Dynamic: classifier
|
|
29
|
+
Dynamic: description
|
|
30
|
+
Dynamic: description-content-type
|
|
31
|
+
Dynamic: home-page
|
|
32
|
+
Dynamic: license
|
|
33
|
+
Dynamic: license-file
|
|
34
|
+
Dynamic: project-url
|
|
35
|
+
Dynamic: requires-dist
|
|
36
|
+
Dynamic: requires-python
|
|
37
|
+
Dynamic: summary
|
|
38
|
+
|
|
39
|
+
# hds
|
|
40
|
+
Functions for EDA, Statistics and Machine Learning
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
hds/__init__.py,sha256=h3_7wcJaYa3pQ08kHrH-AEFEvspzRhu3cyCWO1Gx7wk,69
|
|
2
|
+
hds/plot.py,sha256=aQttCmBujnSqEMrEaZ4mOk0Mc4cvG1kODLjU0gOyNwo,31394
|
|
3
|
+
hds/stat.py,sha256=LaMdVo9tb9O6RPdQSRvubsTl9_wT61kmo6ui3NhlRyM,6376
|
|
4
|
+
hds-0.1.1.dist-info/licenses/LICENSE,sha256=aK7Q5YQ7GY5IWIFdyPs3DaO16O8sC2pyz21hkvgC7to,1073
|
|
5
|
+
hds-0.1.1.dist-info/METADATA,sha256=Vk_xj_qMlJtQVPRP0IAlgTG87Pqb5F9saoNTc_VGY_I,1097
|
|
6
|
+
hds-0.1.1.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
7
|
+
hds-0.1.1.dist-info/top_level.txt,sha256=isOCYkUWALVxGgvOWuhpwXqDbd-5ds7UrfdZQwKgSbs,14
|
|
8
|
+
hds-0.1.1.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2023 HelloDataScience
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|