aiwf 0.3.22 → 0.3.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +3 -2
- package/src/cli/index.js +39 -336
- package/src/lib/installer.js +294 -90
- package/src/utils/messages.js +24 -0
- package/src/DEPENDENCY_MAP.md +0 -90
- package/src/cli/cache-cli.js +0 -459
- package/src/cli/checkpoint-cli.js +0 -417
- package/src/cli/language-cli.js +0 -286
- package/src/cli/sprint-cli.js +0 -287
- package/src/commands/ai-tool.js +0 -383
- package/src/commands/compress.js +0 -60
- package/src/commands/create-project.js +0 -160
- package/src/commands/evaluate.js +0 -459
- package/src/commands/persona.js +0 -309
- package/src/commands/sprint-task.js +0 -262
- package/src/commands/token.js +0 -312
- package/src/lib/ai-persona-manager.js +0 -711
- package/src/lib/cache-system.js +0 -332
- package/src/lib/context-engine.js +0 -568
- package/src/lib/github-integration.js +0 -402
- package/src/lib/memory-profiler.js +0 -471
- package/src/lib/metrics-collector.js +0 -885
- package/src/lib/offline-detector.js +0 -386
- package/src/lib/resource-loader-enhanced.js +0 -399
- package/src/lib/resource-loader.js +0 -250
- package/src/lib/resources/commands/ai-persona.js +0 -602
- package/src/lib/resources/commands/compress-context.js +0 -389
- package/src/lib/resources/commands/evaluate.js +0 -246
- package/src/lib/resources/commands/persona-context-apply.js +0 -562
- package/src/lib/resources/commands/token-tracking.js +0 -443
- package/src/lib/resources/config/commit-patterns.js +0 -103
- package/src/lib/resources/config/language.json +0 -6
- package/src/lib/resources/personas/PERSONA_INDEX.md +0 -57
- package/src/lib/resources/personas/analyst.json +0 -44
- package/src/lib/resources/personas/architect/best_practices.md +0 -136
- package/src/lib/resources/personas/architect/knowledge_base.md +0 -377
- package/src/lib/resources/personas/architect.json +0 -44
- package/src/lib/resources/personas/backend/best_practices.md +0 -766
- package/src/lib/resources/personas/backend/knowledge_base.md +0 -1070
- package/src/lib/resources/personas/data_analyst/best_practices.md +0 -594
- package/src/lib/resources/personas/data_analyst/knowledge_base.md +0 -1057
- package/src/lib/resources/personas/developer.json +0 -44
- package/src/lib/resources/personas/developer.md +0 -37
- package/src/lib/resources/personas/evaluation_criteria.json +0 -137
- package/src/lib/resources/personas/frontend/best_practices.md +0 -445
- package/src/lib/resources/personas/frontend/knowledge_base.md +0 -729
- package/src/lib/resources/personas/persona-index.json +0 -31
- package/src/lib/resources/personas/reviewer.json +0 -44
- package/src/lib/resources/personas/security/best_practices.md +0 -307
- package/src/lib/resources/personas/security/knowledge_base.md +0 -498
- package/src/lib/resources/personas/tester.json +0 -44
- package/src/lib/resources/templates/README.md +0 -78
- package/src/lib/resources/templates/api-server/config.json +0 -64
- package/src/lib/resources/templates/api-server/template/.aiwf/config.json +0 -56
- package/src/lib/resources/templates/api-server/template/.aiwf/feature-ledger.json +0 -42
- package/src/lib/resources/templates/api-server/template/.aiwf/personas/backend-engineer.json +0 -40
- package/src/lib/resources/templates/api-server/template/.aiwf/scripts/cli.js +0 -111
- package/src/lib/resources/templates/api-server/template/.env.example +0 -29
- package/src/lib/resources/templates/api-server/template/.eslintrc.json +0 -23
- package/src/lib/resources/templates/api-server/template/README.md +0 -171
- package/src/lib/resources/templates/api-server/template/jest.config.js +0 -26
- package/src/lib/resources/templates/api-server/template/nodemon.json +0 -9
- package/src/lib/resources/templates/api-server/template/package.json +0 -61
- package/src/lib/resources/templates/api-server/template/src/app.ts +0 -60
- package/src/lib/resources/templates/api-server/template/src/config/swagger.ts +0 -38
- package/src/lib/resources/templates/api-server/template/src/controllers/aiwfController.ts +0 -54
- package/src/lib/resources/templates/api-server/template/src/controllers/authController.ts +0 -128
- package/src/lib/resources/templates/api-server/template/src/controllers/statusController.ts +0 -28
- package/src/lib/resources/templates/api-server/template/src/index.ts +0 -28
- package/src/lib/resources/templates/api-server/template/src/middleware/aiwfMiddleware.ts +0 -50
- package/src/lib/resources/templates/api-server/template/src/middleware/authMiddleware.ts +0 -41
- package/src/lib/resources/templates/api-server/template/src/middleware/errorHandler.ts +0 -42
- package/src/lib/resources/templates/api-server/template/src/middleware/notFoundHandler.ts +0 -14
- package/src/lib/resources/templates/api-server/template/src/middleware/requestLogger.ts +0 -20
- package/src/lib/resources/templates/api-server/template/src/routes/aiwf.ts +0 -31
- package/src/lib/resources/templates/api-server/template/src/routes/index.ts +0 -13
- package/src/lib/resources/templates/api-server/template/src/routes/v1/index.ts +0 -67
- package/src/lib/resources/templates/api-server/template/src/utils/logger.ts +0 -37
- package/src/lib/resources/templates/api-server/template/tests/app.test.ts +0 -40
- package/src/lib/resources/templates/api-server/template/tsconfig.json +0 -43
- package/src/lib/resources/templates/npm-library/config.json +0 -61
- package/src/lib/resources/templates/npm-library/template/LICENSE +0 -21
- package/src/lib/resources/templates/npm-library/template/README.md +0 -201
- package/src/lib/resources/templates/npm-library/template/package.json +0 -78
- package/src/lib/resources/templates/web-app/config.json +0 -53
- package/src/lib/resources/templates/web-app/template/.aiwf/config.json +0 -54
- package/src/lib/resources/templates/web-app/template/.aiwf/feature-ledger.json +0 -37
- package/src/lib/resources/templates/web-app/template/.aiwf/personas/fullstack-developer.json +0 -40
- package/src/lib/resources/templates/web-app/template/.aiwf/scripts/cli.js +0 -110
- package/src/lib/resources/templates/web-app/template/.eslintrc.cjs +0 -20
- package/src/lib/resources/templates/web-app/template/README.md +0 -151
- package/src/lib/resources/templates/web-app/template/index.html +0 -14
- package/src/lib/resources/templates/web-app/template/package.json +0 -44
- package/src/lib/resources/templates/web-app/template/postcss.config.js +0 -6
- package/src/lib/resources/templates/web-app/template/public/vite.svg +0 -1
- package/src/lib/resources/templates/web-app/template/src/App.tsx +0 -21
- package/src/lib/resources/templates/web-app/template/src/components/Layout.tsx +0 -57
- package/src/lib/resources/templates/web-app/template/src/components/aiwf/ContextStatus.tsx +0 -107
- package/src/lib/resources/templates/web-app/template/src/components/aiwf/TokenUsage.tsx +0 -71
- package/src/lib/resources/templates/web-app/template/src/index.css +0 -60
- package/src/lib/resources/templates/web-app/template/src/main.tsx +0 -10
- package/src/lib/resources/templates/web-app/template/src/pages/AiwfDashboard.tsx +0 -57
- package/src/lib/resources/templates/web-app/template/src/pages/HomePage.tsx +0 -89
- package/src/lib/resources/templates/web-app/template/src/pages/NotFound.tsx +0 -25
- package/src/lib/resources/templates/web-app/template/src/stores/aiwfStore.ts +0 -126
- package/src/lib/resources/templates/web-app/template/src/types/global.d.ts +0 -9
- package/src/lib/resources/templates/web-app/template/src/vite-env.d.ts +0 -1
- package/src/lib/resources/templates/web-app/template/tailwind.config.js +0 -30
- package/src/lib/resources/templates/web-app/template/tsconfig.json +0 -37
- package/src/lib/resources/templates/web-app/template/tsconfig.node.json +0 -10
- package/src/lib/resources/templates/web-app/template/vite.config.ts +0 -23
- package/src/lib/resources/utils/background-monitor.js +0 -223
- package/src/lib/resources/utils/compression-metrics.js +0 -1031
- package/src/lib/resources/utils/compression-strategies.js +0 -557
- package/src/lib/resources/utils/content-normalizer.js +0 -455
- package/src/lib/resources/utils/context-compressor.js +0 -583
- package/src/lib/resources/utils/context-rule-parser.js +0 -180
- package/src/lib/resources/utils/context-token-monitor.js +0 -463
- package/src/lib/resources/utils/context-update-manager.js +0 -502
- package/src/lib/resources/utils/git-utils.js +0 -220
- package/src/lib/resources/utils/importance-classifier.js +0 -645
- package/src/lib/resources/utils/information-filter.js +0 -710
- package/src/lib/resources/utils/persona-aware-compressor.js +0 -598
- package/src/lib/resources/utils/prompt-injector.js +0 -242
- package/src/lib/resources/utils/simplified-evaluator.js +0 -119
- package/src/lib/resources/utils/text-summarizer.js +0 -492
- package/src/lib/resources/utils/token-counter.js +0 -177
- package/src/lib/resources/utils/token-monitor.js +0 -484
- package/src/lib/resources/utils/token-reporter.js +0 -452
- package/src/lib/resources/utils/token-storage.js +0 -433
- package/src/lib/resources/utils/token-tracker.js +0 -283
- package/src/lib/state/priority-calculator.js +0 -217
- package/src/lib/state/state-index.js +0 -155
- package/src/lib/state/task-scanner.js +0 -328
- package/src/lib/task-analyzer.js +0 -599
- package/src/lib/template-cache-system.js +0 -559
- package/src/lib/template-downloader.js +0 -467
- package/src/lib/template-version-manager.js +0 -491
- package/src/lib/token-optimizer.js +0 -517
- package/src/utils/engineering-guard.js +0 -399
|
@@ -1,1057 +0,0 @@
|
|
|
1
|
-
# Data Analyst 페르소나 Knowledge Base
|
|
2
|
-
|
|
3
|
-
## 데이터 분석 도구 및 라이브러리
|
|
4
|
-
|
|
5
|
-
### Python 데이터 분석 스택
|
|
6
|
-
|
|
7
|
-
#### Pandas 고급 기능
|
|
8
|
-
```python
|
|
9
|
-
import pandas as pd
|
|
10
|
-
import numpy as np
|
|
11
|
-
|
|
12
|
-
# 1. 멀티 인덱싱
|
|
13
|
-
df = pd.DataFrame({
|
|
14
|
-
'date': pd.date_range('2024-01-01', periods=100),
|
|
15
|
-
'category': np.random.choice(['A', 'B', 'C'], 100),
|
|
16
|
-
'subcategory': np.random.choice(['X', 'Y'], 100),
|
|
17
|
-
'value': np.random.randn(100),
|
|
18
|
-
'quantity': np.random.randint(1, 100, 100)
|
|
19
|
-
})
|
|
20
|
-
|
|
21
|
-
# 멀티 인덱스 생성
|
|
22
|
-
df_multi = df.set_index(['category', 'subcategory'])
|
|
23
|
-
|
|
24
|
-
# 멀티 인덱스 접근
|
|
25
|
-
df_multi.loc[('A', 'X')] # 특정 조합
|
|
26
|
-
df_multi.xs('X', level='subcategory') # 특정 레벨
|
|
27
|
-
|
|
28
|
-
# 2. Window Functions
|
|
29
|
-
# 이동 평균
|
|
30
|
-
df['ma_7'] = df.groupby('category')['value'].transform(
|
|
31
|
-
lambda x: x.rolling(7, min_periods=1).mean()
|
|
32
|
-
)
|
|
33
|
-
|
|
34
|
-
# 누적 합계
|
|
35
|
-
df['cumsum'] = df.groupby('category')['value'].cumsum()
|
|
36
|
-
|
|
37
|
-
# Expanding window
|
|
38
|
-
df['expanding_mean'] = df.groupby('category')['value'].expanding().mean()
|
|
39
|
-
|
|
40
|
-
# 3. 피벗 테이블 고급 기능
|
|
41
|
-
pivot_table = pd.pivot_table(
|
|
42
|
-
df,
|
|
43
|
-
values=['value', 'quantity'],
|
|
44
|
-
index=['date'],
|
|
45
|
-
columns=['category'],
|
|
46
|
-
aggfunc={
|
|
47
|
-
'value': ['mean', 'std'],
|
|
48
|
-
'quantity': 'sum'
|
|
49
|
-
},
|
|
50
|
-
fill_value=0,
|
|
51
|
-
margins=True,
|
|
52
|
-
margins_name='Total'
|
|
53
|
-
)
|
|
54
|
-
|
|
55
|
-
# 4. 시계열 리샘플링
|
|
56
|
-
df_ts = df.set_index('date')
|
|
57
|
-
monthly_summary = df_ts.resample('M').agg({
|
|
58
|
-
'value': ['sum', 'mean', 'std'],
|
|
59
|
-
'quantity': 'sum'
|
|
60
|
-
})
|
|
61
|
-
|
|
62
|
-
# 5. 메모리 최적화
|
|
63
|
-
def optimize_dataframe(df):
|
|
64
|
-
"""데이터프레임 메모리 사용량 최적화"""
|
|
65
|
-
for col in df.columns:
|
|
66
|
-
col_type = df[col].dtype
|
|
67
|
-
|
|
68
|
-
if col_type != 'object':
|
|
69
|
-
c_min = df[col].min()
|
|
70
|
-
c_max = df[col].max()
|
|
71
|
-
|
|
72
|
-
if str(col_type)[:3] == 'int':
|
|
73
|
-
if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:
|
|
74
|
-
df[col] = df[col].astype(np.int8)
|
|
75
|
-
elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:
|
|
76
|
-
df[col] = df[col].astype(np.int16)
|
|
77
|
-
elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:
|
|
78
|
-
df[col] = df[col].astype(np.int32)
|
|
79
|
-
else:
|
|
80
|
-
if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:
|
|
81
|
-
df[col] = df[col].astype(np.float16)
|
|
82
|
-
elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:
|
|
83
|
-
df[col] = df[col].astype(np.float32)
|
|
84
|
-
else:
|
|
85
|
-
df[col] = df[col].astype('category')
|
|
86
|
-
|
|
87
|
-
return df
|
|
88
|
-
```
|
|
89
|
-
|
|
90
|
-
#### NumPy 고급 연산
|
|
91
|
-
```python
|
|
92
|
-
# 1. Broadcasting
|
|
93
|
-
a = np.array([[1, 2, 3], [4, 5, 6]])
|
|
94
|
-
b = np.array([10, 20, 30])
|
|
95
|
-
result = a + b # Broadcasting 적용
|
|
96
|
-
|
|
97
|
-
# 2. 벡터화 연산
|
|
98
|
-
@np.vectorize
|
|
99
|
-
def custom_function(x):
|
|
100
|
-
if x > 0:
|
|
101
|
-
return np.log(x)
|
|
102
|
-
else:
|
|
103
|
-
return 0
|
|
104
|
-
|
|
105
|
-
vectorized_result = custom_function(np.array([-1, 0, 1, 2, 3]))
|
|
106
|
-
|
|
107
|
-
# 3. 고급 인덱싱
|
|
108
|
-
arr = np.random.randn(10, 10)
|
|
109
|
-
mask = arr > 0
|
|
110
|
-
positive_values = arr[mask]
|
|
111
|
-
|
|
112
|
-
# Fancy indexing
|
|
113
|
-
rows = np.array([0, 2, 4])
|
|
114
|
-
cols = np.array([1, 3, 5])
|
|
115
|
-
selected = arr[rows[:, np.newaxis], cols]
|
|
116
|
-
|
|
117
|
-
# 4. 선형대수 연산
|
|
118
|
-
# 고유값 분해
|
|
119
|
-
eigenvalues, eigenvectors = np.linalg.eig(arr)
|
|
120
|
-
|
|
121
|
-
# 특이값 분해 (SVD)
|
|
122
|
-
U, s, Vt = np.linalg.svd(arr)
|
|
123
|
-
|
|
124
|
-
# 5. 통계 함수
|
|
125
|
-
# 백분위수
|
|
126
|
-
percentiles = np.percentile(arr, [25, 50, 75], axis=0)
|
|
127
|
-
|
|
128
|
-
# 상관계수 행렬
|
|
129
|
-
correlation_matrix = np.corrcoef(arr.T)
|
|
130
|
-
```
|
|
131
|
-
|
|
132
|
-
### SQL 고급 쿼리
|
|
133
|
-
|
|
134
|
-
#### 윈도우 함수
|
|
135
|
-
```sql
|
|
136
|
-
-- 1. 순위 함수
|
|
137
|
-
WITH sales_ranking AS (
|
|
138
|
-
SELECT
|
|
139
|
-
product_id,
|
|
140
|
-
sale_date,
|
|
141
|
-
amount,
|
|
142
|
-
ROW_NUMBER() OVER (PARTITION BY product_id ORDER BY amount DESC) as row_num,
|
|
143
|
-
RANK() OVER (PARTITION BY product_id ORDER BY amount DESC) as rank,
|
|
144
|
-
DENSE_RANK() OVER (PARTITION BY product_id ORDER BY amount DESC) as dense_rank,
|
|
145
|
-
PERCENT_RANK() OVER (PARTITION BY product_id ORDER BY amount DESC) as percent_rank
|
|
146
|
-
FROM sales
|
|
147
|
-
)
|
|
148
|
-
SELECT * FROM sales_ranking WHERE row_num <= 10;
|
|
149
|
-
|
|
150
|
-
-- 2. 이동 평균 및 누적 합계
|
|
151
|
-
SELECT
|
|
152
|
-
date,
|
|
153
|
-
revenue,
|
|
154
|
-
-- 7일 이동 평균
|
|
155
|
-
AVG(revenue) OVER (
|
|
156
|
-
ORDER BY date
|
|
157
|
-
ROWS BETWEEN 6 PRECEDING AND CURRENT ROW
|
|
158
|
-
) as ma_7,
|
|
159
|
-
-- 누적 합계
|
|
160
|
-
SUM(revenue) OVER (
|
|
161
|
-
ORDER BY date
|
|
162
|
-
ROWS BETWEEN UNBOUNDED PRECEDING AND CURRENT ROW
|
|
163
|
-
) as cumulative_revenue,
|
|
164
|
-
-- 전월 대비 성장률
|
|
165
|
-
LAG(revenue, 1) OVER (ORDER BY date) as prev_revenue,
|
|
166
|
-
(revenue - LAG(revenue, 1) OVER (ORDER BY date)) /
|
|
167
|
-
NULLIF(LAG(revenue, 1) OVER (ORDER BY date), 0) * 100 as growth_rate
|
|
168
|
-
FROM daily_revenue;
|
|
169
|
-
|
|
170
|
-
-- 3. 고급 집계
|
|
171
|
-
WITH monthly_stats AS (
|
|
172
|
-
SELECT
|
|
173
|
-
DATE_TRUNC('month', order_date) as month,
|
|
174
|
-
category,
|
|
175
|
-
COUNT(DISTINCT customer_id) as unique_customers,
|
|
176
|
-
COUNT(*) as total_orders,
|
|
177
|
-
SUM(amount) as total_revenue,
|
|
178
|
-
AVG(amount) as avg_order_value,
|
|
179
|
-
STDDEV(amount) as stddev_order_value
|
|
180
|
-
FROM orders
|
|
181
|
-
GROUP BY DATE_TRUNC('month', order_date), category
|
|
182
|
-
),
|
|
183
|
-
category_percentiles AS (
|
|
184
|
-
SELECT
|
|
185
|
-
month,
|
|
186
|
-
category,
|
|
187
|
-
total_revenue,
|
|
188
|
-
PERCENTILE_CONT(0.25) WITHIN GROUP (ORDER BY total_revenue)
|
|
189
|
-
OVER (PARTITION BY month) as q1,
|
|
190
|
-
PERCENTILE_CONT(0.50) WITHIN GROUP (ORDER BY total_revenue)
|
|
191
|
-
OVER (PARTITION BY month) as median,
|
|
192
|
-
PERCENTILE_CONT(0.75) WITHIN GROUP (ORDER BY total_revenue)
|
|
193
|
-
OVER (PARTITION BY month) as q3
|
|
194
|
-
FROM monthly_stats
|
|
195
|
-
)
|
|
196
|
-
SELECT * FROM category_percentiles;
|
|
197
|
-
|
|
198
|
-
-- 4. 재귀 CTE (계층 구조)
|
|
199
|
-
WITH RECURSIVE category_hierarchy AS (
|
|
200
|
-
-- Base case
|
|
201
|
-
SELECT
|
|
202
|
-
id,
|
|
203
|
-
name,
|
|
204
|
-
parent_id,
|
|
205
|
-
0 as level,
|
|
206
|
-
name as path
|
|
207
|
-
FROM categories
|
|
208
|
-
WHERE parent_id IS NULL
|
|
209
|
-
|
|
210
|
-
UNION ALL
|
|
211
|
-
|
|
212
|
-
-- Recursive case
|
|
213
|
-
SELECT
|
|
214
|
-
c.id,
|
|
215
|
-
c.name,
|
|
216
|
-
c.parent_id,
|
|
217
|
-
ch.level + 1,
|
|
218
|
-
ch.path || ' > ' || c.name
|
|
219
|
-
FROM categories c
|
|
220
|
-
JOIN category_hierarchy ch ON c.parent_id = ch.id
|
|
221
|
-
)
|
|
222
|
-
SELECT * FROM category_hierarchy ORDER BY path;
|
|
223
|
-
```
|
|
224
|
-
|
|
225
|
-
### 통계 분석 기법
|
|
226
|
-
|
|
227
|
-
#### 가설 검정
|
|
228
|
-
```python
|
|
229
|
-
from scipy import stats
|
|
230
|
-
import statsmodels.api as sm
|
|
231
|
-
from statsmodels.stats.multicomp import pairwise_tukeyhsd
|
|
232
|
-
|
|
233
|
-
class StatisticalTesting:
|
|
234
|
-
@staticmethod
|
|
235
|
-
def normality_tests(data):
|
|
236
|
-
"""정규성 검정"""
|
|
237
|
-
results = {}
|
|
238
|
-
|
|
239
|
-
# Shapiro-Wilk test
|
|
240
|
-
stat, p_value = stats.shapiro(data)
|
|
241
|
-
results['shapiro'] = {
|
|
242
|
-
'statistic': stat,
|
|
243
|
-
'p_value': p_value,
|
|
244
|
-
'is_normal': p_value > 0.05
|
|
245
|
-
}
|
|
246
|
-
|
|
247
|
-
# Kolmogorov-Smirnov test
|
|
248
|
-
stat, p_value = stats.kstest(data, 'norm',
|
|
249
|
-
args=(data.mean(), data.std()))
|
|
250
|
-
results['ks_test'] = {
|
|
251
|
-
'statistic': stat,
|
|
252
|
-
'p_value': p_value,
|
|
253
|
-
'is_normal': p_value > 0.05
|
|
254
|
-
}
|
|
255
|
-
|
|
256
|
-
# Anderson-Darling test
|
|
257
|
-
result = stats.anderson(data)
|
|
258
|
-
results['anderson'] = {
|
|
259
|
-
'statistic': result.statistic,
|
|
260
|
-
'critical_values': result.critical_values,
|
|
261
|
-
'significance_levels': result.significance_level
|
|
262
|
-
}
|
|
263
|
-
|
|
264
|
-
return results
|
|
265
|
-
|
|
266
|
-
@staticmethod
|
|
267
|
-
def variance_tests(group1, group2):
|
|
268
|
-
"""등분산성 검정"""
|
|
269
|
-
# Levene's test
|
|
270
|
-
stat, p_value = stats.levene(group1, group2)
|
|
271
|
-
|
|
272
|
-
return {
|
|
273
|
-
'levene': {
|
|
274
|
-
'statistic': stat,
|
|
275
|
-
'p_value': p_value,
|
|
276
|
-
'equal_variance': p_value > 0.05
|
|
277
|
-
}
|
|
278
|
-
}
|
|
279
|
-
|
|
280
|
-
@staticmethod
|
|
281
|
-
def anova_analysis(groups, labels):
|
|
282
|
-
"""ANOVA 분석"""
|
|
283
|
-
# One-way ANOVA
|
|
284
|
-
f_stat, p_value = stats.f_oneway(*groups)
|
|
285
|
-
|
|
286
|
-
# Post-hoc test (Tukey HSD)
|
|
287
|
-
data_combined = np.concatenate(groups)
|
|
288
|
-
labels_combined = np.concatenate([
|
|
289
|
-
[label] * len(group)
|
|
290
|
-
for label, group in zip(labels, groups)
|
|
291
|
-
])
|
|
292
|
-
|
|
293
|
-
tukey_result = pairwise_tukeyhsd(
|
|
294
|
-
data_combined,
|
|
295
|
-
labels_combined,
|
|
296
|
-
alpha=0.05
|
|
297
|
-
)
|
|
298
|
-
|
|
299
|
-
return {
|
|
300
|
-
'anova': {
|
|
301
|
-
'f_statistic': f_stat,
|
|
302
|
-
'p_value': p_value,
|
|
303
|
-
'significant': p_value < 0.05
|
|
304
|
-
},
|
|
305
|
-
'post_hoc': str(tukey_result)
|
|
306
|
-
}
|
|
307
|
-
```
|
|
308
|
-
|
|
309
|
-
#### 회귀 분석
|
|
310
|
-
```python
|
|
311
|
-
class RegressionAnalysis:
|
|
312
|
-
def __init__(self, X, y):
|
|
313
|
-
self.X = X
|
|
314
|
-
self.y = y
|
|
315
|
-
self.model = None
|
|
316
|
-
self.results = None
|
|
317
|
-
|
|
318
|
-
def linear_regression(self):
|
|
319
|
-
"""선형 회귀 분석"""
|
|
320
|
-
# 상수항 추가
|
|
321
|
-
X_with_const = sm.add_constant(self.X)
|
|
322
|
-
|
|
323
|
-
# 모델 적합
|
|
324
|
-
self.model = sm.OLS(self.y, X_with_const)
|
|
325
|
-
self.results = self.model.fit()
|
|
326
|
-
|
|
327
|
-
# 결과 요약
|
|
328
|
-
summary = {
|
|
329
|
-
'r_squared': self.results.rsquared,
|
|
330
|
-
'adj_r_squared': self.results.rsquared_adj,
|
|
331
|
-
'f_statistic': self.results.fvalue,
|
|
332
|
-
'f_pvalue': self.results.f_pvalue,
|
|
333
|
-
'coefficients': dict(zip(
|
|
334
|
-
self.X.columns,
|
|
335
|
-
self.results.params[1:]
|
|
336
|
-
)),
|
|
337
|
-
'p_values': dict(zip(
|
|
338
|
-
self.X.columns,
|
|
339
|
-
self.results.pvalues[1:]
|
|
340
|
-
)),
|
|
341
|
-
'confidence_intervals': self.results.conf_int()[1:].values.tolist()
|
|
342
|
-
}
|
|
343
|
-
|
|
344
|
-
return summary
|
|
345
|
-
|
|
346
|
-
def check_assumptions(self):
|
|
347
|
-
"""회귀 가정 검증"""
|
|
348
|
-
residuals = self.results.resid
|
|
349
|
-
fitted = self.results.fittedvalues
|
|
350
|
-
|
|
351
|
-
# 1. 선형성 검정
|
|
352
|
-
linearity_test = stats.pearsonr(fitted, self.y)[1] < 0.05
|
|
353
|
-
|
|
354
|
-
# 2. 정규성 검정
|
|
355
|
-
_, normality_pvalue = stats.jarque_bera(residuals)
|
|
356
|
-
|
|
357
|
-
# 3. 등분산성 검정 (Breusch-Pagan test)
|
|
358
|
-
bp_test = sm.stats.diagnostic.het_breuschpagan(
|
|
359
|
-
residuals, self.results.model.exog
|
|
360
|
-
)
|
|
361
|
-
|
|
362
|
-
# 4. 자기상관 검정 (Durbin-Watson)
|
|
363
|
-
dw_stat = sm.stats.stattools.durbin_watson(residuals)
|
|
364
|
-
|
|
365
|
-
# 5. 다중공선성 (VIF)
|
|
366
|
-
from statsmodels.stats.outliers_influence import variance_inflation_factor
|
|
367
|
-
|
|
368
|
-
vif_data = pd.DataFrame()
|
|
369
|
-
vif_data["Variable"] = self.X.columns
|
|
370
|
-
vif_data["VIF"] = [
|
|
371
|
-
variance_inflation_factor(self.X.values, i)
|
|
372
|
-
for i in range(self.X.shape[1])
|
|
373
|
-
]
|
|
374
|
-
|
|
375
|
-
return {
|
|
376
|
-
'linearity': linearity_test,
|
|
377
|
-
'normality': normality_pvalue > 0.05,
|
|
378
|
-
'homoscedasticity': bp_test[1] > 0.05,
|
|
379
|
-
'autocorrelation': 1.5 < dw_stat < 2.5,
|
|
380
|
-
'multicollinearity': vif_data.to_dict()
|
|
381
|
-
}
|
|
382
|
-
```
|
|
383
|
-
|
|
384
|
-
### 시계열 분석
|
|
385
|
-
|
|
386
|
-
#### 시계열 분해 및 예측
|
|
387
|
-
```python
|
|
388
|
-
from statsmodels.tsa.seasonal import seasonal_decompose
|
|
389
|
-
from statsmodels.tsa.stattools import adfuller, kpss
|
|
390
|
-
from statsmodels.tsa.arima.model import ARIMA
|
|
391
|
-
from prophet import Prophet
|
|
392
|
-
|
|
393
|
-
class TimeSeriesAnalysis:
|
|
394
|
-
def __init__(self, ts_data, date_col, value_col):
|
|
395
|
-
self.data = ts_data.set_index(date_col)[value_col].sort_index()
|
|
396
|
-
self.decomposition = None
|
|
397
|
-
|
|
398
|
-
def test_stationarity(self):
|
|
399
|
-
"""정상성 검정"""
|
|
400
|
-
# ADF Test
|
|
401
|
-
adf_result = adfuller(self.data, autolag='AIC')
|
|
402
|
-
|
|
403
|
-
# KPSS Test
|
|
404
|
-
kpss_result = kpss(self.data, regression='c', nlags='auto')
|
|
405
|
-
|
|
406
|
-
return {
|
|
407
|
-
'adf': {
|
|
408
|
-
'statistic': adf_result[0],
|
|
409
|
-
'p_value': adf_result[1],
|
|
410
|
-
'is_stationary': adf_result[1] < 0.05
|
|
411
|
-
},
|
|
412
|
-
'kpss': {
|
|
413
|
-
'statistic': kpss_result[0],
|
|
414
|
-
'p_value': kpss_result[1],
|
|
415
|
-
'is_stationary': kpss_result[1] > 0.05
|
|
416
|
-
}
|
|
417
|
-
}
|
|
418
|
-
|
|
419
|
-
def decompose_series(self, model='additive', period=None):
|
|
420
|
-
"""시계열 분해"""
|
|
421
|
-
self.decomposition = seasonal_decompose(
|
|
422
|
-
self.data,
|
|
423
|
-
model=model,
|
|
424
|
-
period=period,
|
|
425
|
-
extrapolate_trend='freq'
|
|
426
|
-
)
|
|
427
|
-
|
|
428
|
-
return {
|
|
429
|
-
'trend': self.decomposition.trend,
|
|
430
|
-
'seasonal': self.decomposition.seasonal,
|
|
431
|
-
'residual': self.decomposition.resid
|
|
432
|
-
}
|
|
433
|
-
|
|
434
|
-
def fit_arima(self, order=(1,1,1), seasonal_order=(0,0,0,0)):
|
|
435
|
-
"""ARIMA 모델 적합"""
|
|
436
|
-
model = ARIMA(self.data, order=order, seasonal_order=seasonal_order)
|
|
437
|
-
fitted_model = model.fit()
|
|
438
|
-
|
|
439
|
-
# 예측
|
|
440
|
-
forecast = fitted_model.forecast(steps=30)
|
|
441
|
-
|
|
442
|
-
return {
|
|
443
|
-
'model': fitted_model,
|
|
444
|
-
'aic': fitted_model.aic,
|
|
445
|
-
'bic': fitted_model.bic,
|
|
446
|
-
'forecast': forecast,
|
|
447
|
-
'summary': fitted_model.summary()
|
|
448
|
-
}
|
|
449
|
-
|
|
450
|
-
def prophet_forecast(self, periods=30):
|
|
451
|
-
"""Prophet 예측"""
|
|
452
|
-
# 데이터 준비
|
|
453
|
-
prophet_data = pd.DataFrame({
|
|
454
|
-
'ds': self.data.index,
|
|
455
|
-
'y': self.data.values
|
|
456
|
-
})
|
|
457
|
-
|
|
458
|
-
# 모델 생성 및 학습
|
|
459
|
-
model = Prophet(
|
|
460
|
-
yearly_seasonality=True,
|
|
461
|
-
weekly_seasonality=True,
|
|
462
|
-
daily_seasonality=False,
|
|
463
|
-
changepoint_prior_scale=0.05
|
|
464
|
-
)
|
|
465
|
-
|
|
466
|
-
model.fit(prophet_data)
|
|
467
|
-
|
|
468
|
-
# 예측
|
|
469
|
-
future = model.make_future_dataframe(periods=periods)
|
|
470
|
-
forecast = model.predict(future)
|
|
471
|
-
|
|
472
|
-
return {
|
|
473
|
-
'model': model,
|
|
474
|
-
'forecast': forecast[['ds', 'yhat', 'yhat_lower', 'yhat_upper']],
|
|
475
|
-
'components': model.plot_components(forecast)
|
|
476
|
-
}
|
|
477
|
-
```
|
|
478
|
-
|
|
479
|
-
### 머신러닝 기법
|
|
480
|
-
|
|
481
|
-
#### 특성 공학
|
|
482
|
-
```python
|
|
483
|
-
from sklearn.preprocessing import PolynomialFeatures, StandardScaler
|
|
484
|
-
from sklearn.feature_selection import SelectKBest, f_classif, RFE
|
|
485
|
-
from sklearn.decomposition import PCA
|
|
486
|
-
|
|
487
|
-
class FeatureEngineering:
|
|
488
|
-
def __init__(self, X, y):
|
|
489
|
-
self.X = X
|
|
490
|
-
self.y = y
|
|
491
|
-
|
|
492
|
-
def create_polynomial_features(self, degree=2):
|
|
493
|
-
"""다항 특성 생성"""
|
|
494
|
-
poly = PolynomialFeatures(degree=degree, include_bias=False)
|
|
495
|
-
X_poly = poly.fit_transform(self.X)
|
|
496
|
-
|
|
497
|
-
feature_names = poly.get_feature_names_out(self.X.columns)
|
|
498
|
-
|
|
499
|
-
return pd.DataFrame(X_poly, columns=feature_names, index=self.X.index)
|
|
500
|
-
|
|
501
|
-
def create_interaction_features(self):
|
|
502
|
-
"""상호작용 특성 생성"""
|
|
503
|
-
interactions = pd.DataFrame(index=self.X.index)
|
|
504
|
-
|
|
505
|
-
for i, col1 in enumerate(self.X.columns):
|
|
506
|
-
for col2 in self.X.columns[i+1:]:
|
|
507
|
-
# 곱셈 상호작용
|
|
508
|
-
interactions[f'{col1}_x_{col2}'] = self.X[col1] * self.X[col2]
|
|
509
|
-
|
|
510
|
-
# 나눗셈 상호작용 (0 방지)
|
|
511
|
-
denominator = self.X[col2].replace(0, np.nan)
|
|
512
|
-
interactions[f'{col1}_div_{col2}'] = self.X[col1] / denominator
|
|
513
|
-
|
|
514
|
-
return interactions
|
|
515
|
-
|
|
516
|
-
def create_lag_features(self, n_lags=3):
|
|
517
|
-
"""시차 특성 생성"""
|
|
518
|
-
lag_features = pd.DataFrame(index=self.X.index)
|
|
519
|
-
|
|
520
|
-
for col in self.X.columns:
|
|
521
|
-
for lag in range(1, n_lags + 1):
|
|
522
|
-
lag_features[f'{col}_lag{lag}'] = self.X[col].shift(lag)
|
|
523
|
-
|
|
524
|
-
return lag_features
|
|
525
|
-
|
|
526
|
-
def select_features(self, method='univariate', n_features=10):
|
|
527
|
-
"""특성 선택"""
|
|
528
|
-
if method == 'univariate':
|
|
529
|
-
# Univariate selection
|
|
530
|
-
selector = SelectKBest(score_func=f_classif, k=n_features)
|
|
531
|
-
selector.fit(self.X, self.y)
|
|
532
|
-
|
|
533
|
-
selected_features = self.X.columns[selector.get_support()]
|
|
534
|
-
scores = pd.DataFrame({
|
|
535
|
-
'feature': self.X.columns,
|
|
536
|
-
'score': selector.scores_
|
|
537
|
-
}).sort_values('score', ascending=False)
|
|
538
|
-
|
|
539
|
-
elif method == 'rfe':
|
|
540
|
-
# Recursive Feature Elimination
|
|
541
|
-
from sklearn.ensemble import RandomForestClassifier
|
|
542
|
-
|
|
543
|
-
estimator = RandomForestClassifier(n_estimators=100, random_state=42)
|
|
544
|
-
selector = RFE(estimator, n_features_to_select=n_features)
|
|
545
|
-
selector.fit(self.X, self.y)
|
|
546
|
-
|
|
547
|
-
selected_features = self.X.columns[selector.support_]
|
|
548
|
-
scores = pd.DataFrame({
|
|
549
|
-
'feature': self.X.columns,
|
|
550
|
-
'ranking': selector.ranking_
|
|
551
|
-
}).sort_values('ranking')
|
|
552
|
-
|
|
553
|
-
return selected_features, scores
|
|
554
|
-
|
|
555
|
-
def dimensionality_reduction(self, n_components=0.95):
|
|
556
|
-
"""차원 축소"""
|
|
557
|
-
# 표준화
|
|
558
|
-
scaler = StandardScaler()
|
|
559
|
-
X_scaled = scaler.fit_transform(self.X)
|
|
560
|
-
|
|
561
|
-
# PCA
|
|
562
|
-
pca = PCA(n_components=n_components)
|
|
563
|
-
X_pca = pca.fit_transform(X_scaled)
|
|
564
|
-
|
|
565
|
-
# 주성분 기여도
|
|
566
|
-
component_importance = pd.DataFrame({
|
|
567
|
-
'component': [f'PC{i+1}' for i in range(len(pca.components_))],
|
|
568
|
-
'explained_variance_ratio': pca.explained_variance_ratio_,
|
|
569
|
-
'cumulative_variance_ratio': np.cumsum(pca.explained_variance_ratio_)
|
|
570
|
-
})
|
|
571
|
-
|
|
572
|
-
# 특성 중요도
|
|
573
|
-
feature_importance = pd.DataFrame(
|
|
574
|
-
pca.components_.T,
|
|
575
|
-
columns=[f'PC{i+1}' for i in range(len(pca.components_))],
|
|
576
|
-
index=self.X.columns
|
|
577
|
-
)
|
|
578
|
-
|
|
579
|
-
return {
|
|
580
|
-
'transformed_data': X_pca,
|
|
581
|
-
'component_importance': component_importance,
|
|
582
|
-
'feature_importance': feature_importance,
|
|
583
|
-
'n_components': pca.n_components_
|
|
584
|
-
}
|
|
585
|
-
```
|
|
586
|
-
|
|
587
|
-
#### 클러스터링 분석
|
|
588
|
-
```python
|
|
589
|
-
from sklearn.cluster import KMeans, DBSCAN, AgglomerativeClustering
|
|
590
|
-
from sklearn.metrics import silhouette_score, calinski_harabasz_score
|
|
591
|
-
|
|
592
|
-
class ClusteringAnalysis:
|
|
593
|
-
def __init__(self, X):
|
|
594
|
-
self.X = X
|
|
595
|
-
self.scaler = StandardScaler()
|
|
596
|
-
self.X_scaled = self.scaler.fit_transform(X)
|
|
597
|
-
|
|
598
|
-
def find_optimal_clusters(self, max_k=10):
|
|
599
|
-
"""최적 클러스터 수 찾기"""
|
|
600
|
-
metrics = {
|
|
601
|
-
'k': [],
|
|
602
|
-
'inertia': [],
|
|
603
|
-
'silhouette': [],
|
|
604
|
-
'calinski_harabasz': []
|
|
605
|
-
}
|
|
606
|
-
|
|
607
|
-
for k in range(2, max_k + 1):
|
|
608
|
-
kmeans = KMeans(n_clusters=k, random_state=42)
|
|
609
|
-
labels = kmeans.fit_predict(self.X_scaled)
|
|
610
|
-
|
|
611
|
-
metrics['k'].append(k)
|
|
612
|
-
metrics['inertia'].append(kmeans.inertia_)
|
|
613
|
-
metrics['silhouette'].append(silhouette_score(self.X_scaled, labels))
|
|
614
|
-
metrics['calinski_harabasz'].append(
|
|
615
|
-
calinski_harabasz_score(self.X_scaled, labels)
|
|
616
|
-
)
|
|
617
|
-
|
|
618
|
-
# Elbow method
|
|
619
|
-
differences = np.diff(metrics['inertia'])
|
|
620
|
-
elbow_point = np.argmax(differences[1:] - differences[:-1]) + 2
|
|
621
|
-
|
|
622
|
-
return pd.DataFrame(metrics), elbow_point
|
|
623
|
-
|
|
624
|
-
def perform_clustering(self, method='kmeans', **params):
|
|
625
|
-
"""클러스터링 수행"""
|
|
626
|
-
if method == 'kmeans':
|
|
627
|
-
model = KMeans(**params)
|
|
628
|
-
elif method == 'dbscan':
|
|
629
|
-
model = DBSCAN(**params)
|
|
630
|
-
elif method == 'hierarchical':
|
|
631
|
-
model = AgglomerativeClustering(**params)
|
|
632
|
-
|
|
633
|
-
labels = model.fit_predict(self.X_scaled)
|
|
634
|
-
|
|
635
|
-
# 클러스터 프로파일
|
|
636
|
-
cluster_profiles = []
|
|
637
|
-
for cluster in np.unique(labels):
|
|
638
|
-
if cluster != -1: # DBSCAN의 noise points 제외
|
|
639
|
-
cluster_mask = labels == cluster
|
|
640
|
-
profile = {
|
|
641
|
-
'cluster': cluster,
|
|
642
|
-
'size': np.sum(cluster_mask),
|
|
643
|
-
'percentage': np.sum(cluster_mask) / len(labels) * 100
|
|
644
|
-
}
|
|
645
|
-
|
|
646
|
-
# 각 특성의 평균값
|
|
647
|
-
for col in self.X.columns:
|
|
648
|
-
profile[f'{col}_mean'] = self.X.loc[cluster_mask, col].mean()
|
|
649
|
-
profile[f'{col}_std'] = self.X.loc[cluster_mask, col].std()
|
|
650
|
-
|
|
651
|
-
cluster_profiles.append(profile)
|
|
652
|
-
|
|
653
|
-
return {
|
|
654
|
-
'labels': labels,
|
|
655
|
-
'model': model,
|
|
656
|
-
'profiles': pd.DataFrame(cluster_profiles),
|
|
657
|
-
'silhouette_score': silhouette_score(self.X_scaled, labels)
|
|
658
|
-
if len(np.unique(labels)) > 1 else None
|
|
659
|
-
}
|
|
660
|
-
```
|
|
661
|
-
|
|
662
|
-
### 데이터 시각화 고급 기법
|
|
663
|
-
|
|
664
|
-
#### 고급 시각화 패턴
|
|
665
|
-
```python
|
|
666
|
-
import matplotlib.pyplot as plt
|
|
667
|
-
import seaborn as sns
|
|
668
|
-
from matplotlib.patches import Rectangle
|
|
669
|
-
import matplotlib.patches as mpatches
|
|
670
|
-
|
|
671
|
-
class AdvancedVisualization:
|
|
672
|
-
def __init__(self, style='seaborn'):
|
|
673
|
-
plt.style.use(style)
|
|
674
|
-
self.colors = sns.color_palette('husl', 10)
|
|
675
|
-
|
|
676
|
-
def create_waterfall_chart(self, categories, values, title="Waterfall Chart"):
|
|
677
|
-
"""워터폴 차트"""
|
|
678
|
-
fig, ax = plt.subplots(figsize=(12, 8))
|
|
679
|
-
|
|
680
|
-
# 누적 계산
|
|
681
|
-
cumulative = np.cumsum(values)
|
|
682
|
-
cumulative = np.concatenate([[0], cumulative])
|
|
683
|
-
|
|
684
|
-
# 막대 그리기
|
|
685
|
-
for i, (cat, val) in enumerate(zip(categories, values)):
|
|
686
|
-
if val >= 0:
|
|
687
|
-
color = 'green'
|
|
688
|
-
bottom = cumulative[i]
|
|
689
|
-
else:
|
|
690
|
-
color = 'red'
|
|
691
|
-
bottom = cumulative[i+1]
|
|
692
|
-
|
|
693
|
-
ax.bar(i, abs(val), bottom=bottom, color=color,
|
|
694
|
-
edgecolor='black', linewidth=1)
|
|
695
|
-
|
|
696
|
-
# 값 표시
|
|
697
|
-
ax.text(i, bottom + abs(val)/2, f'{val:,.0f}',
|
|
698
|
-
ha='center', va='center', fontweight='bold')
|
|
699
|
-
|
|
700
|
-
# 연결선
|
|
701
|
-
for i in range(len(categories)):
|
|
702
|
-
if i < len(categories) - 1:
|
|
703
|
-
ax.plot([i+0.4, i+0.6], [cumulative[i+1], cumulative[i+1]],
|
|
704
|
-
'k--', alpha=0.5)
|
|
705
|
-
|
|
706
|
-
ax.set_xticks(range(len(categories)))
|
|
707
|
-
ax.set_xticklabels(categories, rotation=45, ha='right')
|
|
708
|
-
ax.set_title(title, fontsize=16, fontweight='bold')
|
|
709
|
-
ax.set_ylabel('Value', fontsize=14)
|
|
710
|
-
|
|
711
|
-
# 그리드
|
|
712
|
-
ax.yaxis.grid(True, alpha=0.3)
|
|
713
|
-
ax.set_axisbelow(True)
|
|
714
|
-
|
|
715
|
-
plt.tight_layout()
|
|
716
|
-
return fig
|
|
717
|
-
|
|
718
|
-
def create_bullet_chart(self, value, target, ranges, title=""):
|
|
719
|
-
"""불릿 차트"""
|
|
720
|
-
fig, ax = plt.subplots(figsize=(8, 3))
|
|
721
|
-
|
|
722
|
-
# 범위 그리기
|
|
723
|
-
colors = ['#d3d3d3', '#a9a9a9', '#808080']
|
|
724
|
-
for i, (start, end) in enumerate(ranges):
|
|
725
|
-
ax.barh(0, end - start, left=start, height=1,
|
|
726
|
-
color=colors[i], alpha=0.5)
|
|
727
|
-
|
|
728
|
-
# 실제 값
|
|
729
|
-
ax.barh(0, value, height=0.5, color='black')
|
|
730
|
-
|
|
731
|
-
# 목표 값
|
|
732
|
-
ax.plot([target, target], [-0.6, 0.6], 'r-', linewidth=3)
|
|
733
|
-
|
|
734
|
-
ax.set_xlim(0, max(ranges[-1][1], value, target) * 1.1)
|
|
735
|
-
ax.set_ylim(-1, 1)
|
|
736
|
-
ax.set_yticks([])
|
|
737
|
-
ax.set_title(title, fontsize=14, fontweight='bold')
|
|
738
|
-
|
|
739
|
-
# 레이블
|
|
740
|
-
ax.text(value/2, 0, f'{value:,.0f}',
|
|
741
|
-
ha='center', va='center', color='white', fontweight='bold')
|
|
742
|
-
ax.text(target, -0.8, f'Target: {target:,.0f}',
|
|
743
|
-
ha='center', fontsize=10)
|
|
744
|
-
|
|
745
|
-
return fig
|
|
746
|
-
|
|
747
|
-
def create_sankey_diagram(self, source, target, value, labels):
|
|
748
|
-
"""생키 다이어그램"""
|
|
749
|
-
import plotly.graph_objects as go
|
|
750
|
-
|
|
751
|
-
# 노드 인덱스 매핑
|
|
752
|
-
all_nodes = list(set(source + target))
|
|
753
|
-
node_indices = {node: i for i, node in enumerate(all_nodes)}
|
|
754
|
-
|
|
755
|
-
source_indices = [node_indices[s] for s in source]
|
|
756
|
-
target_indices = [node_indices[t] for t in target]
|
|
757
|
-
|
|
758
|
-
fig = go.Figure(data=[go.Sankey(
|
|
759
|
-
node=dict(
|
|
760
|
-
pad=15,
|
|
761
|
-
thickness=20,
|
|
762
|
-
line=dict(color="black", width=0.5),
|
|
763
|
-
label=all_nodes,
|
|
764
|
-
color="blue"
|
|
765
|
-
),
|
|
766
|
-
link=dict(
|
|
767
|
-
source=source_indices,
|
|
768
|
-
target=target_indices,
|
|
769
|
-
value=value,
|
|
770
|
-
color="rgba(0,0,255,0.4)"
|
|
771
|
-
)
|
|
772
|
-
)])
|
|
773
|
-
|
|
774
|
-
fig.update_layout(
|
|
775
|
-
title_text="Sankey Diagram",
|
|
776
|
-
font_size=10,
|
|
777
|
-
height=600
|
|
778
|
-
)
|
|
779
|
-
|
|
780
|
-
return fig
|
|
781
|
-
|
|
782
|
-
def create_radar_chart(self, categories, values_dict, title="Radar Chart"):
|
|
783
|
-
"""레이더 차트 (여러 그룹 비교)"""
|
|
784
|
-
angles = np.linspace(0, 2 * np.pi, len(categories), endpoint=False).tolist()
|
|
785
|
-
angles += angles[:1]
|
|
786
|
-
|
|
787
|
-
fig, ax = plt.subplots(figsize=(8, 8), subplot_kw=dict(projection='polar'))
|
|
788
|
-
|
|
789
|
-
for label, values in values_dict.items():
|
|
790
|
-
values = values.tolist()
|
|
791
|
-
values += values[:1]
|
|
792
|
-
ax.plot(angles, values, 'o-', linewidth=2, label=label)
|
|
793
|
-
ax.fill(angles, values, alpha=0.25)
|
|
794
|
-
|
|
795
|
-
ax.set_theta_offset(np.pi / 2)
|
|
796
|
-
ax.set_theta_direction(-1)
|
|
797
|
-
ax.set_thetagrids(np.degrees(angles[:-1]), categories)
|
|
798
|
-
|
|
799
|
-
ax.set_ylim(0, max([max(v) for v in values_dict.values()]) * 1.1)
|
|
800
|
-
ax.set_title(title, fontsize=16, fontweight='bold', pad=20)
|
|
801
|
-
ax.legend(loc='upper right', bbox_to_anchor=(1.3, 1.0))
|
|
802
|
-
|
|
803
|
-
return fig
|
|
804
|
-
```
|
|
805
|
-
|
|
806
|
-
### 리포트 자동화
|
|
807
|
-
|
|
808
|
-
#### 자동 리포트 생성 시스템
|
|
809
|
-
```python
|
|
810
|
-
from datetime import datetime
|
|
811
|
-
import matplotlib.pyplot as plt
|
|
812
|
-
from matplotlib.backends.backend_pdf import PdfPages
|
|
813
|
-
import markdown
|
|
814
|
-
from weasyprint import HTML, CSS
|
|
815
|
-
|
|
816
|
-
class AutomatedReportGenerator:
|
|
817
|
-
def __init__(self, template_engine='jinja2'):
|
|
818
|
-
self.template_engine = template_engine
|
|
819
|
-
self.sections = []
|
|
820
|
-
|
|
821
|
-
def add_section(self, title, content, section_type='text'):
|
|
822
|
-
"""섹션 추가"""
|
|
823
|
-
self.sections.append({
|
|
824
|
-
'title': title,
|
|
825
|
-
'content': content,
|
|
826
|
-
'type': section_type,
|
|
827
|
-
'timestamp': datetime.now()
|
|
828
|
-
})
|
|
829
|
-
|
|
830
|
-
def generate_pdf_report(self, filename='report.pdf'):
|
|
831
|
-
"""PDF 리포트 생성"""
|
|
832
|
-
with PdfPages(filename) as pdf:
|
|
833
|
-
# 표지 페이지
|
|
834
|
-
self._create_cover_page(pdf)
|
|
835
|
-
|
|
836
|
-
# 목차
|
|
837
|
-
self._create_table_of_contents(pdf)
|
|
838
|
-
|
|
839
|
-
# 섹션별 내용
|
|
840
|
-
for i, section in enumerate(self.sections):
|
|
841
|
-
if section['type'] == 'text':
|
|
842
|
-
self._add_text_page(pdf, section)
|
|
843
|
-
elif section['type'] == 'chart':
|
|
844
|
-
self._add_chart_page(pdf, section)
|
|
845
|
-
elif section['type'] == 'table':
|
|
846
|
-
self._add_table_page(pdf, section)
|
|
847
|
-
|
|
848
|
-
# 메타데이터
|
|
849
|
-
d = pdf.infodict()
|
|
850
|
-
d['Title'] = 'Automated Data Analysis Report'
|
|
851
|
-
d['Author'] = 'Data Analysis System'
|
|
852
|
-
d['Subject'] = 'Data Analysis'
|
|
853
|
-
d['Keywords'] = 'Data, Analysis, Report'
|
|
854
|
-
d['CreationDate'] = datetime.now()
|
|
855
|
-
|
|
856
|
-
def _create_cover_page(self, pdf):
|
|
857
|
-
"""표지 페이지 생성"""
|
|
858
|
-
fig = plt.figure(figsize=(8.5, 11))
|
|
859
|
-
fig.text(0.5, 0.6, 'Data Analysis Report',
|
|
860
|
-
ha='center', va='center', fontsize=24, fontweight='bold')
|
|
861
|
-
fig.text(0.5, 0.5, f'Generated on {datetime.now().strftime("%Y-%m-%d")}',
|
|
862
|
-
ha='center', va='center', fontsize=16)
|
|
863
|
-
fig.text(0.5, 0.2, 'Automated Report Generation System',
|
|
864
|
-
ha='center', va='center', fontsize=12, style='italic')
|
|
865
|
-
pdf.savefig(fig, bbox_inches='tight')
|
|
866
|
-
plt.close()
|
|
867
|
-
|
|
868
|
-
def generate_html_report(self, filename='report.html'):
|
|
869
|
-
"""HTML 리포트 생성"""
|
|
870
|
-
html_template = """
|
|
871
|
-
<!DOCTYPE html>
|
|
872
|
-
<html>
|
|
873
|
-
<head>
|
|
874
|
-
<meta charset="utf-8">
|
|
875
|
-
<title>Data Analysis Report</title>
|
|
876
|
-
<style>
|
|
877
|
-
body {{ font-family: Arial, sans-serif; margin: 40px; }}
|
|
878
|
-
h1 {{ color: #333; border-bottom: 2px solid #333; }}
|
|
879
|
-
h2 {{ color: #666; }}
|
|
880
|
-
.chart {{ margin: 20px 0; text-align: center; }}
|
|
881
|
-
.table {{ margin: 20px 0; }}
|
|
882
|
-
table {{ border-collapse: collapse; width: 100%; }}
|
|
883
|
-
th, td {{ border: 1px solid #ddd; padding: 8px; text-align: left; }}
|
|
884
|
-
th {{ background-color: #f2f2f2; font-weight: bold; }}
|
|
885
|
-
.summary-box {{
|
|
886
|
-
background-color: #f0f0f0;
|
|
887
|
-
padding: 15px;
|
|
888
|
-
border-radius: 5px;
|
|
889
|
-
margin: 20px 0;
|
|
890
|
-
}}
|
|
891
|
-
.metric {{
|
|
892
|
-
display: inline-block;
|
|
893
|
-
margin: 10px 20px;
|
|
894
|
-
text-align: center;
|
|
895
|
-
}}
|
|
896
|
-
.metric-value {{
|
|
897
|
-
font-size: 24px;
|
|
898
|
-
font-weight: bold;
|
|
899
|
-
color: #2c3e50;
|
|
900
|
-
}}
|
|
901
|
-
.metric-label {{
|
|
902
|
-
font-size: 14px;
|
|
903
|
-
color: #7f8c8d;
|
|
904
|
-
}}
|
|
905
|
-
</style>
|
|
906
|
-
</head>
|
|
907
|
-
<body>
|
|
908
|
-
<h1>Data Analysis Report</h1>
|
|
909
|
-
<p>Generated on: {date}</p>
|
|
910
|
-
|
|
911
|
-
{content}
|
|
912
|
-
</body>
|
|
913
|
-
</html>
|
|
914
|
-
"""
|
|
915
|
-
|
|
916
|
-
content = self._generate_html_content()
|
|
917
|
-
html = html_template.format(
|
|
918
|
-
date=datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
|
|
919
|
-
content=content
|
|
920
|
-
)
|
|
921
|
-
|
|
922
|
-
with open(filename, 'w', encoding='utf-8') as f:
|
|
923
|
-
f.write(html)
|
|
924
|
-
|
|
925
|
-
def _generate_html_content(self):
|
|
926
|
-
"""HTML 콘텐츠 생성"""
|
|
927
|
-
content_parts = []
|
|
928
|
-
|
|
929
|
-
for section in self.sections:
|
|
930
|
-
if section['type'] == 'text':
|
|
931
|
-
content_parts.append(f"<h2>{section['title']}</h2>")
|
|
932
|
-
content_parts.append(f"<p>{section['content']}</p>")
|
|
933
|
-
|
|
934
|
-
elif section['type'] == 'metrics':
|
|
935
|
-
content_parts.append(f"<h2>{section['title']}</h2>")
|
|
936
|
-
content_parts.append('<div class="summary-box">')
|
|
937
|
-
|
|
938
|
-
for metric, value in section['content'].items():
|
|
939
|
-
content_parts.append(f'''
|
|
940
|
-
<div class="metric">
|
|
941
|
-
<div class="metric-value">{value}</div>
|
|
942
|
-
<div class="metric-label">{metric}</div>
|
|
943
|
-
</div>
|
|
944
|
-
''')
|
|
945
|
-
|
|
946
|
-
content_parts.append('</div>')
|
|
947
|
-
|
|
948
|
-
elif section['type'] == 'table':
|
|
949
|
-
content_parts.append(f"<h2>{section['title']}</h2>")
|
|
950
|
-
content_parts.append('<div class="table">')
|
|
951
|
-
content_parts.append(section['content'].to_html(index=False))
|
|
952
|
-
content_parts.append('</div>')
|
|
953
|
-
|
|
954
|
-
return '\n'.join(content_parts)
|
|
955
|
-
```
|
|
956
|
-
|
|
957
|
-
### 비즈니스 인텔리전스
|
|
958
|
-
|
|
959
|
-
#### KPI 대시보드
|
|
960
|
-
```python
|
|
961
|
-
class KPIDashboard:
|
|
962
|
-
def __init__(self, data):
|
|
963
|
-
self.data = data
|
|
964
|
-
self.kpis = {}
|
|
965
|
-
|
|
966
|
-
def calculate_kpis(self):
|
|
967
|
-
"""주요 KPI 계산"""
|
|
968
|
-
self.kpis = {
|
|
969
|
-
'revenue': {
|
|
970
|
-
'current': self.data['revenue'].sum(),
|
|
971
|
-
'previous': self.data['revenue_prev'].sum(),
|
|
972
|
-
'growth': self._calculate_growth('revenue', 'revenue_prev'),
|
|
973
|
-
'target': self.data['revenue_target'].sum()
|
|
974
|
-
},
|
|
975
|
-
'customers': {
|
|
976
|
-
'total': self.data['customer_id'].nunique(),
|
|
977
|
-
'new': self.data[self.data['is_new_customer']]['customer_id'].nunique(),
|
|
978
|
-
'retention_rate': self._calculate_retention_rate(),
|
|
979
|
-
'churn_rate': self._calculate_churn_rate()
|
|
980
|
-
},
|
|
981
|
-
'operations': {
|
|
982
|
-
'avg_order_value': self.data['revenue'].mean(),
|
|
983
|
-
'conversion_rate': self._calculate_conversion_rate(),
|
|
984
|
-
'fulfillment_time': self.data['fulfillment_hours'].mean()
|
|
985
|
-
}
|
|
986
|
-
}
|
|
987
|
-
|
|
988
|
-
def _calculate_growth(self, current_col, previous_col):
|
|
989
|
-
"""성장률 계산"""
|
|
990
|
-
current = self.data[current_col].sum()
|
|
991
|
-
previous = self.data[previous_col].sum()
|
|
992
|
-
|
|
993
|
-
if previous == 0:
|
|
994
|
-
return float('inf')
|
|
995
|
-
|
|
996
|
-
return ((current - previous) / previous) * 100
|
|
997
|
-
|
|
998
|
-
def create_scorecard(self):
|
|
999
|
-
"""스코어카드 생성"""
|
|
1000
|
-
fig, axes = plt.subplots(2, 3, figsize=(15, 10))
|
|
1001
|
-
axes = axes.flatten()
|
|
1002
|
-
|
|
1003
|
-
# Revenue KPI
|
|
1004
|
-
self._create_kpi_widget(
|
|
1005
|
-
axes[0],
|
|
1006
|
-
'Revenue',
|
|
1007
|
-
self.kpis['revenue']['current'],
|
|
1008
|
-
self.kpis['revenue']['growth'],
|
|
1009
|
-
self.kpis['revenue']['target']
|
|
1010
|
-
)
|
|
1011
|
-
|
|
1012
|
-
# Customer KPIs
|
|
1013
|
-
self._create_kpi_widget(
|
|
1014
|
-
axes[1],
|
|
1015
|
-
'Total Customers',
|
|
1016
|
-
self.kpis['customers']['total'],
|
|
1017
|
-
self.kpis['customers']['new'],
|
|
1018
|
-
None
|
|
1019
|
-
)
|
|
1020
|
-
|
|
1021
|
-
# Retention Rate
|
|
1022
|
-
self._create_gauge_chart(
|
|
1023
|
-
axes[2],
|
|
1024
|
-
'Retention Rate',
|
|
1025
|
-
self.kpis['customers']['retention_rate'],
|
|
1026
|
-
[0, 100]
|
|
1027
|
-
)
|
|
1028
|
-
|
|
1029
|
-
# Continue with other KPIs...
|
|
1030
|
-
|
|
1031
|
-
plt.tight_layout()
|
|
1032
|
-
return fig
|
|
1033
|
-
|
|
1034
|
-
def _create_kpi_widget(self, ax, title, value, change, target=None):
|
|
1035
|
-
"""KPI 위젯 생성"""
|
|
1036
|
-
ax.set_xlim(0, 1)
|
|
1037
|
-
ax.set_ylim(0, 1)
|
|
1038
|
-
ax.axis('off')
|
|
1039
|
-
|
|
1040
|
-
# 제목
|
|
1041
|
-
ax.text(0.5, 0.9, title, ha='center', fontsize=14, fontweight='bold')
|
|
1042
|
-
|
|
1043
|
-
# 값
|
|
1044
|
-
ax.text(0.5, 0.6, f'{value:,.0f}', ha='center', fontsize=24, fontweight='bold')
|
|
1045
|
-
|
|
1046
|
-
# 변화율
|
|
1047
|
-
color = 'green' if change >= 0 else 'red'
|
|
1048
|
-
arrow = '▲' if change >= 0 else '▼'
|
|
1049
|
-
ax.text(0.5, 0.4, f'{arrow} {abs(change):.1f}%',
|
|
1050
|
-
ha='center', fontsize=16, color=color)
|
|
1051
|
-
|
|
1052
|
-
# 목표
|
|
1053
|
-
if target:
|
|
1054
|
-
progress = (value / target) * 100
|
|
1055
|
-
ax.text(0.5, 0.2, f'Target: {progress:.1f}%',
|
|
1056
|
-
ha='center', fontsize=12, color='gray')
|
|
1057
|
-
```
|