exam-sub 1.0.0 → 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/cli.js +10 -0
- package/data/aiop.txt +105 -26
- package/data/bdh.txt +105 -55
- package/data/div.txt +249 -67
- package/data/mlf.txt +225 -119
- package/data/rpa.txt +279 -0
- package/package.json +1 -1
package/data/mlf.txt
CHANGED
|
@@ -1,26 +1,32 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
1
|
+
================================================================================
|
|
2
|
+
Q1: SALES DATA EXPLORATORY DATA ANALYSIS (EDA)
|
|
3
|
+
================================================================================
|
|
4
|
+
|
|
5
|
+
AIM:
|
|
6
|
+
Perform Exploratory Data Analysis on sales data to understand patterns, missing
|
|
7
|
+
values, and unusual transactions.
|
|
8
|
+
|
|
9
|
+
PYTHON PROGRAM:
|
|
10
|
+
--------------------------------------------------------------------------------
|
|
5
11
|
import pandas as pd
|
|
6
12
|
import numpy as np
|
|
7
13
|
import matplotlib.pyplot as plt
|
|
8
14
|
import seaborn as sns
|
|
15
|
+
|
|
16
|
+
# 1. Load Data and Inspect
|
|
9
17
|
df = pd.read_csv("sales_data.csv")
|
|
10
18
|
print(df.head())
|
|
11
19
|
print(df.shape)
|
|
12
20
|
print(df.info())
|
|
13
21
|
|
|
14
|
-
2. Convert Date
|
|
22
|
+
# 2. Convert Date
|
|
15
23
|
df['Date'] = pd.to_datetime(df['Date'])
|
|
16
24
|
print(df['Date'].dtype)
|
|
17
25
|
|
|
18
|
-
3. Identify
|
|
26
|
+
# 3. Identify Variable Types
|
|
19
27
|
print(df.dtypes)
|
|
20
|
-
Typical types: Order_ID – integer/object; Date – DateTime; Product, Category, Region, Customer_Type – categorical;
|
|
21
|
-
Quantity, Unit_Price, Discount, Sales – numerical.
|
|
22
28
|
|
|
23
|
-
4. Check and
|
|
29
|
+
# 4 & 5. Check and Handle Missing Values
|
|
24
30
|
print(df.isnull().sum())
|
|
25
31
|
df['Quantity'] = df['Quantity'].fillna(df['Quantity'].median())
|
|
26
32
|
df['Unit_Price'] = df['Unit_Price'].fillna(df['Unit_Price'].median())
|
|
@@ -29,102 +35,110 @@ df['Sales'] = df['Sales'].fillna(df['Sales'].median())
|
|
|
29
35
|
df['Region'] = df['Region'].fillna(df['Region'].mode()[0])
|
|
30
36
|
df['Category'] = df['Category'].fillna(df['Category'].mode()[0])
|
|
31
37
|
|
|
32
|
-
6. Univariate
|
|
38
|
+
# 6. Univariate Analysis
|
|
33
39
|
sns.histplot(df['Sales'], kde=True)
|
|
34
40
|
plt.title("Sales Distribution")
|
|
35
41
|
plt.show()
|
|
42
|
+
|
|
36
43
|
sns.histplot(df['Quantity'], kde=True)
|
|
37
44
|
plt.title("Quantity Distribution")
|
|
38
45
|
plt.show()
|
|
39
46
|
|
|
40
|
-
7. Bivariate
|
|
47
|
+
# 7. Bivariate Analysis
|
|
41
48
|
# Quantity vs Sales
|
|
42
49
|
sns.scatterplot(x='Quantity', y='Sales', data=df)
|
|
43
50
|
plt.show()
|
|
51
|
+
|
|
44
52
|
# Discount vs Sales
|
|
45
53
|
sns.scatterplot(x='Discount', y='Sales', data=df)
|
|
46
54
|
plt.show()
|
|
55
|
+
|
|
47
56
|
# Region vs Sales
|
|
48
57
|
sns.boxplot(x='Region', y='Sales', data=df)
|
|
49
58
|
plt.show()
|
|
50
59
|
|
|
51
|
-
8. Multivariate
|
|
60
|
+
# 8. Multivariate Analysis
|
|
52
61
|
sns.pairplot(df[['Quantity', 'Unit_Price', 'Discount', 'Sales']])
|
|
53
62
|
plt.show()
|
|
54
63
|
|
|
55
|
-
9. Detect
|
|
56
|
-
|
|
64
|
+
# 9. Detect Sales Outliers using IQR
|
|
57
65
|
Q1 = df['Sales'].quantile(0.25)
|
|
58
66
|
Q3 = df['Sales'].quantile(0.75)
|
|
59
67
|
IQR = Q3 - Q1
|
|
60
68
|
lower = Q1 - 1.5 * IQR
|
|
61
69
|
upper = Q3 + 1.5 * IQR
|
|
70
|
+
|
|
62
71
|
outliers = df[(df['Sales'] < lower) | (df['Sales'] > upper)]
|
|
63
72
|
print("Number of outliers:", len(outliers))
|
|
64
73
|
print(outliers)
|
|
65
74
|
|
|
66
|
-
10. Handle
|
|
75
|
+
# 10 & 11. Handle Outliers and Plot Final Distribution
|
|
67
76
|
df_clean = df[(df['Sales'] >= lower) & (df['Sales'] <= upper)]
|
|
68
77
|
sns.histplot(df_clean['Sales'], kde=True)
|
|
69
78
|
plt.title("Sales Distribution After Outlier Removal")
|
|
70
79
|
plt.show()
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
80
|
+
--------------------------------------------------------------------------------
|
|
81
|
+
|
|
82
|
+
CONCLUSION:
|
|
83
|
+
EDA helps understand sales distribution, relationships, regional patterns,
|
|
84
|
+
missing values, and unusual transactions. IQR is used to detect sales outliers.
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
================================================================================
|
|
88
|
+
Q2: AUTOMATED EDA AND STATISTICAL RELATIONSHIP ANALYSIS
|
|
89
|
+
================================================================================
|
|
90
|
+
|
|
91
|
+
AIM:
|
|
92
|
+
Analyze student academic performance and relationships between study habits,
|
|
93
|
+
attendance, assignments, and marks using manual and automated EDA packages.
|
|
94
|
+
|
|
95
|
+
PYTHON PROGRAM:
|
|
96
|
+
--------------------------------------------------------------------------------
|
|
77
97
|
import pandas as pd
|
|
78
98
|
import numpy as np
|
|
79
99
|
import matplotlib.pyplot as plt
|
|
80
100
|
import seaborn as sns
|
|
101
|
+
|
|
102
|
+
# A. Data Loading and Basic Analysis
|
|
81
103
|
df = pd.read_csv("student_performance.csv")
|
|
82
104
|
print(df.head())
|
|
83
105
|
print(df.tail())
|
|
84
106
|
print("Shape:", df.shape)
|
|
85
107
|
print("Columns:", df.columns)
|
|
86
|
-
print("Data Types
|
|
87
|
-
print(df.
|
|
88
|
-
print("
|
|
89
|
-
print(df.describe())
|
|
90
|
-
print("Missing values:")
|
|
91
|
-
print(df.isnull().sum())
|
|
108
|
+
print("Data Types:\n", df.dtypes)
|
|
109
|
+
print("Statistical Summary:\n", df.describe())
|
|
110
|
+
print("Missing values:\n", df.isnull().sum())
|
|
92
111
|
print("Duplicates:", df.duplicated().sum())
|
|
93
|
-
print("Unique values
|
|
94
|
-
|
|
95
|
-
Handle missing and duplicate values
|
|
112
|
+
print("Unique values:\n", df.nunique())
|
|
113
|
+
|
|
114
|
+
# Handle missing and duplicate values
|
|
96
115
|
df = df.fillna(df.median(numeric_only=True))
|
|
97
116
|
df = df.drop_duplicates()
|
|
98
|
-
|
|
99
|
-
#
|
|
100
|
-
# pip install dtale
|
|
117
|
+
|
|
118
|
+
# B1. D-Tale
|
|
119
|
+
# Terminal install: pip install dtale
|
|
101
120
|
import dtale
|
|
102
121
|
dtale.show(df)
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
# Install once:
|
|
107
|
-
# pip install ydata-profiling
|
|
122
|
+
|
|
123
|
+
# B2. YData Profiling
|
|
124
|
+
# Terminal install: pip install ydata-profiling
|
|
108
125
|
from ydata_profiling import ProfileReport
|
|
109
126
|
profile = ProfileReport(df, title="Student Performance Report")
|
|
110
127
|
profile.to_file("student_profile.html")
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
# Install once:
|
|
115
|
-
# pip install sweetviz
|
|
128
|
+
|
|
129
|
+
# B3. Sweetviz
|
|
130
|
+
# Terminal install: pip install sweetviz
|
|
116
131
|
import sweetviz as sv
|
|
117
132
|
report = sv.analyze(df)
|
|
118
133
|
report.show_html("sweetviz_report.html")
|
|
119
|
-
|
|
120
|
-
B4. AutoViz
|
|
121
|
-
#
|
|
122
|
-
# pip install autoviz
|
|
134
|
+
|
|
135
|
+
# B4. AutoViz
|
|
136
|
+
# Terminal install: pip install autoviz
|
|
123
137
|
from autoviz.AutoViz_Class import AutoViz_Class
|
|
124
138
|
AV = AutoViz_Class()
|
|
125
139
|
AV.AutoViz("student_performance.csv")
|
|
126
|
-
|
|
127
|
-
C. Covariance and Correlation
|
|
140
|
+
|
|
141
|
+
# C. Covariance and Correlation Analysis
|
|
128
142
|
num_cols = [
|
|
129
143
|
'Study_Hours',
|
|
130
144
|
'Attendance',
|
|
@@ -132,36 +146,53 @@ num_cols = [
|
|
|
132
146
|
'Internal_Marks',
|
|
133
147
|
'Final_Marks'
|
|
134
148
|
]
|
|
149
|
+
|
|
135
150
|
covariance = df[num_cols].cov()
|
|
136
|
-
print(covariance)
|
|
151
|
+
print("Covariance Matrix:\n", covariance)
|
|
152
|
+
|
|
137
153
|
correlation = df[num_cols].corr()
|
|
138
|
-
print(correlation)
|
|
154
|
+
print("Correlation Matrix:\n", correlation)
|
|
155
|
+
|
|
139
156
|
plt.figure(figsize=(8, 6))
|
|
140
157
|
sns.heatmap(correlation, annot=True, cmap='coolwarm')
|
|
141
158
|
plt.title("Correlation Matrix")
|
|
142
159
|
plt.show()
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
160
|
+
--------------------------------------------------------------------------------
|
|
161
|
+
|
|
162
|
+
CONCLUSION:
|
|
163
|
+
Automated EDA tools (D-Tale, YData Profiling, Sweetviz, AutoViz) accelerate
|
|
164
|
+
analysis by generating statistics, plots, correlation matrices, missing-value
|
|
165
|
+
reports, and warnings automatically.
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
================================================================================
|
|
169
|
+
Q3: STUDENT PERFORMANCE PREDICTION (LINEAR REGRESSION)
|
|
170
|
+
================================================================================
|
|
171
|
+
|
|
172
|
+
AIM:
|
|
173
|
+
Predict student Final_Marks using Multiple Linear Regression.
|
|
174
|
+
|
|
175
|
+
PYTHON PROGRAM:
|
|
176
|
+
--------------------------------------------------------------------------------
|
|
152
177
|
import pandas as pd
|
|
153
178
|
import numpy as np
|
|
179
|
+
import matplotlib.pyplot as plt
|
|
180
|
+
from sklearn.model_selection import train_test_split
|
|
181
|
+
from sklearn.linear_model import LinearRegression
|
|
182
|
+
from sklearn.metrics import r2_score, mean_squared_error
|
|
183
|
+
|
|
184
|
+
# Task 1: Load and Preprocess Data
|
|
154
185
|
df = pd.read_csv("student_performance.csv")
|
|
155
186
|
print(df.head())
|
|
156
187
|
print(df.shape)
|
|
157
188
|
print(df.dtypes)
|
|
158
189
|
print(df.describe())
|
|
159
190
|
print(df.isnull().sum())
|
|
191
|
+
|
|
160
192
|
df = df.fillna(df.median(numeric_only=True))
|
|
161
193
|
df = df.drop_duplicates()
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
from sklearn.linear_model import LinearRegression
|
|
194
|
+
|
|
195
|
+
# Task 2: Apply Linear Regression
|
|
165
196
|
X = df[
|
|
166
197
|
[
|
|
167
198
|
'Study_Hours',
|
|
@@ -172,116 +203,162 @@ X = df[
|
|
|
172
203
|
]
|
|
173
204
|
]
|
|
174
205
|
y = df['Final_Marks']
|
|
206
|
+
|
|
175
207
|
X_train, X_test, y_train, y_test = train_test_split(
|
|
176
208
|
X, y, test_size=0.20, random_state=42
|
|
177
209
|
)
|
|
210
|
+
|
|
178
211
|
model = LinearRegression()
|
|
179
212
|
model.fit(X_train, y_train)
|
|
213
|
+
|
|
180
214
|
print("Intercept:", model.intercept_)
|
|
181
215
|
for col, coef in zip(X.columns, model.coef_):
|
|
182
|
-
print(col
|
|
216
|
+
print(f"{col}: {coef}")
|
|
217
|
+
|
|
183
218
|
y_pred = model.predict(X_test)
|
|
184
|
-
print("Predicted Final Marks
|
|
185
|
-
|
|
186
|
-
Task 3
|
|
187
|
-
from sklearn.metrics import r2_score, mean_squared_error
|
|
219
|
+
print("Predicted Final Marks:\n", y_pred)
|
|
220
|
+
|
|
221
|
+
# Task 3: Evaluate Model
|
|
188
222
|
r2 = r2_score(y_test, y_pred)
|
|
189
223
|
mse = mean_squared_error(y_test, y_pred)
|
|
190
224
|
rmse = np.sqrt(mse)
|
|
191
225
|
residuals = y_test - y_pred
|
|
192
226
|
RSS = np.sum(residuals ** 2)
|
|
193
|
-
|
|
227
|
+
|
|
228
|
+
print("R2 Score:", r2)
|
|
194
229
|
print("MSE:", mse)
|
|
195
230
|
print("RMSE:", rmse)
|
|
196
|
-
print("Residuals
|
|
197
|
-
print(
|
|
198
|
-
|
|
231
|
+
print("Residuals:\n", residuals)
|
|
232
|
+
print("Residual Sum of Squares (RSS):", RSS)
|
|
233
|
+
|
|
199
234
|
plt.scatter(y_test, y_pred)
|
|
200
235
|
plt.xlabel("Actual Final Marks")
|
|
201
236
|
plt.ylabel("Predicted Final Marks")
|
|
202
237
|
plt.title("Actual vs Predicted Final Marks")
|
|
203
238
|
plt.show()
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
239
|
+
--------------------------------------------------------------------------------
|
|
240
|
+
|
|
241
|
+
CONCLUSION:
|
|
242
|
+
A higher R² indicates the model explains a significant portion of variance in
|
|
243
|
+
Final_Marks. Lower MSE and RMSE indicate smaller prediction error.
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
================================================================================
|
|
247
|
+
Q4: SALES DATA ANALYSIS AND VISUALIZATION USING PYTHON
|
|
248
|
+
================================================================================
|
|
249
|
+
|
|
250
|
+
AIM:
|
|
251
|
+
Load, clean, transform, analyze, and visualize sales performance metrics.
|
|
252
|
+
|
|
253
|
+
PYTHON PROGRAM:
|
|
254
|
+
--------------------------------------------------------------------------------
|
|
209
255
|
import pandas as pd
|
|
210
256
|
import numpy as np
|
|
211
257
|
import matplotlib.pyplot as plt
|
|
212
258
|
import seaborn as sns
|
|
259
|
+
|
|
260
|
+
# 1. Load and Inspect
|
|
213
261
|
df = pd.read_csv("sales_data.csv")
|
|
214
262
|
print(df.head())
|
|
215
263
|
print(df.columns)
|
|
216
264
|
print(df.dtypes)
|
|
217
265
|
print(df.isnull().sum())
|
|
218
|
-
|
|
266
|
+
|
|
267
|
+
# 2. Fill Missing Price using Category Mean
|
|
219
268
|
df['Price'] = df.groupby('Category')['Price'].transform(
|
|
220
269
|
lambda x: x.fillna(x.mean())
|
|
221
270
|
)
|
|
222
|
-
print(df['Price'].isnull().sum())
|
|
223
|
-
|
|
271
|
+
print("Missing Price values:", df['Price'].isnull().sum())
|
|
272
|
+
|
|
273
|
+
# 3. Create TotalAmount Column
|
|
224
274
|
df['TotalAmount'] = df['Quantity'] * df['Price']
|
|
225
275
|
print(df.head())
|
|
226
|
-
|
|
276
|
+
|
|
277
|
+
# 4. Data Analysis
|
|
227
278
|
# Highest total sales region
|
|
228
279
|
region_sales = df.groupby('Region')['TotalAmount'].sum()
|
|
229
|
-
print(region_sales)
|
|
280
|
+
print("Region Sales:\n", region_sales)
|
|
230
281
|
print("Highest Sales Region:", region_sales.idxmax())
|
|
282
|
+
|
|
231
283
|
# Total Electronics revenue
|
|
232
284
|
electronics_sales = df[
|
|
233
285
|
df['Category'] == 'Electronics'
|
|
234
286
|
]['TotalAmount'].sum()
|
|
235
287
|
print("Electronics Revenue:", electronics_sales)
|
|
288
|
+
|
|
236
289
|
# Highest quantity in a single invoice
|
|
237
290
|
highest = df.loc[df['Quantity'].idxmax()]
|
|
238
|
-
print(highest)
|
|
291
|
+
print("Invoice with Highest Quantity:\n", highest)
|
|
239
292
|
|
|
240
|
-
5. Data
|
|
241
|
-
# Bar
|
|
293
|
+
# 5. Data Visualization
|
|
294
|
+
# Bar Chart: Sales by Region
|
|
242
295
|
region_sales.plot(kind='bar')
|
|
243
296
|
plt.title("Total Sales by Region")
|
|
244
297
|
plt.xlabel("Region")
|
|
245
298
|
plt.ylabel("Sales")
|
|
246
299
|
plt.show()
|
|
247
|
-
|
|
300
|
+
|
|
301
|
+
# Pie Chart: Sales by Category
|
|
248
302
|
category_sales = df.groupby('Category')['TotalAmount'].sum()
|
|
249
303
|
category_sales.plot(kind='pie', autopct='%1.1f%%')
|
|
250
304
|
plt.title("Sales Distribution by Category")
|
|
251
305
|
plt.ylabel("")
|
|
252
306
|
plt.show()
|
|
253
|
-
|
|
307
|
+
|
|
308
|
+
# Line Chart: Monthly Sales
|
|
254
309
|
df['Date'] = pd.to_datetime(df['Date'])
|
|
255
310
|
monthly_sales = df.groupby(
|
|
256
311
|
df['Date'].dt.to_period('M')
|
|
257
312
|
)['TotalAmount'].sum()
|
|
258
313
|
monthly_sales.plot(kind='line', marker='o')
|
|
259
|
-
plt.title("Monthly Sales")
|
|
314
|
+
plt.title("Monthly Sales Trend")
|
|
260
315
|
plt.xlabel("Month")
|
|
261
316
|
plt.ylabel("Sales")
|
|
262
317
|
plt.show()
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
318
|
+
--------------------------------------------------------------------------------
|
|
319
|
+
|
|
320
|
+
CONCLUSION:
|
|
321
|
+
The dataset was successfully cleaned and transformed. Key insights including
|
|
322
|
+
top regions, electronics revenue, and sales trends were identified via plots.
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
================================================================================
|
|
326
|
+
Q5: EMPLOYEE PROMOTION PREDICTION USING LOGISTIC REGRESSION
|
|
327
|
+
================================================================================
|
|
328
|
+
|
|
329
|
+
AIM:
|
|
330
|
+
Predict promotion probability based on years of work experience using Logistic
|
|
331
|
+
Regression.
|
|
332
|
+
|
|
333
|
+
PYTHON PROGRAM:
|
|
334
|
+
--------------------------------------------------------------------------------
|
|
267
335
|
import numpy as np
|
|
268
336
|
import pandas as pd
|
|
269
337
|
import matplotlib.pyplot as plt
|
|
270
338
|
from sklearn.linear_model import LogisticRegression
|
|
271
|
-
|
|
272
|
-
|
|
339
|
+
|
|
340
|
+
# Create Sample Dataset
|
|
341
|
+
experience = np.array([1, 2, 3, 4, 5, 6, 7, 8, 9, 10])
|
|
342
|
+
promotion = np.array([0, 0, 0, 0, 0, 1, 1, 1, 1, 1])
|
|
343
|
+
|
|
273
344
|
df = pd.DataFrame({
|
|
274
345
|
'Experience': experience,
|
|
275
346
|
'Promotion': promotion
|
|
276
347
|
})
|
|
348
|
+
|
|
277
349
|
X = experience.reshape(-1, 1)
|
|
278
350
|
y = promotion
|
|
351
|
+
|
|
352
|
+
# Train Logistic Regression Model
|
|
279
353
|
model = LogisticRegression()
|
|
280
354
|
model.fit(X, y)
|
|
281
|
-
|
|
355
|
+
|
|
356
|
+
# Predicted Probability
|
|
282
357
|
probability = model.predict_proba(X)[:, 1]
|
|
283
|
-
|
|
358
|
+
|
|
359
|
+
# Decision Threshold 0.5
|
|
284
360
|
predicted = (probability >= 0.5).astype(int)
|
|
361
|
+
|
|
285
362
|
result = pd.DataFrame({
|
|
286
363
|
'Experience': experience,
|
|
287
364
|
'Actual': promotion,
|
|
@@ -289,39 +366,62 @@ result = pd.DataFrame({
|
|
|
289
366
|
'Predicted': predicted
|
|
290
367
|
})
|
|
291
368
|
print(result)
|
|
292
|
-
|
|
369
|
+
|
|
370
|
+
# Plot Sigmoid Curve
|
|
293
371
|
x_curve = np.linspace(1, 10, 100).reshape(-1, 1)
|
|
294
372
|
y_curve = model.predict_proba(x_curve)[:, 1]
|
|
295
|
-
|
|
296
|
-
plt.
|
|
297
|
-
plt.
|
|
373
|
+
|
|
374
|
+
plt.scatter(experience, promotion, color='red', label='Actual Data')
|
|
375
|
+
plt.plot(x_curve, y_curve, color='blue', label='Logistic Curve')
|
|
376
|
+
plt.axhline(0.5, linestyle='--', color='gray', label='Threshold (0.5)')
|
|
298
377
|
plt.xlabel("Years of Experience")
|
|
299
378
|
plt.ylabel("Probability of Promotion")
|
|
300
|
-
plt.title("Logistic Regression - Promotion")
|
|
379
|
+
plt.title("Logistic Regression - Promotion Prediction")
|
|
380
|
+
plt.legend()
|
|
301
381
|
plt.show()
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
382
|
+
--------------------------------------------------------------------------------
|
|
383
|
+
|
|
384
|
+
CONCLUSION:
|
|
385
|
+
Logistic Regression maps input features to probabilities between 0 and 1.
|
|
386
|
+
Probabilities >= 0.5 are classified as Promoted (1), otherwise Not Promoted (0).
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
================================================================================
|
|
390
|
+
Q6: LOAN APPROVAL PREDICTION USING LOGISTIC REGRESSION
|
|
391
|
+
================================================================================
|
|
392
|
+
|
|
393
|
+
AIM:
|
|
394
|
+
Predict loan approval probability based on monthly income using Logistic Regression.
|
|
395
|
+
|
|
396
|
+
PYTHON PROGRAM:
|
|
397
|
+
--------------------------------------------------------------------------------
|
|
307
398
|
import numpy as np
|
|
308
399
|
import pandas as pd
|
|
309
400
|
import matplotlib.pyplot as plt
|
|
310
401
|
from sklearn.linear_model import LogisticRegression
|
|
311
|
-
|
|
312
|
-
|
|
402
|
+
|
|
403
|
+
# Create Sample Dataset
|
|
404
|
+
income = np.array([15, 18, 20, 22, 25, 28, 30, 35, 40, 45])
|
|
405
|
+
loan_approved = np.array([0, 0, 0, 0, 1, 1, 1, 1, 1, 1])
|
|
406
|
+
|
|
313
407
|
df = pd.DataFrame({
|
|
314
408
|
'Income': income,
|
|
315
409
|
'Loan_Approved': loan_approved
|
|
316
410
|
})
|
|
411
|
+
|
|
317
412
|
X = income.reshape(-1, 1)
|
|
318
413
|
y = loan_approved
|
|
414
|
+
|
|
415
|
+
# Train Logistic Regression Model
|
|
319
416
|
model = LogisticRegression()
|
|
320
417
|
model.fit(X, y)
|
|
321
|
-
|
|
418
|
+
|
|
419
|
+
# Predicted Probability
|
|
322
420
|
probability = model.predict_proba(X)[:, 1]
|
|
323
|
-
|
|
421
|
+
|
|
422
|
+
# Decision Threshold 0.5
|
|
324
423
|
prediction = (probability >= 0.5).astype(int)
|
|
424
|
+
|
|
325
425
|
result = pd.DataFrame({
|
|
326
426
|
'Income': income,
|
|
327
427
|
'Actual': loan_approved,
|
|
@@ -329,19 +429,25 @@ result = pd.DataFrame({
|
|
|
329
429
|
'Prediction': prediction
|
|
330
430
|
})
|
|
331
431
|
print(result)
|
|
332
|
-
|
|
432
|
+
|
|
433
|
+
# Plot Logistic Curve
|
|
333
434
|
x_curve = np.linspace(15, 45, 100).reshape(-1, 1)
|
|
334
435
|
y_curve = model.predict_proba(x_curve)[:, 1]
|
|
335
|
-
|
|
336
|
-
plt.
|
|
337
|
-
plt.
|
|
436
|
+
|
|
437
|
+
plt.scatter(income, loan_approved, color='red', label='Actual Data')
|
|
438
|
+
plt.plot(x_curve, y_curve, color='blue', label='Logistic Curve')
|
|
439
|
+
plt.axhline(0.5, linestyle='--', color='gray', label='Threshold (0.5)')
|
|
338
440
|
plt.xlabel("Monthly Income")
|
|
339
441
|
plt.ylabel("Probability of Loan Approval")
|
|
340
442
|
plt.title("Logistic Regression - Loan Approval")
|
|
443
|
+
plt.legend()
|
|
341
444
|
plt.show()
|
|
342
|
-
|
|
445
|
+
|
|
446
|
+
# Decision Boundary Calculation
|
|
343
447
|
boundary = -model.intercept_[0] / model.coef_[0][0]
|
|
344
|
-
print("Decision Boundary:", boundary)
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
448
|
+
print("Decision Boundary (Income Threshold):", boundary)
|
|
449
|
+
--------------------------------------------------------------------------------
|
|
450
|
+
|
|
451
|
+
CONCLUSION:
|
|
452
|
+
The model calculates a decision boundary for monthly income above which loans
|
|
453
|
+
are classified as Approved (1) and below which they are classified as Rejected (0).
|