# ============================================================
# BOSTON HOUSING DATASET
# K-NEAREST NEIGHBORS (KNN)
# COMPLETE DATA PREPROCESSING + EVALUATION
# GOOGLE COLAB VERSION
# ============================================================
# ============================================================
# 1. IMPORT LIBRARIES
# ============================================================
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
import seaborn as sns
from sklearn.datasets import fetch_openml
from sklearn.model_selection import (
train_test_split,
cross_val_score
)
from sklearn.compose import ColumnTransformer
from sklearn.pipeline import Pipeline
from sklearn.impute import SimpleImputer
from sklearn.preprocessing import (
StandardScaler,
OneHotEncoder
)
from sklearn.neighbors import KNeighborsClassifier
from sklearn.metrics import (
accuracy_score,
precision_score,
recall_score,
f1_score,
confusion_matrix,
classification_report,
ConfusionMatrixDisplay,
roc_curve,
roc_auc_score
)
import warnings
warnings.filterwarnings("ignore")
# ============================================================
# 2. LOAD BOSTON HOUSING DATASET
# ============================================================
boston = fetch_openml(
name="boston",
version=1,
as_frame=True
)
df = boston.frame.copy()
print("Boston Housing Dataset Loaded Successfully!")
# ============================================================
# 3. DISPLAY FIRST 5 ROWS
# ============================================================
print("\nFirst 5 Rows:")
display(
df.head()
)
# ============================================================
# 4. DATASET SHAPE
# ============================================================
print("\nDataset Shape:")
print(
df.shape
)
# ============================================================
# 5. COLUMN NAMES
# ============================================================
print("\nColumn Names:")
print(
df.columns.tolist()
)
# ============================================================
# 6. DATA INFORMATION
# ============================================================
print("\nDataset Information:")
df.info()
# ============================================================
# 7. DATA TYPES
# ============================================================
print("\nData Types:")
print(
df.dtypes
)
# ============================================================
# 8. MISSING VALUES
# ============================================================
print("\nMissing Values:")
print(
df.isnull().sum()
)
# ============================================================
# 9. MISSING VALUE PERCENTAGE
# ============================================================
missing_percentage = (
df.isnull()
.mean()
.mul(100)
.sort_values(
ascending=False
)
)
print("\nMissing Value Percentage:")
print(
missing_percentage
)
# ============================================================
# 10. DUPLICATE VALUES
# ============================================================
print(
"\nNumber of Duplicate Rows:",
df.duplicated().sum()
)
# ============================================================
# 11. REMOVE DUPLICATES
# ============================================================
df = df.drop_duplicates()
print(
"\nShape After Duplicate Removal:"
)
print(
df.shape
)
# ============================================================
# 12. STATISTICAL SUMMARY
# ============================================================
print(
"\nStatistical Summary:"
)
display(
df.describe()
)
# ============================================================
# 13. TARGET VARIABLE
# ============================================================
target = "MEDV"
print(
"\nTarget Variable:",
target
)
# ============================================================
# 14. TARGET DISTRIBUTION
# ============================================================
plt.figure(
figsize=(9,5)
)
sns.histplot(
df[target],
kde=True
)
plt.title(
"Distribution of Boston House Prices"
)
plt.xlabel(
"MEDV"
)
plt.ylabel(
"Frequency"
)
plt.show()
# ============================================================
# 15. OUTLIER DETECTION
# ============================================================
numeric_columns = (
df.select_dtypes(
include=np.number
).columns
)
plt.figure(
figsize=(14,6)
)
sns.boxplot(
data=df[
numeric_columns
]
)
plt.title(
"Boxplot of Boston Housing Features"
)
plt.xticks(
rotation=45
)
plt.show()
# ============================================================
# 16. IQR OUTLIER ANALYSIS
# ============================================================
print(
"\n============================================"
)
print(
" IQR OUTLIER ANALYSIS"
)
print(
"============================================"
)
for column in numeric_columns:
Q1 = df[column].quantile(
0.25
)
Q3 = df[column].quantile(
0.75
)
IQR = Q3 - Q1
lower_bound = (
Q1 - 1.5 * IQR
)
upper_bound = (
Q3 + 1.5 * IQR
)
outliers = df[
(df[column] < lower_bound)
|
(df[column] > upper_bound)
]
print(
column,
"→",
len(outliers),
"outliers"
)
# ============================================================
# 17. OUTLIER CAPPING
# ============================================================
df_capped = df.copy()
for column in numeric_columns:
Q1 = df_capped[column].quantile(
0.25
)
Q3 = df_capped[column].quantile(
0.75
)
IQR = Q3 - Q1
lower_bound = (
Q1 - 1.5 * IQR
)
upper_bound = (
Q3 + 1.5 * IQR
)
df_capped[column] = (
df_capped[column]
.clip(
lower_bound,
upper_bound
)
)
print(
"\nIQR Outlier Capping Completed!"
)
# ============================================================
# 18. CORRELATION MATRIX
# ============================================================
correlation = df.corr(
numeric_only=True
)
print(
"\nCorrelation with MEDV:"
)
print(
correlation[target]
.sort_values(
ascending=False
)
)
# ============================================================
# 19. CORRELATION HEATMAP
# ============================================================
plt.figure(
figsize=(12,9)
)
sns.heatmap(
correlation,
annot=True,
fmt=".2f",
cmap="coolwarm"
)
plt.title(
"Boston Housing Correlation Matrix"
)
plt.show()
# ============================================================
# 20. CONVERT MEDV INTO CLASS
# ============================================================
#
# Original MEDV is continuous.
#
# KNN classification requires classes.
#
# 0 = Low Price
# 1 = High Price
#
# Median MEDV is used as threshold.
# ============================================================
median_price = df[target].median()
print(
"\nMedian House Price:",
median_price
)
df["Price_Class"] = (
df[target] >= median_price
).astype(int)
# ============================================================
# 21. DISPLAY CLASS DISTRIBUTION
# ============================================================
print(
"\nPrice Class Distribution:"
)
print(
df["Price_Class"].value_counts()
)
# ============================================================
# 22. CLASS DISTRIBUTION GRAPH
# ============================================================
plt.figure(
figsize=(7,5)
)
sns.countplot(
x="Price_Class",
data=df
)
plt.title(
"Low Price vs High Price Houses"
)
plt.xlabel(
"Price Class"
)
plt.ylabel(
"Number of Houses"
)
plt.xticks(
[0,1],
[
"Low Price",
"High Price"
]
)
plt.show()
# ============================================================
# 23. FEATURES AND TARGET
# ============================================================
X = df.drop(
columns=[
"MEDV",
"Price_Class"
]
)
y = df["Price_Class"]
print(
"\nFeature Shape:",
X.shape
)
print(
"Target Shape:",
y.shape
)
# ============================================================
# 24. IDENTIFY NUMERICAL FEATURES
# ============================================================
numeric_features = (
X.select_dtypes(
include=np.number
)
.columns
.tolist()
)
categorical_features = (
X.select_dtypes(
exclude=np.number
)
.columns
.tolist()
)
print(
"\nNumerical Features:"
)
print(
numeric_features
)
print(
"\nCategorical Features:"
)
print(
categorical_features
)
# ============================================================
# 25. TRAIN TEST SPLIT
# ============================================================
X_train, X_test, y_train, y_test = (
train_test_split(
X,
y,
test_size=0.20,
random_state=42,
stratify=y
)
)
print(
"\nTraining Shape:",
X_train.shape
)
print(
"Testing Shape:",
X_test.shape
)
# ============================================================
# 26. NUMERICAL PREPROCESSING
# ============================================================
numeric_pipeline = Pipeline(
steps=[
(
"imputer",
SimpleImputer(
strategy="median"
)
),
(
"scaler",
StandardScaler()
)
]
)
# ============================================================
# 27. CATEGORICAL PREPROCESSING
# ============================================================
categorical_pipeline = Pipeline(
steps=[
(
"imputer",
SimpleImputer(
strategy="most_frequent"
)
),
(
"encoder",
OneHotEncoder(
handle_unknown="ignore"
)
)
]
)
# ============================================================
# 28. COMBINE PREPROCESSING
# ============================================================
preprocessor = ColumnTransformer(
transformers=[
(
"numeric",
numeric_pipeline,
numeric_features
),
(
"categorical",
categorical_pipeline,
categorical_features
)
]
)
# ============================================================
# 29. KNN MODEL
# ============================================================
knn_model = Pipeline(
steps=[
(
"preprocessing",
preprocessor
),
(
"model",
KNeighborsClassifier(
n_neighbors=5
)
)
]
)
# ============================================================
# 30. TRAIN KNN MODEL
# ============================================================
knn_model.fit(
X_train,
y_train
)
print(
"\nKNN Model Trained Successfully!"
)
# ============================================================
# 31. PREDICTION
# ============================================================
y_pred = knn_model.predict(
X_test
)
# ============================================================
# 32. PREDICT PROBABILITY
# ============================================================
y_probability = (
knn_model
.predict_proba(
X_test
)[:, 1]
)
# ============================================================
# 33. ACCURACY
# ============================================================
accuracy = accuracy_score(
y_test,
y_pred
)
# ============================================================
# 34. PRECISION
# ============================================================
precision = precision_score(
y_test,
y_pred,
zero_division=0
)
# ============================================================
# 35. RECALL
# ============================================================
recall = recall_score(
y_test,
y_pred,
zero_division=0
)
# ============================================================
# 36. F1 SCORE
# ============================================================
f1 = f1_score(
y_test,
y_pred,
zero_division=0
)
# ============================================================
# 37. ROC-AUC
# ============================================================
roc_auc = roc_auc_score(
y_test,
y_probability
)
# ============================================================
# 38. DISPLAY MODEL RESULTS
# ============================================================
print(
"\n============================================"
)
print(
" KNN RESULTS"
)
print(
"============================================"
)
print(
"Accuracy :",
round(accuracy, 4)
)
print(
"Precision :",
round(precision, 4)
)
print(
"Recall :",
round(recall, 4)
)
print(
"F1 Score :",
round(f1, 4)
)
print(
"ROC-AUC :",
round(roc_auc, 4)
)
print(
"============================================"
)
# ============================================================
# 39. CONFUSION MATRIX
# ============================================================
cm = confusion_matrix(
y_test,
y_pred
)
print(
"\nConfusion Matrix:"
)
print(
cm
)
# ============================================================
# 40. TRUE NEGATIVE / FALSE POSITIVE /
# FALSE NEGATIVE / TRUE POSITIVE
# ============================================================
TN, FP, FN, TP = cm.ravel()
print(
"\nTrue Negative (TN):",
TN
)
print(
"False Positive (FP):",
FP
)
print(
"False Negative (FN):",
FN
)
print(
"True Positive (TP):",
TP
)
# ============================================================
# 41. CONFUSION MATRIX GRAPH
# ============================================================
plt.figure(
figsize=(7,6)
)
sns.heatmap(
cm,
annot=True,
fmt="d",
cmap="Blues",
xticklabels=[
"Predicted Low",
"Predicted High"
],
yticklabels=[
"Actual Low",
"Actual High"
]
)
plt.title(
"KNN Confusion Matrix"
)
plt.xlabel(
"Predicted Class"
)
plt.ylabel(
"Actual Class"
)
plt.show()
# ============================================================
# 42. CLASSIFICATION REPORT
# ============================================================
print(
"\n============================================"
)
print(
" CLASSIFICATION REPORT"
)
print(
"============================================"
)
print(
classification_report(
y_test,
y_pred,
target_names=[
"Low Price",
"High Price"
],
zero_division=0
)
)
# ============================================================
# 43. PERFORMANCE TABLE
# ============================================================
metrics_table = pd.DataFrame({
"Metric": [
"Accuracy",
"Precision",
"Recall",
"F1 Score",
"ROC-AUC"
],
"Score": [
accuracy,
precision,
recall,
f1,
roc_auc
]
})
metrics_table["Score"] = (
metrics_table["Score"]
.round(4)
)
print(
"\nKNN Performance Metrics:"
)
display(
metrics_table
)
# ============================================================
# 44. CROSS VALIDATION
# ============================================================
cv_scores = cross_val_score(
knn_model,
X,
y,
cv=5,
scoring="accuracy"
)
print(
"\n5-Fold Cross Validation Accuracy:"
)
print(
cv_scores
)
print(
"\nMean Cross Validation Accuracy:"
)
print(
round(
cv_scores.mean(),
4
)
)
# ============================================================
# 45. TEST DIFFERENT K VALUES
# ============================================================
k_values = range(
1,
21
)
k_accuracy = []
for k in k_values:
temp_knn = Pipeline(
steps=[
(
"preprocessing",
preprocessor
),
(
"model",
KNeighborsClassifier(
n_neighbors=k
)
)
]
)
temp_knn.fit(
X_train,
y_train
)
temp_prediction = temp_knn.predict(
X_test
)
score = accuracy_score(
y_test,
temp_prediction
)
k_accuracy.append(
score
)
# ============================================================
# 46. K VALUE GRAPH
# ============================================================
plt.figure(
figsize=(10,6)
)
plt.plot(
list(k_values),
k_accuracy,
marker="o"
)
plt.xlabel(
"Number of Neighbors (K)"
)
plt.ylabel(
"Accuracy"
)
plt.title(
"KNN Accuracy for Different K Values"
)
plt.xticks(
list(k_values)
)
plt.grid(
True
)
plt.show()
# ============================================================
# 47. BEST K VALUE
# ============================================================
best_k_index = np.argmax(
k_accuracy
)
best_k = list(
k_values
)[best_k_index]
best_k_accuracy = k_accuracy[
best_k_index
]
print(
"\n============================================"
)
print(
" BEST K VALUE"
)
print(
"============================================"
)
print(
"Best K:",
best_k
)
print(
"Accuracy:",
round(
best_k_accuracy,
4
)
)
# ============================================================
# 48. TRAIN FINAL KNN USING BEST K
# ============================================================
final_knn = Pipeline(
steps=[
(
"preprocessing",
preprocessor
),
(
"model",
KNeighborsClassifier(
n_neighbors=best_k
)
)
]
)
final_knn.fit(
X_train,
y_train
)
final_prediction = final_knn.predict(
X_test
)
final_probability = (
final_knn
.predict_proba(
X_test
)[:, 1]
)
# ============================================================
# 49. FINAL METRICS
# ============================================================
final_accuracy = accuracy_score(
y_test,
final_prediction
)
final_precision = precision_score(
y_test,
final_prediction,
zero_division=0
)
final_recall = recall_score(
y_test,
final_prediction,
zero_division=0
)
final_f1 = f1_score(
y_test,
final_prediction,
zero_division=0
)
final_auc = roc_auc_score(
y_test,
final_probability
)
print(
"\n============================================"
)
print(
" FINAL KNN MODEL RESULTS"
)
print(
"============================================"
)
print(
"Best K :",
best_k
)
print(
"Accuracy :",
round(
final_accuracy,
4
)
)
print(
"Precision :",
round(
final_precision,
4
)
)
print(
"Recall :",
round(
final_recall,
4
)
)
print(
"F1 Score :",
round(
final_f1,
4
)
)
print(
"ROC-AUC :",
round(
final_auc,
4
)
)
print(
"============================================"
)
# ============================================================
# 50. FINAL CONFUSION MATRIX
# ============================================================
final_cm = confusion_matrix(
y_test,
final_prediction
)
plt.figure(
figsize=(7,6)
)
sns.heatmap(
final_cm,
annot=True,
fmt="d",
cmap="Blues",
xticklabels=[
"Predicted Low",
"Predicted High"
],
yticklabels=[
"Actual Low",
"Actual High"
]
)
plt.title(
f"Final KNN Confusion Matrix (K={best_k})"
)
plt.xlabel(
"Predicted Class"
)
plt.ylabel(
"Actual Class"
)
plt.show()
# ============================================================
# 51. FINAL CLASSIFICATION REPORT
# ============================================================
print(
"\nFinal Classification Report:"
)
print(
classification_report(
y_test,
final_prediction,
target_names=[
"Low Price",
"High Price"
],
zero_division=0
)
)
# ============================================================
# 52. ROC CURVE
# ============================================================
fpr, tpr, thresholds = roc_curve(
y_test,
final_probability
)
plt.figure(
figsize=(8,6)
)
plt.plot(
fpr,
tpr,
label=f"KNN ROC-AUC = {final_auc:.4f}"
)
plt.plot(
[0,1],
[0,1],
linestyle="--"
)
plt.xlabel(
"False Positive Rate"
)
plt.ylabel(
"True Positive Rate"
)
plt.title(
"KNN ROC Curve"
)
plt.legend()
plt.show()
# ============================================================
# 53. ACTUAL VS PREDICTED
# ============================================================
prediction_table = pd.DataFrame({
"Actual Class":
y_test.values,
"Predicted Class":
final_prediction,
"High Price Probability":
final_probability
})
prediction_table["Actual Class"] = (
prediction_table[
"Actual Class"
].map({
0: "Low Price",
1: "High Price"
})
)
prediction_table["Predicted Class"] = (
prediction_table[
"Predicted Class"
].map({
0: "Low Price",
1: "High Price"
})
)
prediction_table[
"High Price Probability"
] = (
prediction_table[
"High Price Probability"
].round(4)
)
print(
"\nActual vs Predicted:"
)
display(
prediction_table.head(20)
)
# ============================================================
# 54. NEW HOUSE CLASSIFICATION
# ============================================================
#
# This version avoids the X.mean() categorical error.
# Numerical columns → median
# Categorical columns → mode
# ============================================================
new_house = pd.DataFrame(
index=[0],
columns=X.columns
)
# Numerical features
for column in numeric_features:
new_house.loc[
0,
column
] = X[
column
].median()
# Categorical features
for column in categorical_features:
new_house.loc[
0,
column
] = X[
column
].mode()[0]
# Restore category dtype if necessary
for column in categorical_features:
if str(
X[column].dtype
) == "category":
new_house[column] = pd.Categorical(
new_house[column],
categories=X[
column
].cat.categories
)
# ============================================================
# 55. PREDICT NEW HOUSE
# ============================================================
new_prediction = final_knn.predict(
new_house
)[0]
new_probability = final_knn.predict_proba(
new_house
)[0][1]
if new_prediction == 1:
new_result = "High Price"
else:
new_result = "Low Price"
print(
"\n============================================"
)
print(
" NEW HOUSE CLASSIFICATION"
)
print(
"============================================"
)
print(
"Predicted Class:",
new_result
)
print(
"Probability of High Price:",
round(
new_probability,
4
)
)
print(
"Probability of Low Price:",
round(
1 - new_probability,
4
)
)
# ============================================================
# 56. FINAL KNN SUMMARY
# ============================================================
print(
"\n============================================"
)
print(
" KNN SUMMARY"
)
print(
"============================================"
)
print(
"Algorithm: K-Nearest Neighbors"
)
print(
"Number of Neighbors:",
best_k
)
print(
"Accuracy:",
round(
final_accuracy,
4
)
)
print(
"Precision:",
round(
final_precision,
4
)
)
print(
"Recall:",
round(
final_recall,
4
)
)
print(
"F1 Score:",
round(
final_f1,
4
)
)
print(
"ROC-AUC:",
round(
final_auc,
4
)
)
print(
"Cross Validation Mean:",
round(
cv_scores.mean(),
4
)
)
print(
"============================================"
)
# ============================================================
# 57. KNN THEORY
# ============================================================
print("""
============================================================
K-NEAREST NEIGHBORS (KNN) - THEORY
============================================================
KNN is a supervised machine learning algorithm used for
classification and regression.
For classification, KNN looks at the nearest K observations
and assigns the class that receives the majority vote.
Example:
K = 5
Nearest neighbors:
Neighbor 1 → High Price
Neighbor 2 → High Price
Neighbor 3 → Low Price
Neighbor 4 → High Price
Neighbor 5 → Low Price
High Price = 3
Low Price = 2
Prediction = High Price
============================================================
WHY STANDARDIZATION IS IMPORTANT?
KNN uses distance calculations.
Features with large numerical values can dominate the distance.
StandardScaler converts features approximately to:
Mean = 0
Standard Deviation = 1
Therefore, scaling is particularly important for KNN.
============================================================
K VALUE
Small K:
- Sensitive to noise
- More flexible
- Can overfit
Large K:
- Smoother decision boundary
- Less sensitive to noise
- Can underfit
The code tests K = 1 to K = 20.
============================================================
BOSTON HOUSING CLASSIFICATION
Original target:
MEDV = Continuous house price
Converted target:
0 = Low Price
1 = High Price
Median MEDV is used as the classification threshold.
============================================================
""")
No comments:
Post a Comment