# ============================================================
# BOSTON HOUSING DATASET
# K-MEANS CLUSTERING
# COMPLETE DATA PREPROCESSING + CLUSTER ANALYSIS
# GOOGLE COLAB VERSION
# ============================================================
# ============================================================
# 1. IMPORT LIBRARIES
# ============================================================
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
import seaborn as sns
from sklearn.datasets import fetch_openml
from sklearn.model_selection import train_test_split
from sklearn.pipeline import Pipeline
from sklearn.compose import ColumnTransformer
from sklearn.impute import SimpleImputer
from sklearn.preprocessing import (
StandardScaler,
OneHotEncoder
)
from sklearn.cluster import KMeans
from sklearn.metrics import (
silhouette_score,
adjusted_rand_score,
normalized_mutual_info_score
)
from sklearn.decomposition import PCA
import warnings
warnings.filterwarnings("ignore")
# ============================================================
# 2. LOAD BOSTON HOUSING DATASET
# ============================================================
boston = fetch_openml(
name="boston",
version=1,
as_frame=True
)
df = boston.frame.copy()
print("Boston Housing Dataset Loaded Successfully!")
# ============================================================
# 3. DISPLAY FIRST 5 ROWS
# ============================================================
print("\nFirst 5 Rows:")
display(
df.head()
)
# ============================================================
# 4. DATASET SHAPE
# ============================================================
print("\nDataset Shape:")
print(
df.shape
)
# ============================================================
# 5. COLUMN NAMES
# ============================================================
print("\nColumn Names:")
print(
df.columns.tolist()
)
# ============================================================
# 6. DATA INFORMATION
# ============================================================
print("\nDataset Information:")
df.info()
# ============================================================
# 7. DATA TYPES
# ============================================================
print("\nData Types:")
print(
df.dtypes
)
# ============================================================
# 8. MISSING VALUES
# ============================================================
print("\nMissing Values:")
print(
df.isnull().sum()
)
# ============================================================
# 9. MISSING VALUE PERCENTAGE
# ============================================================
missing_percentage = (
df.isnull()
.mean()
.mul(100)
.sort_values(
ascending=False
)
)
print(
"\nMissing Value Percentage:"
)
print(
missing_percentage
)
# ============================================================
# 10. DUPLICATE RECORDS
# ============================================================
duplicate_count = df.duplicated().sum()
print(
"\nNumber of Duplicate Rows:",
duplicate_count
)
# ============================================================
# 11. REMOVE DUPLICATES
# ============================================================
df = df.drop_duplicates()
print(
"\nShape After Duplicate Removal:"
)
print(
df.shape
)
# ============================================================
# 12. STATISTICAL SUMMARY
# ============================================================
print(
"\nStatistical Summary:"
)
display(
df.describe()
)
# ============================================================
# 13. UNIQUE VALUES
# ============================================================
print(
"\nNumber of Unique Values:"
)
print(
df.nunique()
)
# ============================================================
# 14. TARGET / REFERENCE VARIABLE
# ============================================================
target = "MEDV"
print(
"\nReference Variable:",
target
)
# ============================================================
# 15. MEDV DISTRIBUTION
# ============================================================
plt.figure(
figsize=(9,5)
)
sns.histplot(
df[target],
kde=True
)
plt.title(
"Distribution of Boston House Prices"
)
plt.xlabel(
"MEDV"
)
plt.ylabel(
"Frequency"
)
plt.show()
# ============================================================
# 16. IDENTIFY NUMERICAL COLUMNS
# ============================================================
numeric_columns = (
df.select_dtypes(
include=np.number
).columns.tolist()
)
print(
"\nNumerical Columns:"
)
print(
numeric_columns
)
# ============================================================
# 17. OUTLIER DETECTION USING IQR
# ============================================================
print(
"\n============================================"
)
print(
" IQR OUTLIER ANALYSIS"
)
print(
"============================================"
)
outlier_summary = []
for column in numeric_columns:
Q1 = df[column].quantile(
0.25
)
Q3 = df[column].quantile(
0.75
)
IQR = Q3 - Q1
lower_bound = (
Q1 - 1.5 * IQR
)
upper_bound = (
Q3 + 1.5 * IQR
)
outlier_count = (
(
(df[column] < lower_bound)
|
(df[column] > upper_bound)
)
.sum()
)
outlier_summary.append({
"Feature": column,
"Q1": Q1,
"Q3": Q3,
"IQR": IQR,
"Lower Bound": lower_bound,
"Upper Bound": upper_bound,
"Outliers": outlier_count
})
outlier_table = pd.DataFrame(
outlier_summary
)
display(
outlier_table
)
# ============================================================
# 18. BOX PLOT BEFORE OUTLIER TREATMENT
# ============================================================
plt.figure(
figsize=(15,6)
)
sns.boxplot(
data=df[numeric_columns]
)
plt.title(
"Boston Housing Features - Before Outlier Treatment"
)
plt.xticks(
rotation=45
)
plt.show()
# ============================================================
# 19. IQR OUTLIER CAPPING
# ============================================================
#
# Instead of deleting observations, values outside the
# IQR limits are capped at the lower/upper boundaries.
#
# MEDV is not used for clustering, so it is left untouched.
# ============================================================
df_cluster = df.copy()
feature_columns = [
column
for column in numeric_columns
if column != target
]
for column in feature_columns:
Q1 = df_cluster[column].quantile(
0.25
)
Q3 = df_cluster[column].quantile(
0.75
)
IQR = Q3 - Q1
lower_bound = (
Q1 - 1.5 * IQR
)
upper_bound = (
Q3 + 1.5 * IQR
)
df_cluster[column] = (
df_cluster[column]
.clip(
lower_bound,
upper_bound
)
)
print(
"\nIQR Outlier Capping Completed!"
)
# ============================================================
# 20. BOX PLOT AFTER OUTLIER TREATMENT
# ============================================================
plt.figure(
figsize=(15,6)
)
sns.boxplot(
data=df_cluster[feature_columns]
)
plt.title(
"Boston Housing Features - After IQR Capping"
)
plt.xticks(
rotation=45
)
plt.show()
# ============================================================
# 21. CHECK MISSING VALUES AFTER PREPROCESSING
# ============================================================
print(
"\nMissing Values After Preprocessing:"
)
print(
df_cluster[feature_columns]
.isnull()
.sum()
)
# ============================================================
# 22. PREPARE FEATURES FOR K-MEANS
# ============================================================
#
# IMPORTANT:
#
# MEDV is NOT used to create clusters.
#
# K-Means will find groups using the housing features.
#
# MEDV is retained separately only for later interpretation.
# ============================================================
X = df_cluster.drop(
columns=[target]
)
medv_reference = df_cluster[target].copy()
print(
"\nK-Means Feature Shape:"
)
print(
X.shape
)
# ============================================================
# 23. IDENTIFY NUMERICAL AND CATEGORICAL FEATURES
# ============================================================
numeric_features = (
X.select_dtypes(
include=np.number
)
.columns
.tolist()
)
categorical_features = (
X.select_dtypes(
exclude=np.number
)
.columns
.tolist()
)
print(
"\nNumerical Features:"
)
print(
numeric_features
)
print(
"\nCategorical Features:"
)
print(
categorical_features
)
# ============================================================
# 24. NUMERICAL PREPROCESSING
# ============================================================
numeric_pipeline = Pipeline(
steps=[
(
"imputer",
SimpleImputer(
strategy="median"
)
),
(
"scaler",
StandardScaler()
)
]
)
# ============================================================
# 25. CATEGORICAL PREPROCESSING
# ============================================================
categorical_pipeline = Pipeline(
steps=[
(
"imputer",
SimpleImputer(
strategy="most_frequent"
)
),
(
"encoder",
OneHotEncoder(
handle_unknown="ignore"
)
)
]
)
# ============================================================
# 26. COMBINE PREPROCESSING
# ============================================================
preprocessor = ColumnTransformer(
transformers=[
(
"numeric",
numeric_pipeline,
numeric_features
),
(
"categorical",
categorical_pipeline,
categorical_features
)
]
)
# ============================================================
# 27. TRANSFORM DATA
# ============================================================
X_processed = preprocessor.fit_transform(
X
)
# Convert sparse matrix to dense matrix if necessary
if hasattr(
X_processed,
"toarray"
):
X_processed = X_processed.toarray()
print(
"\nProcessed Data Shape:"
)
print(
X_processed.shape
)
# ============================================================
# 28. CHECK PROCESSED DATA
# ============================================================
print(
"\nFirst 5 Processed Rows:"
)
display(
pd.DataFrame(
X_processed
).head()
)
# ============================================================
# 29. ELBOW METHOD
# ============================================================
#
# Inertia measures the sum of squared distances between
# observations and their assigned cluster centers.
#
# We calculate K = 2 to 10.
# ============================================================
k_values = range(
2,
11
)
inertia_values = []
for k in k_values:
kmeans = KMeans(
n_clusters=k,
random_state=42,
n_init=10
)
kmeans.fit(
X_processed
)
inertia_values.append(
kmeans.inertia_
)
# ============================================================
# 30. ELBOW GRAPH
# ============================================================
plt.figure(
figsize=(10,6)
)
plt.plot(
list(k_values),
inertia_values,
marker="o"
)
plt.xlabel(
"Number of Clusters (K)"
)
plt.ylabel(
"Inertia"
)
plt.title(
"Elbow Method for K-Means"
)
plt.xticks(
list(k_values)
)
plt.grid(
True
)
plt.show()
# ============================================================
# 31. SILHOUETTE SCORE
# ============================================================
#
# Silhouette score measures how well observations fit
# within their own cluster compared with other clusters.
# ============================================================
silhouette_values = []
for k in k_values:
kmeans = KMeans(
n_clusters=k,
random_state=42,
n_init=10
)
labels = kmeans.fit_predict(
X_processed
)
score = silhouette_score(
X_processed,
labels
)
silhouette_values.append(
score
)
# ============================================================
# 32. SILHOUETTE SCORE TABLE
# ============================================================
silhouette_table = pd.DataFrame({
"K": list(k_values),
"Silhouette Score":
silhouette_values
})
silhouette_table[
"Silhouette Score"
] = (
silhouette_table[
"Silhouette Score"
].round(4)
)
print(
"\nSilhouette Scores:"
)
display(
silhouette_table
)
# ============================================================
# 33. SILHOUETTE GRAPH
# ============================================================
plt.figure(
figsize=(10,6)
)
plt.plot(
list(k_values),
silhouette_values,
marker="o"
)
plt.xlabel(
"Number of Clusters (K)"
)
plt.ylabel(
"Silhouette Score"
)
plt.title(
"Silhouette Score for Different K Values"
)
plt.xticks(
list(k_values)
)
plt.grid(
True
)
plt.show()
# ============================================================
# 34. SELECT BEST K USING SILHOUETTE SCORE
# ============================================================
best_k_index = np.argmax(
silhouette_values
)
best_k = list(
k_values
)[best_k_index]
best_silhouette = silhouette_values[
best_k_index
]
print(
"\n============================================"
)
print(
" BEST K VALUE"
)
print(
"============================================"
)
print(
"Best K:",
best_k
)
print(
"Silhouette Score:",
round(
best_silhouette,
4
)
)
# ============================================================
# 35. CREATE FINAL K-MEANS MODEL
# ============================================================
kmeans_final = KMeans(
n_clusters=best_k,
random_state=42,
n_init=10
)
# ============================================================
# 36. FIT K-MEANS
# ============================================================
cluster_labels = kmeans_final.fit_predict(
X_processed
)
print(
"\nK-Means Clustering Completed!"
)
# ============================================================
# 37. ADD CLUSTER LABELS TO DATASET
# ============================================================
df_cluster["Cluster"] = cluster_labels
# ============================================================
# 38. CLUSTER COUNTS
# ============================================================
cluster_counts = (
df_cluster["Cluster"]
.value_counts()
.sort_index()
)
print(
"\nNumber of Houses in Each Cluster:"
)
print(
cluster_counts
)
# ============================================================
# 39. CLUSTER DISTRIBUTION GRAPH
# ============================================================
plt.figure(
figsize=(9,6)
)
sns.countplot(
x="Cluster",
data=df_cluster,
order=sorted(
df_cluster["Cluster"].unique()
)
)
plt.title(
"Number of Houses in Each K-Means Cluster"
)
plt.xlabel(
"Cluster"
)
plt.ylabel(
"Number of Houses"
)
plt.show()
# ============================================================
# 40. FINAL SILHOUETTE SCORE
# ============================================================
final_silhouette = silhouette_score(
X_processed,
cluster_labels
)
print(
"\nFinal Silhouette Score:"
)
print(
round(
final_silhouette,
4
)
)
# ============================================================
# 41. CLUSTER CENTERS
# ============================================================
cluster_centers = (
pd.DataFrame(
kmeans_final.cluster_centers_
)
)
print(
"\nCluster Centers:"
)
display(
cluster_centers
)
# ============================================================
# 42. CLUSTER-WISE HOUSE PRICE ANALYSIS
# ============================================================
#
# MEDV was NOT used to create the clusters.
#
# We now use MEDV only to understand what the clusters
# represent after clustering.
# ============================================================
cluster_price_summary = (
df_cluster
.groupby("Cluster")[target]
.agg([
"count",
"mean",
"median",
"min",
"max"
])
.round(2)
)
print(
"\nCluster-Wise House Price Summary:"
)
display(
cluster_price_summary
)
# ============================================================
# 43. CLUSTER-WISE FEATURE MEANS
# ============================================================
cluster_feature_summary = (
df_cluster
.groupby("Cluster")[
feature_columns
]
.mean()
.round(2)
)
print(
"\nCluster-Wise Feature Means:"
)
display(
cluster_feature_summary
)
# ============================================================
# 44. HEATMAP OF CLUSTER FEATURE MEANS
# ============================================================
plt.figure(
figsize=(15,7)
)
sns.heatmap(
cluster_feature_summary,
annot=True,
fmt=".2f",
cmap="coolwarm"
)
plt.title(
"Cluster-Wise Feature Means"
)
plt.xlabel(
"Features"
)
plt.ylabel(
"Cluster"
)
plt.show()
# ============================================================
# 45. CLUSTER VS MEDV
# ============================================================
plt.figure(
figsize=(10,6)
)
sns.boxplot(
x="Cluster",
y=target,
data=df_cluster
)
plt.title(
"House Price Distribution by K-Means Cluster"
)
plt.xlabel(
"Cluster"
)
plt.ylabel(
"MEDV"
)
plt.show()
# ============================================================
# 46. CLUSTER VS MEDV BAR CHART
# ============================================================
cluster_price_mean = (
df_cluster
.groupby("Cluster")[target]
.mean()
)
plt.figure(
figsize=(9,6)
)
plt.bar(
cluster_price_mean.index,
cluster_price_mean.values
)
plt.xlabel(
"Cluster"
)
plt.ylabel(
"Average MEDV"
)
plt.title(
"Average House Price by Cluster"
)
plt.show()
# ============================================================
# 47. PCA FOR VISUALIZATION
# ============================================================
#
# K-Means may use many features.
#
# PCA reduces the processed data to two dimensions
# so we can visualize the clusters.
# ============================================================
pca = PCA(
n_components=2
)
X_pca = pca.fit_transform(
X_processed
)
print(
"\nPCA Explained Variance:"
)
print(
pca.explained_variance_ratio_
)
print(
"\nTotal Variance Explained:"
)
print(
round(
pca.explained_variance_ratio_.sum(),
4
)
)
# ============================================================
# 48. CREATE PCA DATAFRAME
# ============================================================
pca_df = pd.DataFrame({
"PC1": X_pca[:, 0],
"PC2": X_pca[:, 1],
"Cluster": cluster_labels
})
# ============================================================
# 49. PCA CLUSTER VISUALIZATION
# ============================================================
plt.figure(
figsize=(10,7)
)
sns.scatterplot(
data=pca_df,
x="PC1",
y="PC2",
hue="Cluster",
palette="tab10",
s=70
)
plt.title(
"K-Means Clusters Visualized Using PCA"
)
plt.xlabel(
"Principal Component 1"
)
plt.ylabel(
"Principal Component 2"
)
plt.legend(
title="Cluster"
)
plt.show()
# ============================================================
# 50. PCA WITH CLUSTER CENTERS
# ============================================================
centers_pca = pca.transform(
kmeans_final.cluster_centers_
)
plt.figure(
figsize=(10,7)
)
sns.scatterplot(
data=pca_df,
x="PC1",
y="PC2",
hue="Cluster",
palette="tab10",
s=60
)
plt.scatter(
centers_pca[:, 0],
centers_pca[:, 1],
marker="X",
s=250,
color="black",
label="Cluster Centers"
)
plt.title(
"K-Means Clusters and Cluster Centers"
)
plt.xlabel(
"Principal Component 1"
)
plt.ylabel(
"Principal Component 2"
)
plt.legend()
plt.show()
# ============================================================
# 51. OPTIONAL: CREATE LOW/HIGH PRICE REFERENCE
# ============================================================
#
# This is NOT used for training K-Means.
#
# It is only used to examine how the unsupervised clusters
# correspond to the known MEDV-based categories.
# ============================================================
median_price = df_cluster[target].median()
df_cluster["Price_Class"] = (
df_cluster[target] >= median_price
).astype(int)
df_cluster["Price_Class_Name"] = (
df_cluster["Price_Class"]
.map({
0: "Low Price",
1: "High Price"
})
)
# ============================================================
# 52. CLUSTER VS PRICE CLASS
# ============================================================
cluster_price_class = pd.crosstab(
df_cluster["Cluster"],
df_cluster["Price_Class_Name"]
)
print(
"\nCluster vs Price Class:"
)
display(
cluster_price_class
)
# ============================================================
# 53. CLUSTER VS PRICE CLASS GRAPH
# ============================================================
cluster_price_class.plot(
kind="bar",
figsize=(10,6)
)
plt.title(
"Price Class Distribution Within Each Cluster"
)
plt.xlabel(
"Cluster"
)
plt.ylabel(
"Number of Houses"
)
plt.xticks(
rotation=0
)
plt.show()
# ============================================================
# 54. OPTIONAL UNSUPERVISED AGREEMENT METRICS
# ============================================================
#
# These metrics are NOT training metrics.
#
# They compare the discovered clusters against the
# MEDV-based reference classes only for educational analysis.
# ============================================================
ari = adjusted_rand_score(
df_cluster["Price_Class"],
cluster_labels
)
nmi = normalized_mutual_info_score(
df_cluster["Price_Class"],
cluster_labels
)
print(
"\n============================================"
)
print(
" OPTIONAL CLUSTER AGREEMENT"
)
print(
"============================================"
)
print(
"Adjusted Rand Index:",
round(
ari,
4
)
)
print(
"Normalized Mutual Information:",
round(
nmi,
4
)
)
# ============================================================
# 55. SAVE FINAL CLUSTERED DATA
# ============================================================
final_clustered_data = df_cluster.copy()
print(
"\nFinal Clustered Dataset:"
)
display(
final_clustered_data.head()
)
# ============================================================
# 56. FINAL SUMMARY TABLE
# ============================================================
summary_table = pd.DataFrame({
"Item": [
"Algorithm",
"Number of Clusters",
"Silhouette Score",
"Number of Features",
"Number of Houses",
"PCA Components"
],
"Value": [
"K-Means",
best_k,
round(
final_silhouette,
4
),
len(feature_columns),
len(df_cluster),
2
]
})
print(
"\n============================================"
)
print(
" K-MEANS SUMMARY"
)
print(
"============================================"
)
display(
summary_table
)
# ============================================================
# 57. CLUSTER INTERPRETATION
# ============================================================
print("""
============================================================
K-MEANS CLUSTERING - THEORY
============================================================
K-Means is an UNSUPERVISED MACHINE LEARNING algorithm.
Unlike Linear Regression, Logistic Regression and KNN
classification, K-Means does not require a target variable
during clustering.
The algorithm groups similar observations into K clusters.
============================================================
HOW K-MEANS WORKS
============================================================
Step 1:
Choose the number of clusters K.
Step 2:
Randomly initialize K cluster centers.
Step 3:
Calculate the distance between each observation
and every cluster center.
Step 4:
Assign each observation to the nearest cluster.
Step 5:
Recalculate the cluster centers.
Step 6:
Repeat the process until the cluster assignments
stabilize.
============================================================
ELBOW METHOD
============================================================
The Elbow Method calculates:
INERTIA
Inertia represents the within-cluster sum of squared
distances.
As K increases, inertia normally decreases.
The point where the reduction begins to slow down
is called the "elbow".
============================================================
SILHOUETTE SCORE
============================================================
Silhouette Score measures how well observations fit
their assigned clusters.
Higher values generally indicate better separation
and cohesion.
The score ranges approximately from:
-1 to +1
A value closer to +1 indicates better-defined clusters.
============================================================
IMPORTANT BOSTON HOUSING POINT
============================================================
MEDV was NOT used to create the clusters.
K-Means used the housing characteristics/features.
MEDV was used AFTER clustering only to understand
the characteristics of the resulting clusters.
Therefore this is an UNSUPERVISED learning experiment.
============================================================
PREPROCESSING USED
============================================================
1. Missing-value checking
2. Missing-value imputation
3. Duplicate detection
4. Duplicate removal
5. IQR outlier detection
6. IQR outlier capping
7. Feature scaling
8. Categorical encoding
9. PCA visualization
============================================================
""")
No comments:
Post a Comment