Total Pageviews

Monday, September 28, 2026

kMEANS

 # ============================================================

# BOSTON HOUSING DATASET

# K-MEANS CLUSTERING

# COMPLETE DATA PREPROCESSING + CLUSTER ANALYSIS

# GOOGLE COLAB VERSION

# ============================================================



# ============================================================

# 1. IMPORT LIBRARIES

# ============================================================


import numpy as np

import pandas as pd


import matplotlib.pyplot as plt

import seaborn as sns


from sklearn.datasets import fetch_openml


from sklearn.model_selection import train_test_split


from sklearn.pipeline import Pipeline


from sklearn.compose import ColumnTransformer


from sklearn.impute import SimpleImputer


from sklearn.preprocessing import (

    StandardScaler,

    OneHotEncoder

)


from sklearn.cluster import KMeans


from sklearn.metrics import (

    silhouette_score,

    adjusted_rand_score,

    normalized_mutual_info_score

)


from sklearn.decomposition import PCA


import warnings


warnings.filterwarnings("ignore")



# ============================================================

# 2. LOAD BOSTON HOUSING DATASET

# ============================================================


boston = fetch_openml(

    name="boston",

    version=1,

    as_frame=True

)


df = boston.frame.copy()


print("Boston Housing Dataset Loaded Successfully!")



# ============================================================

# 3. DISPLAY FIRST 5 ROWS

# ============================================================


print("\nFirst 5 Rows:")


display(

    df.head()

)



# ============================================================

# 4. DATASET SHAPE

# ============================================================


print("\nDataset Shape:")


print(

    df.shape

)



# ============================================================

# 5. COLUMN NAMES

# ============================================================


print("\nColumn Names:")


print(

    df.columns.tolist()

)



# ============================================================

# 6. DATA INFORMATION

# ============================================================


print("\nDataset Information:")


df.info()



# ============================================================

# 7. DATA TYPES

# ============================================================


print("\nData Types:")


print(

    df.dtypes

)



# ============================================================

# 8. MISSING VALUES

# ============================================================


print("\nMissing Values:")


print(

    df.isnull().sum()

)



# ============================================================

# 9. MISSING VALUE PERCENTAGE

# ============================================================


missing_percentage = (


    df.isnull()

      .mean()

      .mul(100)

      .sort_values(

          ascending=False

      )


)


print(

    "\nMissing Value Percentage:"

)


print(

    missing_percentage

)



# ============================================================

# 10. DUPLICATE RECORDS

# ============================================================


duplicate_count = df.duplicated().sum()


print(

    "\nNumber of Duplicate Rows:",

    duplicate_count

)



# ============================================================

# 11. REMOVE DUPLICATES

# ============================================================


df = df.drop_duplicates()


print(

    "\nShape After Duplicate Removal:"

)


print(

    df.shape

)



# ============================================================

# 12. STATISTICAL SUMMARY

# ============================================================


print(

    "\nStatistical Summary:"

)


display(

    df.describe()

)



# ============================================================

# 13. UNIQUE VALUES

# ============================================================


print(

    "\nNumber of Unique Values:"

)


print(

    df.nunique()

)



# ============================================================

# 14. TARGET / REFERENCE VARIABLE

# ============================================================


target = "MEDV"


print(

    "\nReference Variable:",

    target

)



# ============================================================

# 15. MEDV DISTRIBUTION

# ============================================================


plt.figure(

    figsize=(9,5)

)


sns.histplot(

    df[target],

    kde=True

)


plt.title(

    "Distribution of Boston House Prices"

)


plt.xlabel(

    "MEDV"

)


plt.ylabel(

    "Frequency"

)


plt.show()



# ============================================================

# 16. IDENTIFY NUMERICAL COLUMNS

# ============================================================


numeric_columns = (


    df.select_dtypes(

        include=np.number

    ).columns.tolist()


)



print(

    "\nNumerical Columns:"

)


print(

    numeric_columns

)



# ============================================================

# 17. OUTLIER DETECTION USING IQR

# ============================================================


print(

    "\n============================================"

)


print(

    "          IQR OUTLIER ANALYSIS"

)


print(

    "============================================"

)



outlier_summary = []



for column in numeric_columns:


    Q1 = df[column].quantile(

        0.25

    )


    Q3 = df[column].quantile(

        0.75

    )


    IQR = Q3 - Q1


    lower_bound = (

        Q1 - 1.5 * IQR

    )


    upper_bound = (

        Q3 + 1.5 * IQR

    )


    outlier_count = (


        (

            (df[column] < lower_bound)

            |

            (df[column] > upper_bound)

        )

        .sum()


    )


    outlier_summary.append({


        "Feature": column,


        "Q1": Q1,


        "Q3": Q3,


        "IQR": IQR,


        "Lower Bound": lower_bound,


        "Upper Bound": upper_bound,


        "Outliers": outlier_count


    })



outlier_table = pd.DataFrame(

    outlier_summary

)


display(

    outlier_table

)



# ============================================================

# 18. BOX PLOT BEFORE OUTLIER TREATMENT

# ============================================================


plt.figure(

    figsize=(15,6)

)


sns.boxplot(

    data=df[numeric_columns]

)


plt.title(

    "Boston Housing Features - Before Outlier Treatment"

)


plt.xticks(

    rotation=45

)


plt.show()



# ============================================================

# 19. IQR OUTLIER CAPPING

# ============================================================

#

# Instead of deleting observations, values outside the

# IQR limits are capped at the lower/upper boundaries.

#

# MEDV is not used for clustering, so it is left untouched.

# ============================================================


df_cluster = df.copy()



feature_columns = [


    column


    for column in numeric_columns


    if column != target


]



for column in feature_columns:


    Q1 = df_cluster[column].quantile(

        0.25

    )


    Q3 = df_cluster[column].quantile(

        0.75

    )


    IQR = Q3 - Q1


    lower_bound = (

        Q1 - 1.5 * IQR

    )


    upper_bound = (

        Q3 + 1.5 * IQR

    )


    df_cluster[column] = (


        df_cluster[column]

        .clip(

            lower_bound,

            upper_bound

        )


    )



print(

    "\nIQR Outlier Capping Completed!"

)



# ============================================================

# 20. BOX PLOT AFTER OUTLIER TREATMENT

# ============================================================


plt.figure(

    figsize=(15,6)

)


sns.boxplot(

    data=df_cluster[feature_columns]

)


plt.title(

    "Boston Housing Features - After IQR Capping"

)


plt.xticks(

    rotation=45

)


plt.show()



# ============================================================

# 21. CHECK MISSING VALUES AFTER PREPROCESSING

# ============================================================


print(

    "\nMissing Values After Preprocessing:"

)


print(

    df_cluster[feature_columns]

    .isnull()

    .sum()

)



# ============================================================

# 22. PREPARE FEATURES FOR K-MEANS

# ============================================================

#

# IMPORTANT:

#

# MEDV is NOT used to create clusters.

#

# K-Means will find groups using the housing features.

#

# MEDV is retained separately only for later interpretation.

# ============================================================


X = df_cluster.drop(

    columns=[target]

)


medv_reference = df_cluster[target].copy()



print(

    "\nK-Means Feature Shape:"

)


print(

    X.shape

)



# ============================================================

# 23. IDENTIFY NUMERICAL AND CATEGORICAL FEATURES

# ============================================================


numeric_features = (


    X.select_dtypes(

        include=np.number

    )

    .columns

    .tolist()


)



categorical_features = (


    X.select_dtypes(

        exclude=np.number

    )

    .columns

    .tolist()


)



print(

    "\nNumerical Features:"

)


print(

    numeric_features

)



print(

    "\nCategorical Features:"

)


print(

    categorical_features

)



# ============================================================

# 24. NUMERICAL PREPROCESSING

# ============================================================


numeric_pipeline = Pipeline(


    steps=[


        (

            "imputer",


            SimpleImputer(

                strategy="median"

            )


        ),


        (

            "scaler",


            StandardScaler()

        )


    ]


)



# ============================================================

# 25. CATEGORICAL PREPROCESSING

# ============================================================


categorical_pipeline = Pipeline(


    steps=[


        (

            "imputer",


            SimpleImputer(

                strategy="most_frequent"

            )


        ),


        (

            "encoder",


            OneHotEncoder(

                handle_unknown="ignore"

            )


        )


    ]


)



# ============================================================

# 26. COMBINE PREPROCESSING

# ============================================================


preprocessor = ColumnTransformer(


    transformers=[


        (

            "numeric",


            numeric_pipeline,


            numeric_features


        ),


        (

            "categorical",


            categorical_pipeline,


            categorical_features


        )


    ]


)



# ============================================================

# 27. TRANSFORM DATA

# ============================================================


X_processed = preprocessor.fit_transform(

    X

)



# Convert sparse matrix to dense matrix if necessary


if hasattr(

    X_processed,

    "toarray"

):


    X_processed = X_processed.toarray()



print(

    "\nProcessed Data Shape:"

)


print(

    X_processed.shape

)



# ============================================================

# 28. CHECK PROCESSED DATA

# ============================================================


print(

    "\nFirst 5 Processed Rows:"

)


display(


    pd.DataFrame(

        X_processed

    ).head()


)



# ============================================================

# 29. ELBOW METHOD

# ============================================================

#

# Inertia measures the sum of squared distances between

# observations and their assigned cluster centers.

#

# We calculate K = 2 to 10.

# ============================================================


k_values = range(

    2,

    11

)


inertia_values = []



for k in k_values:


    kmeans = KMeans(


        n_clusters=k,


        random_state=42,


        n_init=10


    )


    kmeans.fit(

        X_processed

    )


    inertia_values.append(

        kmeans.inertia_

    )



# ============================================================

# 30. ELBOW GRAPH

# ============================================================


plt.figure(

    figsize=(10,6)

)


plt.plot(


    list(k_values),


    inertia_values,


    marker="o"


)


plt.xlabel(

    "Number of Clusters (K)"

)


plt.ylabel(

    "Inertia"

)


plt.title(

    "Elbow Method for K-Means"

)


plt.xticks(

    list(k_values)

)


plt.grid(

    True

)


plt.show()



# ============================================================

# 31. SILHOUETTE SCORE

# ============================================================

#

# Silhouette score measures how well observations fit

# within their own cluster compared with other clusters.

# ============================================================


silhouette_values = []



for k in k_values:


    kmeans = KMeans(


        n_clusters=k,


        random_state=42,


        n_init=10


    )


    labels = kmeans.fit_predict(

        X_processed

    )


    score = silhouette_score(


        X_processed,


        labels


    )


    silhouette_values.append(

        score

    )



# ============================================================

# 32. SILHOUETTE SCORE TABLE

# ============================================================


silhouette_table = pd.DataFrame({


    "K": list(k_values),


    "Silhouette Score":

        silhouette_values


})



silhouette_table[

    "Silhouette Score"

] = (


    silhouette_table[

        "Silhouette Score"

    ].round(4)


)



print(

    "\nSilhouette Scores:"

)


display(

    silhouette_table

)



# ============================================================

# 33. SILHOUETTE GRAPH

# ============================================================


plt.figure(

    figsize=(10,6)

)


plt.plot(


    list(k_values),


    silhouette_values,


    marker="o"


)


plt.xlabel(

    "Number of Clusters (K)"

)


plt.ylabel(

    "Silhouette Score"

)


plt.title(

    "Silhouette Score for Different K Values"

)


plt.xticks(

    list(k_values)

)


plt.grid(

    True

)


plt.show()



# ============================================================

# 34. SELECT BEST K USING SILHOUETTE SCORE

# ============================================================


best_k_index = np.argmax(

    silhouette_values

)


best_k = list(

    k_values

)[best_k_index]



best_silhouette = silhouette_values[

    best_k_index

]



print(

    "\n============================================"

)


print(

    "             BEST K VALUE"

)


print(

    "============================================"

)


print(

    "Best K:",

    best_k

)


print(

    "Silhouette Score:",

    round(

        best_silhouette,

        4

    )

)



# ============================================================

# 35. CREATE FINAL K-MEANS MODEL

# ============================================================


kmeans_final = KMeans(


    n_clusters=best_k,


    random_state=42,


    n_init=10


)



# ============================================================

# 36. FIT K-MEANS

# ============================================================


cluster_labels = kmeans_final.fit_predict(


    X_processed


)



print(

    "\nK-Means Clustering Completed!"

)



# ============================================================

# 37. ADD CLUSTER LABELS TO DATASET

# ============================================================


df_cluster["Cluster"] = cluster_labels



# ============================================================

# 38. CLUSTER COUNTS

# ============================================================


cluster_counts = (


    df_cluster["Cluster"]

    .value_counts()

    .sort_index()


)



print(

    "\nNumber of Houses in Each Cluster:"

)


print(

    cluster_counts

)



# ============================================================

# 39. CLUSTER DISTRIBUTION GRAPH

# ============================================================


plt.figure(

    figsize=(9,6)

)


sns.countplot(


    x="Cluster",


    data=df_cluster,


    order=sorted(

        df_cluster["Cluster"].unique()

    )


)


plt.title(

    "Number of Houses in Each K-Means Cluster"

)


plt.xlabel(

    "Cluster"

)


plt.ylabel(

    "Number of Houses"

)


plt.show()



# ============================================================

# 40. FINAL SILHOUETTE SCORE

# ============================================================


final_silhouette = silhouette_score(


    X_processed,


    cluster_labels


)



print(

    "\nFinal Silhouette Score:"

)


print(

    round(

        final_silhouette,

        4

    )

)



# ============================================================

# 41. CLUSTER CENTERS

# ============================================================


cluster_centers = (


    pd.DataFrame(

        kmeans_final.cluster_centers_

    )


)



print(

    "\nCluster Centers:"

)


display(

    cluster_centers

)



# ============================================================

# 42. CLUSTER-WISE HOUSE PRICE ANALYSIS

# ============================================================

#

# MEDV was NOT used to create the clusters.

#

# We now use MEDV only to understand what the clusters

# represent after clustering.

# ============================================================


cluster_price_summary = (


    df_cluster

    .groupby("Cluster")[target]

    .agg([

        "count",

        "mean",

        "median",

        "min",

        "max"

    ])

    .round(2)


)



print(

    "\nCluster-Wise House Price Summary:"

)


display(

    cluster_price_summary

)



# ============================================================

# 43. CLUSTER-WISE FEATURE MEANS

# ============================================================


cluster_feature_summary = (


    df_cluster

    .groupby("Cluster")[

        feature_columns

    ]

    .mean()

    .round(2)


)



print(

    "\nCluster-Wise Feature Means:"

)


display(

    cluster_feature_summary

)



# ============================================================

# 44. HEATMAP OF CLUSTER FEATURE MEANS

# ============================================================


plt.figure(

    figsize=(15,7)

)


sns.heatmap(


    cluster_feature_summary,


    annot=True,


    fmt=".2f",


    cmap="coolwarm"


)


plt.title(

    "Cluster-Wise Feature Means"

)


plt.xlabel(

    "Features"

)


plt.ylabel(

    "Cluster"

)


plt.show()



# ============================================================

# 45. CLUSTER VS MEDV

# ============================================================


plt.figure(

    figsize=(10,6)

)


sns.boxplot(


    x="Cluster",


    y=target,


    data=df_cluster


)


plt.title(

    "House Price Distribution by K-Means Cluster"

)


plt.xlabel(

    "Cluster"

)


plt.ylabel(

    "MEDV"

)


plt.show()



# ============================================================

# 46. CLUSTER VS MEDV BAR CHART

# ============================================================


cluster_price_mean = (


    df_cluster

    .groupby("Cluster")[target]

    .mean()


)



plt.figure(

    figsize=(9,6)

)


plt.bar(


    cluster_price_mean.index,


    cluster_price_mean.values


)


plt.xlabel(

    "Cluster"

)


plt.ylabel(

    "Average MEDV"

)


plt.title(

    "Average House Price by Cluster"

)


plt.show()



# ============================================================

# 47. PCA FOR VISUALIZATION

# ============================================================

#

# K-Means may use many features.

#

# PCA reduces the processed data to two dimensions

# so we can visualize the clusters.

# ============================================================


pca = PCA(

    n_components=2

)



X_pca = pca.fit_transform(

    X_processed

)



print(

    "\nPCA Explained Variance:"

)


print(

    pca.explained_variance_ratio_

)



print(

    "\nTotal Variance Explained:"

)


print(

    round(

        pca.explained_variance_ratio_.sum(),

        4

    )

)



# ============================================================

# 48. CREATE PCA DATAFRAME

# ============================================================


pca_df = pd.DataFrame({


    "PC1": X_pca[:, 0],


    "PC2": X_pca[:, 1],


    "Cluster": cluster_labels


})



# ============================================================

# 49. PCA CLUSTER VISUALIZATION

# ============================================================


plt.figure(

    figsize=(10,7)

)


sns.scatterplot(


    data=pca_df,


    x="PC1",


    y="PC2",


    hue="Cluster",


    palette="tab10",


    s=70


)


plt.title(

    "K-Means Clusters Visualized Using PCA"

)


plt.xlabel(

    "Principal Component 1"

)


plt.ylabel(

    "Principal Component 2"

)


plt.legend(

    title="Cluster"

)


plt.show()



# ============================================================

# 50. PCA WITH CLUSTER CENTERS

# ============================================================


centers_pca = pca.transform(


    kmeans_final.cluster_centers_


)



plt.figure(

    figsize=(10,7)

)


sns.scatterplot(


    data=pca_df,


    x="PC1",


    y="PC2",


    hue="Cluster",


    palette="tab10",


    s=60


)



plt.scatter(


    centers_pca[:, 0],


    centers_pca[:, 1],


    marker="X",


    s=250,


    color="black",


    label="Cluster Centers"


)



plt.title(

    "K-Means Clusters and Cluster Centers"

)


plt.xlabel(

    "Principal Component 1"

)


plt.ylabel(

    "Principal Component 2"

)


plt.legend()


plt.show()



# ============================================================

# 51. OPTIONAL: CREATE LOW/HIGH PRICE REFERENCE

# ============================================================

#

# This is NOT used for training K-Means.

#

# It is only used to examine how the unsupervised clusters

# correspond to the known MEDV-based categories.

# ============================================================


median_price = df_cluster[target].median()



df_cluster["Price_Class"] = (


    df_cluster[target] >= median_price


).astype(int)



df_cluster["Price_Class_Name"] = (


    df_cluster["Price_Class"]

    .map({


        0: "Low Price",


        1: "High Price"


    })


)



# ============================================================

# 52. CLUSTER VS PRICE CLASS

# ============================================================


cluster_price_class = pd.crosstab(


    df_cluster["Cluster"],


    df_cluster["Price_Class_Name"]


)



print(

    "\nCluster vs Price Class:"

)


display(

    cluster_price_class

)



# ============================================================

# 53. CLUSTER VS PRICE CLASS GRAPH

# ============================================================


cluster_price_class.plot(


    kind="bar",


    figsize=(10,6)


)


plt.title(

    "Price Class Distribution Within Each Cluster"

)


plt.xlabel(

    "Cluster"

)


plt.ylabel(

    "Number of Houses"

)


plt.xticks(

    rotation=0

)


plt.show()



# ============================================================

# 54. OPTIONAL UNSUPERVISED AGREEMENT METRICS

# ============================================================

#

# These metrics are NOT training metrics.

#

# They compare the discovered clusters against the

# MEDV-based reference classes only for educational analysis.

# ============================================================


ari = adjusted_rand_score(


    df_cluster["Price_Class"],


    cluster_labels


)



nmi = normalized_mutual_info_score(


    df_cluster["Price_Class"],


    cluster_labels


)



print(

    "\n============================================"

)


print(

    "      OPTIONAL CLUSTER AGREEMENT"

)


print(

    "============================================"

)


print(

    "Adjusted Rand Index:",

    round(

        ari,

        4

    )

)


print(

    "Normalized Mutual Information:",

    round(

        nmi,

        4

    )

)



# ============================================================

# 55. SAVE FINAL CLUSTERED DATA

# ============================================================


final_clustered_data = df_cluster.copy()



print(

    "\nFinal Clustered Dataset:"

)


display(

    final_clustered_data.head()

)



# ============================================================

# 56. FINAL SUMMARY TABLE

# ============================================================


summary_table = pd.DataFrame({


    "Item": [


        "Algorithm",


        "Number of Clusters",


        "Silhouette Score",


        "Number of Features",


        "Number of Houses",


        "PCA Components"


    ],


    "Value": [


        "K-Means",


        best_k,


        round(

            final_silhouette,

            4

        ),


        len(feature_columns),


        len(df_cluster),


        2


    ]


})



print(

    "\n============================================"

)


print(

    "             K-MEANS SUMMARY"

)


print(

    "============================================"

)


display(

    summary_table

)



# ============================================================

# 57. CLUSTER INTERPRETATION

# ============================================================


print("""

============================================================

K-MEANS CLUSTERING - THEORY

============================================================


K-Means is an UNSUPERVISED MACHINE LEARNING algorithm.


Unlike Linear Regression, Logistic Regression and KNN

classification, K-Means does not require a target variable

during clustering.


The algorithm groups similar observations into K clusters.


============================================================


HOW K-MEANS WORKS

============================================================


Step 1:

Choose the number of clusters K.


Step 2:

Randomly initialize K cluster centers.


Step 3:

Calculate the distance between each observation

and every cluster center.


Step 4:

Assign each observation to the nearest cluster.


Step 5:

Recalculate the cluster centers.


Step 6:

Repeat the process until the cluster assignments

stabilize.


============================================================


ELBOW METHOD

============================================================


The Elbow Method calculates:


INERTIA


Inertia represents the within-cluster sum of squared

distances.


As K increases, inertia normally decreases.


The point where the reduction begins to slow down

is called the "elbow".


============================================================


SILHOUETTE SCORE

============================================================


Silhouette Score measures how well observations fit

their assigned clusters.


Higher values generally indicate better separation

and cohesion.


The score ranges approximately from:


-1 to +1


A value closer to +1 indicates better-defined clusters.


============================================================


IMPORTANT BOSTON HOUSING POINT

============================================================


MEDV was NOT used to create the clusters.


K-Means used the housing characteristics/features.


MEDV was used AFTER clustering only to understand

the characteristics of the resulting clusters.


Therefore this is an UNSUPERVISED learning experiment.


============================================================


PREPROCESSING USED

============================================================


1. Missing-value checking

2. Missing-value imputation

3. Duplicate detection

4. Duplicate removal

5. IQR outlier detection

6. IQR outlier capping

7. Feature scaling

8. Categorical encoding

9. PCA visualization


============================================================

""")

No comments:

Post a Comment