Total Pageviews

Monday, September 28, 2026

kMEDOID

 # ================================================================

# K-MEDOIDS CLUSTERING - BOSTON HOUSING DATASET

# Complete Google Colab Single-Cell Code

# ================================================================


# -----------------------------

# 1. INSTALL REQUIRED LIBRARY

# -----------------------------

!pip install scikit-learn-extra -q


# -----------------------------

# 2. IMPORT LIBRARIES

# -----------------------------

import numpy as np

import pandas as pd

import matplotlib.pyplot as plt


from sklearn.datasets import fetch_openml

from sklearn.preprocessing import StandardScaler

from sklearn.decomposition import PCA

from sklearn.metrics import silhouette_score


from sklearn_extra.cluster import KMedoids


import warnings

warnings.filterwarnings("ignore")


# -----------------------------

# 3. LOAD BOSTON HOUSING DATA

# -----------------------------

print("=" * 70)

print("BOSTON HOUSING DATASET - K-MEDOIDS CLUSTERING")

print("=" * 70)


boston = fetch_openml(

    name="boston",

    version=1,

    as_frame=True

)


df = boston.frame.copy()


print("\nDataset Shape:", df.shape)


# -----------------------------

# 4. DISPLAY DATA

# -----------------------------

print("\nFirst 5 Rows:")

display(df.head())


print("\nDataset Information:")

print(df.info())


# -----------------------------

# 5. CONVERT NUMERIC COLUMNS

# -----------------------------

df = df.apply(pd.to_numeric, errors="coerce")


# -----------------------------

# 6. CHECK MISSING VALUES

# -----------------------------

print("\nMissing Values:")

print(df.isnull().sum())


# -----------------------------

# 7. HANDLE MISSING VALUES

# -----------------------------

df = df.dropna()


print("\nShape After Removing Missing Values:")

print(df.shape)


# -----------------------------

# 8. CHECK DUPLICATES

# -----------------------------

print("\nDuplicate Rows:")

print(df.duplicated().sum())


# Remove duplicates

df = df.drop_duplicates()


print("Shape After Removing Duplicates:")

print(df.shape)


# -----------------------------

# 9. SEPARATE FEATURES

# -----------------------------

# MEDV is the house-price target.

# It will NOT be used for clustering.


X = df.drop("MEDV", axis=1)


y = df["MEDV"]


print("\nFeatures Used for K-Medoids:")

print(X.columns.tolist())


print("\nNumber of Features:", X.shape[1])


# -----------------------------

# 10. STANDARDIZATION

# -----------------------------

scaler = StandardScaler()


X_scaled = scaler.fit_transform(X)


X_scaled = pd.DataFrame(

    X_scaled,

    columns=X.columns

)


print("\nScaled Data:")

display(X_scaled.head())


# -----------------------------

# 11. CHECK SCALED DATA

# -----------------------------

print("\nMean After Scaling:")

print(X_scaled.mean().round(2))


print("\nStandard Deviation After Scaling:")

print(X_scaled.std().round(2))


# -----------------------------

# 12. FIND BEST K

# USING SILHOUETTE SCORE

# -----------------------------

silhouette_scores = []


print("\n" + "=" * 70)

print("SILHOUETTE ANALYSIS")

print("=" * 70)


for k in range(2, 11):


    model = KMedoids(

        n_clusters=k,

        metric="euclidean",

        method="alternate",

        init="k-medoids++",

        max_iter=300,

        random_state=42

    )


    labels = model.fit_predict(X_scaled)


    score = silhouette_score(

        X_scaled,

        labels

    )


    silhouette_scores.append(score)


    print(

        "K =", k,

        "| Silhouette Score =",

        round(score, 4)

    )


# -----------------------------

# 13. PLOT SILHOUETTE SCORE

# -----------------------------

plt.figure(figsize=(9, 5))


plt.plot(

    range(2, 11),

    silhouette_scores,

    marker="o"

)


plt.xlabel("Number of Clusters (K)")

plt.ylabel("Silhouette Score")

plt.title("K-Medoids - Silhouette Score")


plt.xticks(range(2, 11))

plt.grid(True)


plt.show()


# -----------------------------

# 14. AUTOMATICALLY SELECT K

# -----------------------------

best_k = list(range(2, 11))[

    np.argmax(silhouette_scores)

]


print("\nBest K According to Silhouette Score:", best_k)


# -----------------------------

# 15. TRAIN FINAL K-MEDOIDS

# -----------------------------

kmedoids = KMedoids(

    n_clusters=best_k,

    metric="euclidean",

    method="alternate",

    init="k-medoids++",

    max_iter=300,

    random_state=42

)


kmedoids.fit(X_scaled)


# -----------------------------

# 16. GET CLUSTER LABELS

# -----------------------------

cluster_labels = kmedoids.labels_


df_result = df.copy()


df_result["Cluster"] = cluster_labels


print("\n" + "=" * 70)

print("K-MEDOIDS CLUSTERING RESULT")

print("=" * 70)


print("\nFirst 10 Cluster Assignments:")

display(df_result.head(10))


# -----------------------------

# 17. CLUSTER COUNTS

# -----------------------------

cluster_counts = (

    df_result["Cluster"]

    .value_counts()

    .sort_index()

)


print("\nNumber of Observations in Each Cluster:")

print(cluster_counts)


# -----------------------------

# 18. CLUSTER SIZE GRAPH

# -----------------------------

plt.figure(figsize=(8, 5))


cluster_counts.plot(

    kind="bar"

)


plt.xlabel("Cluster")

plt.ylabel("Number of Houses")

plt.title("Number of Houses in Each K-Medoids Cluster")


plt.xticks(rotation=0)

plt.grid(axis="y")


plt.show()


# -----------------------------

# 19. MEDOID INDICES

# -----------------------------

medoid_indices = kmedoids.medoid_indices_


print("\nMedoid Indices:")

print(medoid_indices)


# -----------------------------

# 20. ACTUAL MEDOID OBSERVATIONS

# -----------------------------

medoids = df.iloc[

    medoid_indices

].copy()


medoids["Cluster"] = range(best_k)


print("\nActual Medoid Observations:")

display(medoids)


# -----------------------------

# 21. MEDOID HOUSE VALUES

# -----------------------------

print("\nMEDV of Medoid Observations:")


for cluster_number, index in enumerate(medoid_indices):


    print(

        "Cluster",

        cluster_number,

        "| MEDV =",

        round(df.iloc[index]["MEDV"], 2)

    )


# -----------------------------

# 22. SILHOUETTE SCORE

# -----------------------------

final_silhouette = silhouette_score(

    X_scaled,

    cluster_labels

)


print("\nFinal Silhouette Score:")

print(round(final_silhouette, 4))


# -----------------------------

# 23. K-MEDOIDS INERTIA

# -----------------------------

print("\nK-Medoids Inertia:")

print(round(kmedoids.inertia_, 4))


# -----------------------------

# 24. PCA DIMENSION REDUCTION

# -----------------------------

pca = PCA(

    n_components=2

)


X_pca = pca.fit_transform(

    X_scaled

)


print("\nPCA Explained Variance:")

print(

    pca.explained_variance_ratio_

)


print(

    "\nTotal Explained Variance:",

    round(

        sum(pca.explained_variance_ratio_) * 100,

        2

    ),

    "%"

)


# -----------------------------

# 25. PCA CLUSTER VISUALIZATION

# -----------------------------

plt.figure(figsize=(10, 7))


scatter = plt.scatter(

    X_pca[:, 0],

    X_pca[:, 1],

    c=cluster_labels,

    s=45,

    alpha=0.7

)


# PCA coordinates of actual medoids

medoid_pca = X_pca[

    medoid_indices

]


plt.scatter(

    medoid_pca[:, 0],

    medoid_pca[:, 1],

    marker="X",

    s=250,

    edgecolors="black",

    linewidths=2,

    label="Medoids"

)


plt.xlabel("Principal Component 1")

plt.ylabel("Principal Component 2")


plt.title(

    "K-Medoids Clustering - Boston Housing"

)


plt.colorbar(

    scatter,

    label="Cluster"

)


plt.legend()

plt.grid(True)


plt.show()


# -----------------------------

# 26. CLUSTER-WISE ANALYSIS

# -----------------------------

cluster_analysis = (

    df_result

    .groupby("Cluster")

    .mean(numeric_only=True)

)


print("\n" + "=" * 70)

print("CLUSTER-WISE FEATURE ANALYSIS")

print("=" * 70)


display(

    cluster_analysis.round(2)

)


# -----------------------------

# 27. HOUSE PRICE ANALYSIS

# MEDV WAS NOT USED FOR TRAINING

# -----------------------------

cluster_report = (

    df_result

    .groupby("Cluster")

    .agg(

        Number_of_Houses=("MEDV", "count"),

        Average_MEDV=("MEDV", "mean"),

        Minimum_MEDV=("MEDV", "min"),

        Maximum_MEDV=("MEDV", "max")

    )

)


print("\nCluster House-Value Analysis:")

display(

    cluster_report.round(2)

)


# -----------------------------

# 28. AVERAGE MEDV GRAPH

# -----------------------------

plt.figure(figsize=(8, 5))


cluster_report[

    "Average_MEDV"

].plot(

    kind="bar"

)


plt.xlabel("Cluster")

plt.ylabel("Average MEDV")

plt.title(

    "Average House Value by K-Medoids Cluster"

)


plt.xticks(rotation=0)

plt.grid(axis="y")


plt.show()


# -----------------------------

# 29. FINAL SUMMARY

# -----------------------------

print("\n" + "=" * 70)

print("FINAL K-MEDOIDS SUMMARY")

print("=" * 70)


print("Dataset Size       :", df.shape)

print("Number of Features :", X.shape[1])

print("Selected K          :", best_k)

print("Silhouette Score    :", round(final_silhouette, 4))

print("Inertia             :", round(kmedoids.inertia_, 4))


print("\nCluster Sizes:")

print(cluster_counts)


print("\nMedoid Indices:")

print(medoid_indices)


print("\nK-Medoids completed successfully.")

No comments:

Post a Comment