# ================================================================
# K-MEDOIDS CLUSTERING - BOSTON HOUSING DATASET
# Complete Google Colab Single-Cell Code
# ================================================================
# -----------------------------
# 1. INSTALL REQUIRED LIBRARY
# -----------------------------
!pip install scikit-learn-extra -q
# -----------------------------
# 2. IMPORT LIBRARIES
# -----------------------------
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
from sklearn.datasets import fetch_openml
from sklearn.preprocessing import StandardScaler
from sklearn.decomposition import PCA
from sklearn.metrics import silhouette_score
from sklearn_extra.cluster import KMedoids
import warnings
warnings.filterwarnings("ignore")
# -----------------------------
# 3. LOAD BOSTON HOUSING DATA
# -----------------------------
print("=" * 70)
print("BOSTON HOUSING DATASET - K-MEDOIDS CLUSTERING")
print("=" * 70)
boston = fetch_openml(
name="boston",
version=1,
as_frame=True
)
df = boston.frame.copy()
print("\nDataset Shape:", df.shape)
# -----------------------------
# 4. DISPLAY DATA
# -----------------------------
print("\nFirst 5 Rows:")
display(df.head())
print("\nDataset Information:")
print(df.info())
# -----------------------------
# 5. CONVERT NUMERIC COLUMNS
# -----------------------------
df = df.apply(pd.to_numeric, errors="coerce")
# -----------------------------
# 6. CHECK MISSING VALUES
# -----------------------------
print("\nMissing Values:")
print(df.isnull().sum())
# -----------------------------
# 7. HANDLE MISSING VALUES
# -----------------------------
df = df.dropna()
print("\nShape After Removing Missing Values:")
print(df.shape)
# -----------------------------
# 8. CHECK DUPLICATES
# -----------------------------
print("\nDuplicate Rows:")
print(df.duplicated().sum())
# Remove duplicates
df = df.drop_duplicates()
print("Shape After Removing Duplicates:")
print(df.shape)
# -----------------------------
# 9. SEPARATE FEATURES
# -----------------------------
# MEDV is the house-price target.
# It will NOT be used for clustering.
X = df.drop("MEDV", axis=1)
y = df["MEDV"]
print("\nFeatures Used for K-Medoids:")
print(X.columns.tolist())
print("\nNumber of Features:", X.shape[1])
# -----------------------------
# 10. STANDARDIZATION
# -----------------------------
scaler = StandardScaler()
X_scaled = scaler.fit_transform(X)
X_scaled = pd.DataFrame(
X_scaled,
columns=X.columns
)
print("\nScaled Data:")
display(X_scaled.head())
# -----------------------------
# 11. CHECK SCALED DATA
# -----------------------------
print("\nMean After Scaling:")
print(X_scaled.mean().round(2))
print("\nStandard Deviation After Scaling:")
print(X_scaled.std().round(2))
# -----------------------------
# 12. FIND BEST K
# USING SILHOUETTE SCORE
# -----------------------------
silhouette_scores = []
print("\n" + "=" * 70)
print("SILHOUETTE ANALYSIS")
print("=" * 70)
for k in range(2, 11):
model = KMedoids(
n_clusters=k,
metric="euclidean",
method="alternate",
init="k-medoids++",
max_iter=300,
random_state=42
)
labels = model.fit_predict(X_scaled)
score = silhouette_score(
X_scaled,
labels
)
silhouette_scores.append(score)
print(
"K =", k,
"| Silhouette Score =",
round(score, 4)
)
# -----------------------------
# 13. PLOT SILHOUETTE SCORE
# -----------------------------
plt.figure(figsize=(9, 5))
plt.plot(
range(2, 11),
silhouette_scores,
marker="o"
)
plt.xlabel("Number of Clusters (K)")
plt.ylabel("Silhouette Score")
plt.title("K-Medoids - Silhouette Score")
plt.xticks(range(2, 11))
plt.grid(True)
plt.show()
# -----------------------------
# 14. AUTOMATICALLY SELECT K
# -----------------------------
best_k = list(range(2, 11))[
np.argmax(silhouette_scores)
]
print("\nBest K According to Silhouette Score:", best_k)
# -----------------------------
# 15. TRAIN FINAL K-MEDOIDS
# -----------------------------
kmedoids = KMedoids(
n_clusters=best_k,
metric="euclidean",
method="alternate",
init="k-medoids++",
max_iter=300,
random_state=42
)
kmedoids.fit(X_scaled)
# -----------------------------
# 16. GET CLUSTER LABELS
# -----------------------------
cluster_labels = kmedoids.labels_
df_result = df.copy()
df_result["Cluster"] = cluster_labels
print("\n" + "=" * 70)
print("K-MEDOIDS CLUSTERING RESULT")
print("=" * 70)
print("\nFirst 10 Cluster Assignments:")
display(df_result.head(10))
# -----------------------------
# 17. CLUSTER COUNTS
# -----------------------------
cluster_counts = (
df_result["Cluster"]
.value_counts()
.sort_index()
)
print("\nNumber of Observations in Each Cluster:")
print(cluster_counts)
# -----------------------------
# 18. CLUSTER SIZE GRAPH
# -----------------------------
plt.figure(figsize=(8, 5))
cluster_counts.plot(
kind="bar"
)
plt.xlabel("Cluster")
plt.ylabel("Number of Houses")
plt.title("Number of Houses in Each K-Medoids Cluster")
plt.xticks(rotation=0)
plt.grid(axis="y")
plt.show()
# -----------------------------
# 19. MEDOID INDICES
# -----------------------------
medoid_indices = kmedoids.medoid_indices_
print("\nMedoid Indices:")
print(medoid_indices)
# -----------------------------
# 20. ACTUAL MEDOID OBSERVATIONS
# -----------------------------
medoids = df.iloc[
medoid_indices
].copy()
medoids["Cluster"] = range(best_k)
print("\nActual Medoid Observations:")
display(medoids)
# -----------------------------
# 21. MEDOID HOUSE VALUES
# -----------------------------
print("\nMEDV of Medoid Observations:")
for cluster_number, index in enumerate(medoid_indices):
print(
"Cluster",
cluster_number,
"| MEDV =",
round(df.iloc[index]["MEDV"], 2)
)
# -----------------------------
# 22. SILHOUETTE SCORE
# -----------------------------
final_silhouette = silhouette_score(
X_scaled,
cluster_labels
)
print("\nFinal Silhouette Score:")
print(round(final_silhouette, 4))
# -----------------------------
# 23. K-MEDOIDS INERTIA
# -----------------------------
print("\nK-Medoids Inertia:")
print(round(kmedoids.inertia_, 4))
# -----------------------------
# 24. PCA DIMENSION REDUCTION
# -----------------------------
pca = PCA(
n_components=2
)
X_pca = pca.fit_transform(
X_scaled
)
print("\nPCA Explained Variance:")
print(
pca.explained_variance_ratio_
)
print(
"\nTotal Explained Variance:",
round(
sum(pca.explained_variance_ratio_) * 100,
2
),
"%"
)
# -----------------------------
# 25. PCA CLUSTER VISUALIZATION
# -----------------------------
plt.figure(figsize=(10, 7))
scatter = plt.scatter(
X_pca[:, 0],
X_pca[:, 1],
c=cluster_labels,
s=45,
alpha=0.7
)
# PCA coordinates of actual medoids
medoid_pca = X_pca[
medoid_indices
]
plt.scatter(
medoid_pca[:, 0],
medoid_pca[:, 1],
marker="X",
s=250,
edgecolors="black",
linewidths=2,
label="Medoids"
)
plt.xlabel("Principal Component 1")
plt.ylabel("Principal Component 2")
plt.title(
"K-Medoids Clustering - Boston Housing"
)
plt.colorbar(
scatter,
label="Cluster"
)
plt.legend()
plt.grid(True)
plt.show()
# -----------------------------
# 26. CLUSTER-WISE ANALYSIS
# -----------------------------
cluster_analysis = (
df_result
.groupby("Cluster")
.mean(numeric_only=True)
)
print("\n" + "=" * 70)
print("CLUSTER-WISE FEATURE ANALYSIS")
print("=" * 70)
display(
cluster_analysis.round(2)
)
# -----------------------------
# 27. HOUSE PRICE ANALYSIS
# MEDV WAS NOT USED FOR TRAINING
# -----------------------------
cluster_report = (
df_result
.groupby("Cluster")
.agg(
Number_of_Houses=("MEDV", "count"),
Average_MEDV=("MEDV", "mean"),
Minimum_MEDV=("MEDV", "min"),
Maximum_MEDV=("MEDV", "max")
)
)
print("\nCluster House-Value Analysis:")
display(
cluster_report.round(2)
)
# -----------------------------
# 28. AVERAGE MEDV GRAPH
# -----------------------------
plt.figure(figsize=(8, 5))
cluster_report[
"Average_MEDV"
].plot(
kind="bar"
)
plt.xlabel("Cluster")
plt.ylabel("Average MEDV")
plt.title(
"Average House Value by K-Medoids Cluster"
)
plt.xticks(rotation=0)
plt.grid(axis="y")
plt.show()
# -----------------------------
# 29. FINAL SUMMARY
# -----------------------------
print("\n" + "=" * 70)
print("FINAL K-MEDOIDS SUMMARY")
print("=" * 70)
print("Dataset Size :", df.shape)
print("Number of Features :", X.shape[1])
print("Selected K :", best_k)
print("Silhouette Score :", round(final_silhouette, 4))
print("Inertia :", round(kmedoids.inertia_, 4))
print("\nCluster Sizes:")
print(cluster_counts)
print("\nMedoid Indices:")
print(medoid_indices)
print("\nK-Medoids completed successfully.")
No comments:
Post a Comment