Total Pageviews

Monday, September 28, 2026

KNN

 # ============================================================

# BOSTON HOUSING DATASET

# K-NEAREST NEIGHBORS (KNN)

# COMPLETE DATA PREPROCESSING + EVALUATION

# GOOGLE COLAB VERSION

# ============================================================



# ============================================================

# 1. IMPORT LIBRARIES

# ============================================================


import numpy as np

import pandas as pd


import matplotlib.pyplot as plt

import seaborn as sns


from sklearn.datasets import fetch_openml


from sklearn.model_selection import (

    train_test_split,

    cross_val_score

)


from sklearn.compose import ColumnTransformer


from sklearn.pipeline import Pipeline


from sklearn.impute import SimpleImputer


from sklearn.preprocessing import (

    StandardScaler,

    OneHotEncoder

)


from sklearn.neighbors import KNeighborsClassifier


from sklearn.metrics import (

    accuracy_score,

    precision_score,

    recall_score,

    f1_score,

    confusion_matrix,

    classification_report,

    ConfusionMatrixDisplay,

    roc_curve,

    roc_auc_score

)


import warnings


warnings.filterwarnings("ignore")



# ============================================================

# 2. LOAD BOSTON HOUSING DATASET

# ============================================================


boston = fetch_openml(

    name="boston",

    version=1,

    as_frame=True

)


df = boston.frame.copy()


print("Boston Housing Dataset Loaded Successfully!")



# ============================================================

# 3. DISPLAY FIRST 5 ROWS

# ============================================================


print("\nFirst 5 Rows:")


display(

    df.head()

)



# ============================================================

# 4. DATASET SHAPE

# ============================================================


print("\nDataset Shape:")


print(

    df.shape

)



# ============================================================

# 5. COLUMN NAMES

# ============================================================


print("\nColumn Names:")


print(

    df.columns.tolist()

)



# ============================================================

# 6. DATA INFORMATION

# ============================================================


print("\nDataset Information:")


df.info()



# ============================================================

# 7. DATA TYPES

# ============================================================


print("\nData Types:")


print(

    df.dtypes

)



# ============================================================

# 8. MISSING VALUES

# ============================================================


print("\nMissing Values:")


print(

    df.isnull().sum()

)



# ============================================================

# 9. MISSING VALUE PERCENTAGE

# ============================================================


missing_percentage = (


    df.isnull()

      .mean()

      .mul(100)

      .sort_values(

          ascending=False

      )


)


print("\nMissing Value Percentage:")


print(

    missing_percentage

)



# ============================================================

# 10. DUPLICATE VALUES

# ============================================================


print(

    "\nNumber of Duplicate Rows:",

    df.duplicated().sum()

)



# ============================================================

# 11. REMOVE DUPLICATES

# ============================================================


df = df.drop_duplicates()


print(

    "\nShape After Duplicate Removal:"

)


print(

    df.shape

)



# ============================================================

# 12. STATISTICAL SUMMARY

# ============================================================


print(

    "\nStatistical Summary:"

)


display(

    df.describe()

)



# ============================================================

# 13. TARGET VARIABLE

# ============================================================


target = "MEDV"


print(

    "\nTarget Variable:",

    target

)



# ============================================================

# 14. TARGET DISTRIBUTION

# ============================================================


plt.figure(

    figsize=(9,5)

)


sns.histplot(

    df[target],

    kde=True

)


plt.title(

    "Distribution of Boston House Prices"

)


plt.xlabel(

    "MEDV"

)


plt.ylabel(

    "Frequency"

)


plt.show()



# ============================================================

# 15. OUTLIER DETECTION

# ============================================================


numeric_columns = (


    df.select_dtypes(

        include=np.number

    ).columns


)



plt.figure(

    figsize=(14,6)

)


sns.boxplot(

    data=df[

        numeric_columns

    ]

)


plt.title(

    "Boxplot of Boston Housing Features"

)


plt.xticks(

    rotation=45

)


plt.show()



# ============================================================

# 16. IQR OUTLIER ANALYSIS

# ============================================================


print(

    "\n============================================"

)


print(

    "          IQR OUTLIER ANALYSIS"

)


print(

    "============================================"

)



for column in numeric_columns:


    Q1 = df[column].quantile(

        0.25

    )


    Q3 = df[column].quantile(

        0.75

    )


    IQR = Q3 - Q1


    lower_bound = (

        Q1 - 1.5 * IQR

    )


    upper_bound = (

        Q3 + 1.5 * IQR

    )


    outliers = df[

        (df[column] < lower_bound)

        |

        (df[column] > upper_bound)

    ]


    print(

        column,

        "→",

        len(outliers),

        "outliers"

    )



# ============================================================

# 17. OUTLIER CAPPING

# ============================================================


df_capped = df.copy()


for column in numeric_columns:


    Q1 = df_capped[column].quantile(

        0.25

    )


    Q3 = df_capped[column].quantile(

        0.75

    )


    IQR = Q3 - Q1


    lower_bound = (

        Q1 - 1.5 * IQR

    )


    upper_bound = (

        Q3 + 1.5 * IQR

    )


    df_capped[column] = (


        df_capped[column]

        .clip(

            lower_bound,

            upper_bound

        )


    )



print(

    "\nIQR Outlier Capping Completed!"

)



# ============================================================

# 18. CORRELATION MATRIX

# ============================================================


correlation = df.corr(

    numeric_only=True

)


print(

    "\nCorrelation with MEDV:"

)


print(

    correlation[target]

    .sort_values(

        ascending=False

    )

)



# ============================================================

# 19. CORRELATION HEATMAP

# ============================================================


plt.figure(

    figsize=(12,9)

)


sns.heatmap(

    correlation,

    annot=True,

    fmt=".2f",

    cmap="coolwarm"

)


plt.title(

    "Boston Housing Correlation Matrix"

)


plt.show()



# ============================================================

# 20. CONVERT MEDV INTO CLASS

# ============================================================

#

# Original MEDV is continuous.

#

# KNN classification requires classes.

#

# 0 = Low Price

# 1 = High Price

#

# Median MEDV is used as threshold.

# ============================================================


median_price = df[target].median()


print(

    "\nMedian House Price:",

    median_price

)



df["Price_Class"] = (


    df[target] >= median_price


).astype(int)



# ============================================================

# 21. DISPLAY CLASS DISTRIBUTION

# ============================================================


print(

    "\nPrice Class Distribution:"

)


print(

    df["Price_Class"].value_counts()

)



# ============================================================

# 22. CLASS DISTRIBUTION GRAPH

# ============================================================


plt.figure(

    figsize=(7,5)

)


sns.countplot(

    x="Price_Class",

    data=df

)


plt.title(

    "Low Price vs High Price Houses"

)


plt.xlabel(

    "Price Class"

)


plt.ylabel(

    "Number of Houses"

)


plt.xticks(

    [0,1],

    [

        "Low Price",

        "High Price"

    ]

)


plt.show()



# ============================================================

# 23. FEATURES AND TARGET

# ============================================================


X = df.drop(

    columns=[

        "MEDV",

        "Price_Class"

    ]

)


y = df["Price_Class"]



print(

    "\nFeature Shape:",

    X.shape

)


print(

    "Target Shape:",

    y.shape

)



# ============================================================

# 24. IDENTIFY NUMERICAL FEATURES

# ============================================================


numeric_features = (


    X.select_dtypes(

        include=np.number

    )

    .columns

    .tolist()


)



categorical_features = (


    X.select_dtypes(

        exclude=np.number

    )

    .columns

    .tolist()


)



print(

    "\nNumerical Features:"

)


print(

    numeric_features

)


print(

    "\nCategorical Features:"

)


print(

    categorical_features

)



# ============================================================

# 25. TRAIN TEST SPLIT

# ============================================================


X_train, X_test, y_train, y_test = (


    train_test_split(


        X,

        y,


        test_size=0.20,


        random_state=42,


        stratify=y


    )


)



print(

    "\nTraining Shape:",

    X_train.shape

)


print(

    "Testing Shape:",

    X_test.shape

)



# ============================================================

# 26. NUMERICAL PREPROCESSING

# ============================================================


numeric_pipeline = Pipeline(


    steps=[


        (

            "imputer",


            SimpleImputer(

                strategy="median"

            )

        ),


        (

            "scaler",


            StandardScaler()

        )


    ]


)



# ============================================================

# 27. CATEGORICAL PREPROCESSING

# ============================================================


categorical_pipeline = Pipeline(


    steps=[


        (

            "imputer",


            SimpleImputer(

                strategy="most_frequent"

            )

        ),


        (

            "encoder",


            OneHotEncoder(

                handle_unknown="ignore"

            )

        )


    ]


)



# ============================================================

# 28. COMBINE PREPROCESSING

# ============================================================


preprocessor = ColumnTransformer(


    transformers=[


        (

            "numeric",


            numeric_pipeline,


            numeric_features

        ),


        (

            "categorical",


            categorical_pipeline,


            categorical_features

        )


    ]


)



# ============================================================

# 29. KNN MODEL

# ============================================================


knn_model = Pipeline(


    steps=[


        (

            "preprocessing",


            preprocessor

        ),


        (

            "model",


            KNeighborsClassifier(

                n_neighbors=5

            )

        )


    ]


)



# ============================================================

# 30. TRAIN KNN MODEL

# ============================================================


knn_model.fit(


    X_train,


    y_train


)



print(

    "\nKNN Model Trained Successfully!"

)



# ============================================================

# 31. PREDICTION

# ============================================================


y_pred = knn_model.predict(

    X_test

)



# ============================================================

# 32. PREDICT PROBABILITY

# ============================================================


y_probability = (


    knn_model

    .predict_proba(

        X_test

    )[:, 1]


)



# ============================================================

# 33. ACCURACY

# ============================================================


accuracy = accuracy_score(


    y_test,


    y_pred


)



# ============================================================

# 34. PRECISION

# ============================================================


precision = precision_score(


    y_test,


    y_pred,


    zero_division=0


)



# ============================================================

# 35. RECALL

# ============================================================


recall = recall_score(


    y_test,


    y_pred,


    zero_division=0


)



# ============================================================

# 36. F1 SCORE

# ============================================================


f1 = f1_score(


    y_test,


    y_pred,


    zero_division=0


)



# ============================================================

# 37. ROC-AUC

# ============================================================


roc_auc = roc_auc_score(


    y_test,


    y_probability


)



# ============================================================

# 38. DISPLAY MODEL RESULTS

# ============================================================


print(

    "\n============================================"

)


print(

    "             KNN RESULTS"

)


print(

    "============================================"

)


print(

    "Accuracy  :",

    round(accuracy, 4)

)


print(

    "Precision :",

    round(precision, 4)

)


print(

    "Recall    :",

    round(recall, 4)

)


print(

    "F1 Score  :",

    round(f1, 4)

)


print(

    "ROC-AUC   :",

    round(roc_auc, 4)

)


print(

    "============================================"

)



# ============================================================

# 39. CONFUSION MATRIX

# ============================================================


cm = confusion_matrix(


    y_test,


    y_pred


)



print(

    "\nConfusion Matrix:"

)


print(

    cm

)



# ============================================================

# 40. TRUE NEGATIVE / FALSE POSITIVE /

#     FALSE NEGATIVE / TRUE POSITIVE

# ============================================================


TN, FP, FN, TP = cm.ravel()



print(

    "\nTrue Negative (TN):",

    TN

)


print(

    "False Positive (FP):",

    FP

)


print(

    "False Negative (FN):",

    FN

)


print(

    "True Positive (TP):",

    TP

)



# ============================================================

# 41. CONFUSION MATRIX GRAPH

# ============================================================


plt.figure(

    figsize=(7,6)

)


sns.heatmap(


    cm,


    annot=True,


    fmt="d",


    cmap="Blues",


    xticklabels=[

        "Predicted Low",

        "Predicted High"

    ],


    yticklabels=[

        "Actual Low",

        "Actual High"

    ]


)


plt.title(

    "KNN Confusion Matrix"

)


plt.xlabel(

    "Predicted Class"

)


plt.ylabel(

    "Actual Class"

)


plt.show()



# ============================================================

# 42. CLASSIFICATION REPORT

# ============================================================


print(

    "\n============================================"

)


print(

    "          CLASSIFICATION REPORT"

)


print(

    "============================================"

)


print(


    classification_report(


        y_test,


        y_pred,


        target_names=[

            "Low Price",

            "High Price"

        ],


        zero_division=0


    )


)



# ============================================================

# 43. PERFORMANCE TABLE

# ============================================================


metrics_table = pd.DataFrame({


    "Metric": [


        "Accuracy",


        "Precision",


        "Recall",


        "F1 Score",


        "ROC-AUC"


    ],


    "Score": [


        accuracy,


        precision,


        recall,


        f1,


        roc_auc


    ]


})



metrics_table["Score"] = (


    metrics_table["Score"]

    .round(4)


)



print(

    "\nKNN Performance Metrics:"

)


display(

    metrics_table

)



# ============================================================

# 44. CROSS VALIDATION

# ============================================================


cv_scores = cross_val_score(


    knn_model,


    X,


    y,


    cv=5,


    scoring="accuracy"


)



print(

    "\n5-Fold Cross Validation Accuracy:"

)


print(

    cv_scores

)



print(

    "\nMean Cross Validation Accuracy:"

)


print(

    round(

        cv_scores.mean(),

        4

    )

)



# ============================================================

# 45. TEST DIFFERENT K VALUES

# ============================================================


k_values = range(

    1,

    21

)


k_accuracy = []



for k in k_values:


    temp_knn = Pipeline(


        steps=[


            (

                "preprocessing",


                preprocessor

            ),


            (

                "model",


                KNeighborsClassifier(

                    n_neighbors=k

                )


            )


        ]


    )



    temp_knn.fit(

        X_train,

        y_train

    )



    temp_prediction = temp_knn.predict(

        X_test

    )



    score = accuracy_score(


        y_test,


        temp_prediction


    )



    k_accuracy.append(

        score

    )



# ============================================================

# 46. K VALUE GRAPH

# ============================================================


plt.figure(

    figsize=(10,6)

)


plt.plot(


    list(k_values),


    k_accuracy,


    marker="o"


)


plt.xlabel(

    "Number of Neighbors (K)"

)


plt.ylabel(

    "Accuracy"

)


plt.title(

    "KNN Accuracy for Different K Values"

)


plt.xticks(

    list(k_values)

)


plt.grid(

    True

)


plt.show()



# ============================================================

# 47. BEST K VALUE

# ============================================================


best_k_index = np.argmax(

    k_accuracy

)


best_k = list(

    k_values

)[best_k_index]


best_k_accuracy = k_accuracy[

    best_k_index

]



print(

    "\n============================================"

)


print(

    "             BEST K VALUE"

)


print(

    "============================================"

)


print(

    "Best K:",

    best_k

)


print(

    "Accuracy:",

    round(

        best_k_accuracy,

        4

    )

)



# ============================================================

# 48. TRAIN FINAL KNN USING BEST K

# ============================================================


final_knn = Pipeline(


    steps=[


        (

            "preprocessing",


            preprocessor

        ),


        (

            "model",


            KNeighborsClassifier(


                n_neighbors=best_k


            )


        )


    ]


)



final_knn.fit(


    X_train,


    y_train


)



final_prediction = final_knn.predict(

    X_test

)



final_probability = (


    final_knn

    .predict_proba(

        X_test

    )[:, 1]


)



# ============================================================

# 49. FINAL METRICS

# ============================================================


final_accuracy = accuracy_score(


    y_test,


    final_prediction


)



final_precision = precision_score(


    y_test,


    final_prediction,


    zero_division=0


)



final_recall = recall_score(


    y_test,


    final_prediction,


    zero_division=0


)



final_f1 = f1_score(


    y_test,


    final_prediction,


    zero_division=0


)



final_auc = roc_auc_score(


    y_test,


    final_probability


)



print(

    "\n============================================"

)


print(

    "       FINAL KNN MODEL RESULTS"

)


print(

    "============================================"

)


print(

    "Best K       :",

    best_k

)


print(

    "Accuracy     :",

    round(

        final_accuracy,

        4

    )

)


print(

    "Precision    :",

    round(

        final_precision,

        4

    )

)


print(

    "Recall       :",

    round(

        final_recall,

        4

    )

)


print(

    "F1 Score     :",

    round(

        final_f1,

        4

    )

)


print(

    "ROC-AUC      :",

    round(

        final_auc,

        4

    )

)


print(

    "============================================"

)



# ============================================================

# 50. FINAL CONFUSION MATRIX

# ============================================================


final_cm = confusion_matrix(


    y_test,


    final_prediction


)



plt.figure(

    figsize=(7,6)

)


sns.heatmap(


    final_cm,


    annot=True,


    fmt="d",


    cmap="Blues",


    xticklabels=[

        "Predicted Low",

        "Predicted High"

    ],


    yticklabels=[

        "Actual Low",

        "Actual High"

    ]


)


plt.title(

    f"Final KNN Confusion Matrix (K={best_k})"

)


plt.xlabel(

    "Predicted Class"

)


plt.ylabel(

    "Actual Class"

)


plt.show()



# ============================================================

# 51. FINAL CLASSIFICATION REPORT

# ============================================================


print(

    "\nFinal Classification Report:"

)


print(


    classification_report(


        y_test,


        final_prediction,


        target_names=[

            "Low Price",

            "High Price"

        ],


        zero_division=0


    )


)



# ============================================================

# 52. ROC CURVE

# ============================================================


fpr, tpr, thresholds = roc_curve(


    y_test,


    final_probability


)



plt.figure(

    figsize=(8,6)

)


plt.plot(


    fpr,


    tpr,


    label=f"KNN ROC-AUC = {final_auc:.4f}"


)


plt.plot(


    [0,1],


    [0,1],


    linestyle="--"


)


plt.xlabel(

    "False Positive Rate"

)


plt.ylabel(

    "True Positive Rate"

)


plt.title(

    "KNN ROC Curve"

)


plt.legend()


plt.show()



# ============================================================

# 53. ACTUAL VS PREDICTED

# ============================================================


prediction_table = pd.DataFrame({


    "Actual Class":

        y_test.values,


    "Predicted Class":

        final_prediction,


    "High Price Probability":

        final_probability


})



prediction_table["Actual Class"] = (


    prediction_table[

        "Actual Class"

    ].map({


        0: "Low Price",


        1: "High Price"


    })


)



prediction_table["Predicted Class"] = (


    prediction_table[

        "Predicted Class"

    ].map({


        0: "Low Price",


        1: "High Price"


    })


)



prediction_table[

    "High Price Probability"

] = (


    prediction_table[

        "High Price Probability"

    ].round(4)


)



print(

    "\nActual vs Predicted:"

)


display(

    prediction_table.head(20)

)



# ============================================================

# 54. NEW HOUSE CLASSIFICATION

# ============================================================

#

# This version avoids the X.mean() categorical error.

# Numerical columns → median

# Categorical columns → mode

# ============================================================


new_house = pd.DataFrame(

    index=[0],

    columns=X.columns

)



# Numerical features


for column in numeric_features:


    new_house.loc[

        0,

        column

    ] = X[

        column

    ].median()



# Categorical features


for column in categorical_features:


    new_house.loc[

        0,

        column

    ] = X[

        column

    ].mode()[0]



# Restore category dtype if necessary


for column in categorical_features:


    if str(

        X[column].dtype

    ) == "category":


        new_house[column] = pd.Categorical(


            new_house[column],


            categories=X[

                column

            ].cat.categories


        )



# ============================================================

# 55. PREDICT NEW HOUSE

# ============================================================


new_prediction = final_knn.predict(


    new_house


)[0]



new_probability = final_knn.predict_proba(


    new_house


)[0][1]



if new_prediction == 1:


    new_result = "High Price"


else:


    new_result = "Low Price"



print(

    "\n============================================"

)


print(

    "        NEW HOUSE CLASSIFICATION"

)


print(

    "============================================"

)


print(

    "Predicted Class:",

    new_result

)


print(

    "Probability of High Price:",

    round(

        new_probability,

        4

    )

)


print(

    "Probability of Low Price:",

    round(

        1 - new_probability,

        4

    )

)



# ============================================================

# 56. FINAL KNN SUMMARY

# ============================================================


print(

    "\n============================================"

)


print(

    "             KNN SUMMARY"

)


print(

    "============================================"

)


print(

    "Algorithm: K-Nearest Neighbors"

)


print(

    "Number of Neighbors:",

    best_k

)


print(

    "Accuracy:",

    round(

        final_accuracy,

        4

    )

)


print(

    "Precision:",

    round(

        final_precision,

        4

    )

)


print(

    "Recall:",

    round(

        final_recall,

        4

    )

)


print(

    "F1 Score:",

    round(

        final_f1,

        4

    )

)


print(

    "ROC-AUC:",

    round(

        final_auc,

        4

    )

)


print(

    "Cross Validation Mean:",

    round(

        cv_scores.mean(),

        4

    )

)


print(

    "============================================"

)



# ============================================================

# 57. KNN THEORY

# ============================================================


print("""

============================================================

K-NEAREST NEIGHBORS (KNN) - THEORY

============================================================


KNN is a supervised machine learning algorithm used for

classification and regression.


For classification, KNN looks at the nearest K observations

and assigns the class that receives the majority vote.


Example:


K = 5


Nearest neighbors:


Neighbor 1 → High Price

Neighbor 2 → High Price

Neighbor 3 → Low Price

Neighbor 4 → High Price

Neighbor 5 → Low Price


High Price = 3

Low Price  = 2


Prediction = High Price


============================================================


WHY STANDARDIZATION IS IMPORTANT?


KNN uses distance calculations.


Features with large numerical values can dominate the distance.


StandardScaler converts features approximately to:


Mean = 0

Standard Deviation = 1


Therefore, scaling is particularly important for KNN.


============================================================


K VALUE


Small K:

- Sensitive to noise

- More flexible

- Can overfit


Large K:

- Smoother decision boundary

- Less sensitive to noise

- Can underfit


The code tests K = 1 to K = 20.


============================================================


BOSTON HOUSING CLASSIFICATION


Original target:


MEDV = Continuous house price


Converted target:


0 = Low Price

1 = High Price


Median MEDV is used as the classification threshold.


============================================================

""")

kMEANS

 # ============================================================

# BOSTON HOUSING DATASET

# K-MEANS CLUSTERING

# COMPLETE DATA PREPROCESSING + CLUSTER ANALYSIS

# GOOGLE COLAB VERSION

# ============================================================



# ============================================================

# 1. IMPORT LIBRARIES

# ============================================================


import numpy as np

import pandas as pd


import matplotlib.pyplot as plt

import seaborn as sns


from sklearn.datasets import fetch_openml


from sklearn.model_selection import train_test_split


from sklearn.pipeline import Pipeline


from sklearn.compose import ColumnTransformer


from sklearn.impute import SimpleImputer


from sklearn.preprocessing import (

    StandardScaler,

    OneHotEncoder

)


from sklearn.cluster import KMeans


from sklearn.metrics import (

    silhouette_score,

    adjusted_rand_score,

    normalized_mutual_info_score

)


from sklearn.decomposition import PCA


import warnings


warnings.filterwarnings("ignore")



# ============================================================

# 2. LOAD BOSTON HOUSING DATASET

# ============================================================


boston = fetch_openml(

    name="boston",

    version=1,

    as_frame=True

)


df = boston.frame.copy()


print("Boston Housing Dataset Loaded Successfully!")



# ============================================================

# 3. DISPLAY FIRST 5 ROWS

# ============================================================


print("\nFirst 5 Rows:")


display(

    df.head()

)



# ============================================================

# 4. DATASET SHAPE

# ============================================================


print("\nDataset Shape:")


print(

    df.shape

)



# ============================================================

# 5. COLUMN NAMES

# ============================================================


print("\nColumn Names:")


print(

    df.columns.tolist()

)



# ============================================================

# 6. DATA INFORMATION

# ============================================================


print("\nDataset Information:")


df.info()



# ============================================================

# 7. DATA TYPES

# ============================================================


print("\nData Types:")


print(

    df.dtypes

)



# ============================================================

# 8. MISSING VALUES

# ============================================================


print("\nMissing Values:")


print(

    df.isnull().sum()

)



# ============================================================

# 9. MISSING VALUE PERCENTAGE

# ============================================================


missing_percentage = (


    df.isnull()

      .mean()

      .mul(100)

      .sort_values(

          ascending=False

      )


)


print(

    "\nMissing Value Percentage:"

)


print(

    missing_percentage

)



# ============================================================

# 10. DUPLICATE RECORDS

# ============================================================


duplicate_count = df.duplicated().sum()


print(

    "\nNumber of Duplicate Rows:",

    duplicate_count

)



# ============================================================

# 11. REMOVE DUPLICATES

# ============================================================


df = df.drop_duplicates()


print(

    "\nShape After Duplicate Removal:"

)


print(

    df.shape

)



# ============================================================

# 12. STATISTICAL SUMMARY

# ============================================================


print(

    "\nStatistical Summary:"

)


display(

    df.describe()

)



# ============================================================

# 13. UNIQUE VALUES

# ============================================================


print(

    "\nNumber of Unique Values:"

)


print(

    df.nunique()

)



# ============================================================

# 14. TARGET / REFERENCE VARIABLE

# ============================================================


target = "MEDV"


print(

    "\nReference Variable:",

    target

)



# ============================================================

# 15. MEDV DISTRIBUTION

# ============================================================


plt.figure(

    figsize=(9,5)

)


sns.histplot(

    df[target],

    kde=True

)


plt.title(

    "Distribution of Boston House Prices"

)


plt.xlabel(

    "MEDV"

)


plt.ylabel(

    "Frequency"

)


plt.show()



# ============================================================

# 16. IDENTIFY NUMERICAL COLUMNS

# ============================================================


numeric_columns = (


    df.select_dtypes(

        include=np.number

    ).columns.tolist()


)



print(

    "\nNumerical Columns:"

)


print(

    numeric_columns

)



# ============================================================

# 17. OUTLIER DETECTION USING IQR

# ============================================================


print(

    "\n============================================"

)


print(

    "          IQR OUTLIER ANALYSIS"

)


print(

    "============================================"

)



outlier_summary = []



for column in numeric_columns:


    Q1 = df[column].quantile(

        0.25

    )


    Q3 = df[column].quantile(

        0.75

    )


    IQR = Q3 - Q1


    lower_bound = (

        Q1 - 1.5 * IQR

    )


    upper_bound = (

        Q3 + 1.5 * IQR

    )


    outlier_count = (


        (

            (df[column] < lower_bound)

            |

            (df[column] > upper_bound)

        )

        .sum()


    )


    outlier_summary.append({


        "Feature": column,


        "Q1": Q1,


        "Q3": Q3,


        "IQR": IQR,


        "Lower Bound": lower_bound,


        "Upper Bound": upper_bound,


        "Outliers": outlier_count


    })



outlier_table = pd.DataFrame(

    outlier_summary

)


display(

    outlier_table

)



# ============================================================

# 18. BOX PLOT BEFORE OUTLIER TREATMENT

# ============================================================


plt.figure(

    figsize=(15,6)

)


sns.boxplot(

    data=df[numeric_columns]

)


plt.title(

    "Boston Housing Features - Before Outlier Treatment"

)


plt.xticks(

    rotation=45

)


plt.show()



# ============================================================

# 19. IQR OUTLIER CAPPING

# ============================================================

#

# Instead of deleting observations, values outside the

# IQR limits are capped at the lower/upper boundaries.

#

# MEDV is not used for clustering, so it is left untouched.

# ============================================================


df_cluster = df.copy()



feature_columns = [


    column


    for column in numeric_columns


    if column != target


]



for column in feature_columns:


    Q1 = df_cluster[column].quantile(

        0.25

    )


    Q3 = df_cluster[column].quantile(

        0.75

    )


    IQR = Q3 - Q1


    lower_bound = (

        Q1 - 1.5 * IQR

    )


    upper_bound = (

        Q3 + 1.5 * IQR

    )


    df_cluster[column] = (


        df_cluster[column]

        .clip(

            lower_bound,

            upper_bound

        )


    )



print(

    "\nIQR Outlier Capping Completed!"

)



# ============================================================

# 20. BOX PLOT AFTER OUTLIER TREATMENT

# ============================================================


plt.figure(

    figsize=(15,6)

)


sns.boxplot(

    data=df_cluster[feature_columns]

)


plt.title(

    "Boston Housing Features - After IQR Capping"

)


plt.xticks(

    rotation=45

)


plt.show()



# ============================================================

# 21. CHECK MISSING VALUES AFTER PREPROCESSING

# ============================================================


print(

    "\nMissing Values After Preprocessing:"

)


print(

    df_cluster[feature_columns]

    .isnull()

    .sum()

)



# ============================================================

# 22. PREPARE FEATURES FOR K-MEANS

# ============================================================

#

# IMPORTANT:

#

# MEDV is NOT used to create clusters.

#

# K-Means will find groups using the housing features.

#

# MEDV is retained separately only for later interpretation.

# ============================================================


X = df_cluster.drop(

    columns=[target]

)


medv_reference = df_cluster[target].copy()



print(

    "\nK-Means Feature Shape:"

)


print(

    X.shape

)



# ============================================================

# 23. IDENTIFY NUMERICAL AND CATEGORICAL FEATURES

# ============================================================


numeric_features = (


    X.select_dtypes(

        include=np.number

    )

    .columns

    .tolist()


)



categorical_features = (


    X.select_dtypes(

        exclude=np.number

    )

    .columns

    .tolist()


)



print(

    "\nNumerical Features:"

)


print(

    numeric_features

)



print(

    "\nCategorical Features:"

)


print(

    categorical_features

)



# ============================================================

# 24. NUMERICAL PREPROCESSING

# ============================================================


numeric_pipeline = Pipeline(


    steps=[


        (

            "imputer",


            SimpleImputer(

                strategy="median"

            )


        ),


        (

            "scaler",


            StandardScaler()

        )


    ]


)



# ============================================================

# 25. CATEGORICAL PREPROCESSING

# ============================================================


categorical_pipeline = Pipeline(


    steps=[


        (

            "imputer",


            SimpleImputer(

                strategy="most_frequent"

            )


        ),


        (

            "encoder",


            OneHotEncoder(

                handle_unknown="ignore"

            )


        )


    ]


)



# ============================================================

# 26. COMBINE PREPROCESSING

# ============================================================


preprocessor = ColumnTransformer(


    transformers=[


        (

            "numeric",


            numeric_pipeline,


            numeric_features


        ),


        (

            "categorical",


            categorical_pipeline,


            categorical_features


        )


    ]


)



# ============================================================

# 27. TRANSFORM DATA

# ============================================================


X_processed = preprocessor.fit_transform(

    X

)



# Convert sparse matrix to dense matrix if necessary


if hasattr(

    X_processed,

    "toarray"

):


    X_processed = X_processed.toarray()



print(

    "\nProcessed Data Shape:"

)


print(

    X_processed.shape

)



# ============================================================

# 28. CHECK PROCESSED DATA

# ============================================================


print(

    "\nFirst 5 Processed Rows:"

)


display(


    pd.DataFrame(

        X_processed

    ).head()


)



# ============================================================

# 29. ELBOW METHOD

# ============================================================

#

# Inertia measures the sum of squared distances between

# observations and their assigned cluster centers.

#

# We calculate K = 2 to 10.

# ============================================================


k_values = range(

    2,

    11

)


inertia_values = []



for k in k_values:


    kmeans = KMeans(


        n_clusters=k,


        random_state=42,


        n_init=10


    )


    kmeans.fit(

        X_processed

    )


    inertia_values.append(

        kmeans.inertia_

    )



# ============================================================

# 30. ELBOW GRAPH

# ============================================================


plt.figure(

    figsize=(10,6)

)


plt.plot(


    list(k_values),


    inertia_values,


    marker="o"


)


plt.xlabel(

    "Number of Clusters (K)"

)


plt.ylabel(

    "Inertia"

)


plt.title(

    "Elbow Method for K-Means"

)


plt.xticks(

    list(k_values)

)


plt.grid(

    True

)


plt.show()



# ============================================================

# 31. SILHOUETTE SCORE

# ============================================================

#

# Silhouette score measures how well observations fit

# within their own cluster compared with other clusters.

# ============================================================


silhouette_values = []



for k in k_values:


    kmeans = KMeans(


        n_clusters=k,


        random_state=42,


        n_init=10


    )


    labels = kmeans.fit_predict(

        X_processed

    )


    score = silhouette_score(


        X_processed,


        labels


    )


    silhouette_values.append(

        score

    )



# ============================================================

# 32. SILHOUETTE SCORE TABLE

# ============================================================


silhouette_table = pd.DataFrame({


    "K": list(k_values),


    "Silhouette Score":

        silhouette_values


})



silhouette_table[

    "Silhouette Score"

] = (


    silhouette_table[

        "Silhouette Score"

    ].round(4)


)



print(

    "\nSilhouette Scores:"

)


display(

    silhouette_table

)



# ============================================================

# 33. SILHOUETTE GRAPH

# ============================================================


plt.figure(

    figsize=(10,6)

)


plt.plot(


    list(k_values),


    silhouette_values,


    marker="o"


)


plt.xlabel(

    "Number of Clusters (K)"

)


plt.ylabel(

    "Silhouette Score"

)


plt.title(

    "Silhouette Score for Different K Values"

)


plt.xticks(

    list(k_values)

)


plt.grid(

    True

)


plt.show()



# ============================================================

# 34. SELECT BEST K USING SILHOUETTE SCORE

# ============================================================


best_k_index = np.argmax(

    silhouette_values

)


best_k = list(

    k_values

)[best_k_index]



best_silhouette = silhouette_values[

    best_k_index

]



print(

    "\n============================================"

)


print(

    "             BEST K VALUE"

)


print(

    "============================================"

)


print(

    "Best K:",

    best_k

)


print(

    "Silhouette Score:",

    round(

        best_silhouette,

        4

    )

)



# ============================================================

# 35. CREATE FINAL K-MEANS MODEL

# ============================================================


kmeans_final = KMeans(


    n_clusters=best_k,


    random_state=42,


    n_init=10


)



# ============================================================

# 36. FIT K-MEANS

# ============================================================


cluster_labels = kmeans_final.fit_predict(


    X_processed


)



print(

    "\nK-Means Clustering Completed!"

)



# ============================================================

# 37. ADD CLUSTER LABELS TO DATASET

# ============================================================


df_cluster["Cluster"] = cluster_labels



# ============================================================

# 38. CLUSTER COUNTS

# ============================================================


cluster_counts = (


    df_cluster["Cluster"]

    .value_counts()

    .sort_index()


)



print(

    "\nNumber of Houses in Each Cluster:"

)


print(

    cluster_counts

)



# ============================================================

# 39. CLUSTER DISTRIBUTION GRAPH

# ============================================================


plt.figure(

    figsize=(9,6)

)


sns.countplot(


    x="Cluster",


    data=df_cluster,


    order=sorted(

        df_cluster["Cluster"].unique()

    )


)


plt.title(

    "Number of Houses in Each K-Means Cluster"

)


plt.xlabel(

    "Cluster"

)


plt.ylabel(

    "Number of Houses"

)


plt.show()



# ============================================================

# 40. FINAL SILHOUETTE SCORE

# ============================================================


final_silhouette = silhouette_score(


    X_processed,


    cluster_labels


)



print(

    "\nFinal Silhouette Score:"

)


print(

    round(

        final_silhouette,

        4

    )

)



# ============================================================

# 41. CLUSTER CENTERS

# ============================================================


cluster_centers = (


    pd.DataFrame(

        kmeans_final.cluster_centers_

    )


)



print(

    "\nCluster Centers:"

)


display(

    cluster_centers

)



# ============================================================

# 42. CLUSTER-WISE HOUSE PRICE ANALYSIS

# ============================================================

#

# MEDV was NOT used to create the clusters.

#

# We now use MEDV only to understand what the clusters

# represent after clustering.

# ============================================================


cluster_price_summary = (


    df_cluster

    .groupby("Cluster")[target]

    .agg([

        "count",

        "mean",

        "median",

        "min",

        "max"

    ])

    .round(2)


)



print(

    "\nCluster-Wise House Price Summary:"

)


display(

    cluster_price_summary

)



# ============================================================

# 43. CLUSTER-WISE FEATURE MEANS

# ============================================================


cluster_feature_summary = (


    df_cluster

    .groupby("Cluster")[

        feature_columns

    ]

    .mean()

    .round(2)


)



print(

    "\nCluster-Wise Feature Means:"

)


display(

    cluster_feature_summary

)



# ============================================================

# 44. HEATMAP OF CLUSTER FEATURE MEANS

# ============================================================


plt.figure(

    figsize=(15,7)

)


sns.heatmap(


    cluster_feature_summary,


    annot=True,


    fmt=".2f",


    cmap="coolwarm"


)


plt.title(

    "Cluster-Wise Feature Means"

)


plt.xlabel(

    "Features"

)


plt.ylabel(

    "Cluster"

)


plt.show()



# ============================================================

# 45. CLUSTER VS MEDV

# ============================================================


plt.figure(

    figsize=(10,6)

)


sns.boxplot(


    x="Cluster",


    y=target,


    data=df_cluster


)


plt.title(

    "House Price Distribution by K-Means Cluster"

)


plt.xlabel(

    "Cluster"

)


plt.ylabel(

    "MEDV"

)


plt.show()



# ============================================================

# 46. CLUSTER VS MEDV BAR CHART

# ============================================================


cluster_price_mean = (


    df_cluster

    .groupby("Cluster")[target]

    .mean()


)



plt.figure(

    figsize=(9,6)

)


plt.bar(


    cluster_price_mean.index,


    cluster_price_mean.values


)


plt.xlabel(

    "Cluster"

)


plt.ylabel(

    "Average MEDV"

)


plt.title(

    "Average House Price by Cluster"

)


plt.show()



# ============================================================

# 47. PCA FOR VISUALIZATION

# ============================================================

#

# K-Means may use many features.

#

# PCA reduces the processed data to two dimensions

# so we can visualize the clusters.

# ============================================================


pca = PCA(

    n_components=2

)



X_pca = pca.fit_transform(

    X_processed

)



print(

    "\nPCA Explained Variance:"

)


print(

    pca.explained_variance_ratio_

)



print(

    "\nTotal Variance Explained:"

)


print(

    round(

        pca.explained_variance_ratio_.sum(),

        4

    )

)



# ============================================================

# 48. CREATE PCA DATAFRAME

# ============================================================


pca_df = pd.DataFrame({


    "PC1": X_pca[:, 0],


    "PC2": X_pca[:, 1],


    "Cluster": cluster_labels


})



# ============================================================

# 49. PCA CLUSTER VISUALIZATION

# ============================================================


plt.figure(

    figsize=(10,7)

)


sns.scatterplot(


    data=pca_df,


    x="PC1",


    y="PC2",


    hue="Cluster",


    palette="tab10",


    s=70


)


plt.title(

    "K-Means Clusters Visualized Using PCA"

)


plt.xlabel(

    "Principal Component 1"

)


plt.ylabel(

    "Principal Component 2"

)


plt.legend(

    title="Cluster"

)


plt.show()



# ============================================================

# 50. PCA WITH CLUSTER CENTERS

# ============================================================


centers_pca = pca.transform(


    kmeans_final.cluster_centers_


)



plt.figure(

    figsize=(10,7)

)


sns.scatterplot(


    data=pca_df,


    x="PC1",


    y="PC2",


    hue="Cluster",


    palette="tab10",


    s=60


)



plt.scatter(


    centers_pca[:, 0],


    centers_pca[:, 1],


    marker="X",


    s=250,


    color="black",


    label="Cluster Centers"


)



plt.title(

    "K-Means Clusters and Cluster Centers"

)


plt.xlabel(

    "Principal Component 1"

)


plt.ylabel(

    "Principal Component 2"

)


plt.legend()


plt.show()



# ============================================================

# 51. OPTIONAL: CREATE LOW/HIGH PRICE REFERENCE

# ============================================================

#

# This is NOT used for training K-Means.

#

# It is only used to examine how the unsupervised clusters

# correspond to the known MEDV-based categories.

# ============================================================


median_price = df_cluster[target].median()



df_cluster["Price_Class"] = (


    df_cluster[target] >= median_price


).astype(int)



df_cluster["Price_Class_Name"] = (


    df_cluster["Price_Class"]

    .map({


        0: "Low Price",


        1: "High Price"


    })


)



# ============================================================

# 52. CLUSTER VS PRICE CLASS

# ============================================================


cluster_price_class = pd.crosstab(


    df_cluster["Cluster"],


    df_cluster["Price_Class_Name"]


)



print(

    "\nCluster vs Price Class:"

)


display(

    cluster_price_class

)



# ============================================================

# 53. CLUSTER VS PRICE CLASS GRAPH

# ============================================================


cluster_price_class.plot(


    kind="bar",


    figsize=(10,6)


)


plt.title(

    "Price Class Distribution Within Each Cluster"

)


plt.xlabel(

    "Cluster"

)


plt.ylabel(

    "Number of Houses"

)


plt.xticks(

    rotation=0

)


plt.show()



# ============================================================

# 54. OPTIONAL UNSUPERVISED AGREEMENT METRICS

# ============================================================

#

# These metrics are NOT training metrics.

#

# They compare the discovered clusters against the

# MEDV-based reference classes only for educational analysis.

# ============================================================


ari = adjusted_rand_score(


    df_cluster["Price_Class"],


    cluster_labels


)



nmi = normalized_mutual_info_score(


    df_cluster["Price_Class"],


    cluster_labels


)



print(

    "\n============================================"

)


print(

    "      OPTIONAL CLUSTER AGREEMENT"

)


print(

    "============================================"

)


print(

    "Adjusted Rand Index:",

    round(

        ari,

        4

    )

)


print(

    "Normalized Mutual Information:",

    round(

        nmi,

        4

    )

)



# ============================================================

# 55. SAVE FINAL CLUSTERED DATA

# ============================================================


final_clustered_data = df_cluster.copy()



print(

    "\nFinal Clustered Dataset:"

)


display(

    final_clustered_data.head()

)



# ============================================================

# 56. FINAL SUMMARY TABLE

# ============================================================


summary_table = pd.DataFrame({


    "Item": [


        "Algorithm",


        "Number of Clusters",


        "Silhouette Score",


        "Number of Features",


        "Number of Houses",


        "PCA Components"


    ],


    "Value": [


        "K-Means",


        best_k,


        round(

            final_silhouette,

            4

        ),


        len(feature_columns),


        len(df_cluster),


        2


    ]


})



print(

    "\n============================================"

)


print(

    "             K-MEANS SUMMARY"

)


print(

    "============================================"

)


display(

    summary_table

)



# ============================================================

# 57. CLUSTER INTERPRETATION

# ============================================================


print("""

============================================================

K-MEANS CLUSTERING - THEORY

============================================================


K-Means is an UNSUPERVISED MACHINE LEARNING algorithm.


Unlike Linear Regression, Logistic Regression and KNN

classification, K-Means does not require a target variable

during clustering.


The algorithm groups similar observations into K clusters.


============================================================


HOW K-MEANS WORKS

============================================================


Step 1:

Choose the number of clusters K.


Step 2:

Randomly initialize K cluster centers.


Step 3:

Calculate the distance between each observation

and every cluster center.


Step 4:

Assign each observation to the nearest cluster.


Step 5:

Recalculate the cluster centers.


Step 6:

Repeat the process until the cluster assignments

stabilize.


============================================================


ELBOW METHOD

============================================================


The Elbow Method calculates:


INERTIA


Inertia represents the within-cluster sum of squared

distances.


As K increases, inertia normally decreases.


The point where the reduction begins to slow down

is called the "elbow".


============================================================


SILHOUETTE SCORE

============================================================


Silhouette Score measures how well observations fit

their assigned clusters.


Higher values generally indicate better separation

and cohesion.


The score ranges approximately from:


-1 to +1


A value closer to +1 indicates better-defined clusters.


============================================================


IMPORTANT BOSTON HOUSING POINT

============================================================


MEDV was NOT used to create the clusters.


K-Means used the housing characteristics/features.


MEDV was used AFTER clustering only to understand

the characteristics of the resulting clusters.


Therefore this is an UNSUPERVISED learning experiment.


============================================================


PREPROCESSING USED

============================================================


1. Missing-value checking

2. Missing-value imputation

3. Duplicate detection

4. Duplicate removal

5. IQR outlier detection

6. IQR outlier capping

7. Feature scaling

8. Categorical encoding

9. PCA visualization


============================================================

""")

kMEDOID

 # ================================================================

# K-MEDOIDS CLUSTERING - BOSTON HOUSING DATASET

# Complete Google Colab Single-Cell Code

# ================================================================


# -----------------------------

# 1. INSTALL REQUIRED LIBRARY

# -----------------------------

!pip install scikit-learn-extra -q


# -----------------------------

# 2. IMPORT LIBRARIES

# -----------------------------

import numpy as np

import pandas as pd

import matplotlib.pyplot as plt


from sklearn.datasets import fetch_openml

from sklearn.preprocessing import StandardScaler

from sklearn.decomposition import PCA

from sklearn.metrics import silhouette_score


from sklearn_extra.cluster import KMedoids


import warnings

warnings.filterwarnings("ignore")


# -----------------------------

# 3. LOAD BOSTON HOUSING DATA

# -----------------------------

print("=" * 70)

print("BOSTON HOUSING DATASET - K-MEDOIDS CLUSTERING")

print("=" * 70)


boston = fetch_openml(

    name="boston",

    version=1,

    as_frame=True

)


df = boston.frame.copy()


print("\nDataset Shape:", df.shape)


# -----------------------------

# 4. DISPLAY DATA

# -----------------------------

print("\nFirst 5 Rows:")

display(df.head())


print("\nDataset Information:")

print(df.info())


# -----------------------------

# 5. CONVERT NUMERIC COLUMNS

# -----------------------------

df = df.apply(pd.to_numeric, errors="coerce")


# -----------------------------

# 6. CHECK MISSING VALUES

# -----------------------------

print("\nMissing Values:")

print(df.isnull().sum())


# -----------------------------

# 7. HANDLE MISSING VALUES

# -----------------------------

df = df.dropna()


print("\nShape After Removing Missing Values:")

print(df.shape)


# -----------------------------

# 8. CHECK DUPLICATES

# -----------------------------

print("\nDuplicate Rows:")

print(df.duplicated().sum())


# Remove duplicates

df = df.drop_duplicates()


print("Shape After Removing Duplicates:")

print(df.shape)


# -----------------------------

# 9. SEPARATE FEATURES

# -----------------------------

# MEDV is the house-price target.

# It will NOT be used for clustering.


X = df.drop("MEDV", axis=1)


y = df["MEDV"]


print("\nFeatures Used for K-Medoids:")

print(X.columns.tolist())


print("\nNumber of Features:", X.shape[1])


# -----------------------------

# 10. STANDARDIZATION

# -----------------------------

scaler = StandardScaler()


X_scaled = scaler.fit_transform(X)


X_scaled = pd.DataFrame(

    X_scaled,

    columns=X.columns

)


print("\nScaled Data:")

display(X_scaled.head())


# -----------------------------

# 11. CHECK SCALED DATA

# -----------------------------

print("\nMean After Scaling:")

print(X_scaled.mean().round(2))


print("\nStandard Deviation After Scaling:")

print(X_scaled.std().round(2))


# -----------------------------

# 12. FIND BEST K

# USING SILHOUETTE SCORE

# -----------------------------

silhouette_scores = []


print("\n" + "=" * 70)

print("SILHOUETTE ANALYSIS")

print("=" * 70)


for k in range(2, 11):


    model = KMedoids(

        n_clusters=k,

        metric="euclidean",

        method="alternate",

        init="k-medoids++",

        max_iter=300,

        random_state=42

    )


    labels = model.fit_predict(X_scaled)


    score = silhouette_score(

        X_scaled,

        labels

    )


    silhouette_scores.append(score)


    print(

        "K =", k,

        "| Silhouette Score =",

        round(score, 4)

    )


# -----------------------------

# 13. PLOT SILHOUETTE SCORE

# -----------------------------

plt.figure(figsize=(9, 5))


plt.plot(

    range(2, 11),

    silhouette_scores,

    marker="o"

)


plt.xlabel("Number of Clusters (K)")

plt.ylabel("Silhouette Score")

plt.title("K-Medoids - Silhouette Score")


plt.xticks(range(2, 11))

plt.grid(True)


plt.show()


# -----------------------------

# 14. AUTOMATICALLY SELECT K

# -----------------------------

best_k = list(range(2, 11))[

    np.argmax(silhouette_scores)

]


print("\nBest K According to Silhouette Score:", best_k)


# -----------------------------

# 15. TRAIN FINAL K-MEDOIDS

# -----------------------------

kmedoids = KMedoids(

    n_clusters=best_k,

    metric="euclidean",

    method="alternate",

    init="k-medoids++",

    max_iter=300,

    random_state=42

)


kmedoids.fit(X_scaled)


# -----------------------------

# 16. GET CLUSTER LABELS

# -----------------------------

cluster_labels = kmedoids.labels_


df_result = df.copy()


df_result["Cluster"] = cluster_labels


print("\n" + "=" * 70)

print("K-MEDOIDS CLUSTERING RESULT")

print("=" * 70)


print("\nFirst 10 Cluster Assignments:")

display(df_result.head(10))


# -----------------------------

# 17. CLUSTER COUNTS

# -----------------------------

cluster_counts = (

    df_result["Cluster"]

    .value_counts()

    .sort_index()

)


print("\nNumber of Observations in Each Cluster:")

print(cluster_counts)


# -----------------------------

# 18. CLUSTER SIZE GRAPH

# -----------------------------

plt.figure(figsize=(8, 5))


cluster_counts.plot(

    kind="bar"

)


plt.xlabel("Cluster")

plt.ylabel("Number of Houses")

plt.title("Number of Houses in Each K-Medoids Cluster")


plt.xticks(rotation=0)

plt.grid(axis="y")


plt.show()


# -----------------------------

# 19. MEDOID INDICES

# -----------------------------

medoid_indices = kmedoids.medoid_indices_


print("\nMedoid Indices:")

print(medoid_indices)


# -----------------------------

# 20. ACTUAL MEDOID OBSERVATIONS

# -----------------------------

medoids = df.iloc[

    medoid_indices

].copy()


medoids["Cluster"] = range(best_k)


print("\nActual Medoid Observations:")

display(medoids)


# -----------------------------

# 21. MEDOID HOUSE VALUES

# -----------------------------

print("\nMEDV of Medoid Observations:")


for cluster_number, index in enumerate(medoid_indices):


    print(

        "Cluster",

        cluster_number,

        "| MEDV =",

        round(df.iloc[index]["MEDV"], 2)

    )


# -----------------------------

# 22. SILHOUETTE SCORE

# -----------------------------

final_silhouette = silhouette_score(

    X_scaled,

    cluster_labels

)


print("\nFinal Silhouette Score:")

print(round(final_silhouette, 4))


# -----------------------------

# 23. K-MEDOIDS INERTIA

# -----------------------------

print("\nK-Medoids Inertia:")

print(round(kmedoids.inertia_, 4))


# -----------------------------

# 24. PCA DIMENSION REDUCTION

# -----------------------------

pca = PCA(

    n_components=2

)


X_pca = pca.fit_transform(

    X_scaled

)


print("\nPCA Explained Variance:")

print(

    pca.explained_variance_ratio_

)


print(

    "\nTotal Explained Variance:",

    round(

        sum(pca.explained_variance_ratio_) * 100,

        2

    ),

    "%"

)


# -----------------------------

# 25. PCA CLUSTER VISUALIZATION

# -----------------------------

plt.figure(figsize=(10, 7))


scatter = plt.scatter(

    X_pca[:, 0],

    X_pca[:, 1],

    c=cluster_labels,

    s=45,

    alpha=0.7

)


# PCA coordinates of actual medoids

medoid_pca = X_pca[

    medoid_indices

]


plt.scatter(

    medoid_pca[:, 0],

    medoid_pca[:, 1],

    marker="X",

    s=250,

    edgecolors="black",

    linewidths=2,

    label="Medoids"

)


plt.xlabel("Principal Component 1")

plt.ylabel("Principal Component 2")


plt.title(

    "K-Medoids Clustering - Boston Housing"

)


plt.colorbar(

    scatter,

    label="Cluster"

)


plt.legend()

plt.grid(True)


plt.show()


# -----------------------------

# 26. CLUSTER-WISE ANALYSIS

# -----------------------------

cluster_analysis = (

    df_result

    .groupby("Cluster")

    .mean(numeric_only=True)

)


print("\n" + "=" * 70)

print("CLUSTER-WISE FEATURE ANALYSIS")

print("=" * 70)


display(

    cluster_analysis.round(2)

)


# -----------------------------

# 27. HOUSE PRICE ANALYSIS

# MEDV WAS NOT USED FOR TRAINING

# -----------------------------

cluster_report = (

    df_result

    .groupby("Cluster")

    .agg(

        Number_of_Houses=("MEDV", "count"),

        Average_MEDV=("MEDV", "mean"),

        Minimum_MEDV=("MEDV", "min"),

        Maximum_MEDV=("MEDV", "max")

    )

)


print("\nCluster House-Value Analysis:")

display(

    cluster_report.round(2)

)


# -----------------------------

# 28. AVERAGE MEDV GRAPH

# -----------------------------

plt.figure(figsize=(8, 5))


cluster_report[

    "Average_MEDV"

].plot(

    kind="bar"

)


plt.xlabel("Cluster")

plt.ylabel("Average MEDV")

plt.title(

    "Average House Value by K-Medoids Cluster"

)


plt.xticks(rotation=0)

plt.grid(axis="y")


plt.show()


# -----------------------------

# 29. FINAL SUMMARY

# -----------------------------

print("\n" + "=" * 70)

print("FINAL K-MEDOIDS SUMMARY")

print("=" * 70)


print("Dataset Size       :", df.shape)

print("Number of Features :", X.shape[1])

print("Selected K          :", best_k)

print("Silhouette Score    :", round(final_silhouette, 4))

print("Inertia             :", round(kmedoids.inertia_, 4))


print("\nCluster Sizes:")

print(cluster_counts)


print("\nMedoid Indices:")

print(medoid_indices)


print("\nK-Medoids completed successfully.")