Total Pageviews

Monday, September 28, 2026

KNN

 # ============================================================

# BOSTON HOUSING DATASET

# K-NEAREST NEIGHBORS (KNN)

# COMPLETE DATA PREPROCESSING + EVALUATION

# GOOGLE COLAB VERSION

# ============================================================



# ============================================================

# 1. IMPORT LIBRARIES

# ============================================================


import numpy as np

import pandas as pd


import matplotlib.pyplot as plt

import seaborn as sns


from sklearn.datasets import fetch_openml


from sklearn.model_selection import (

    train_test_split,

    cross_val_score

)


from sklearn.compose import ColumnTransformer


from sklearn.pipeline import Pipeline


from sklearn.impute import SimpleImputer


from sklearn.preprocessing import (

    StandardScaler,

    OneHotEncoder

)


from sklearn.neighbors import KNeighborsClassifier


from sklearn.metrics import (

    accuracy_score,

    precision_score,

    recall_score,

    f1_score,

    confusion_matrix,

    classification_report,

    ConfusionMatrixDisplay,

    roc_curve,

    roc_auc_score

)


import warnings


warnings.filterwarnings("ignore")



# ============================================================

# 2. LOAD BOSTON HOUSING DATASET

# ============================================================


boston = fetch_openml(

    name="boston",

    version=1,

    as_frame=True

)


df = boston.frame.copy()


print("Boston Housing Dataset Loaded Successfully!")



# ============================================================

# 3. DISPLAY FIRST 5 ROWS

# ============================================================


print("\nFirst 5 Rows:")


display(

    df.head()

)



# ============================================================

# 4. DATASET SHAPE

# ============================================================


print("\nDataset Shape:")


print(

    df.shape

)



# ============================================================

# 5. COLUMN NAMES

# ============================================================


print("\nColumn Names:")


print(

    df.columns.tolist()

)



# ============================================================

# 6. DATA INFORMATION

# ============================================================


print("\nDataset Information:")


df.info()



# ============================================================

# 7. DATA TYPES

# ============================================================


print("\nData Types:")


print(

    df.dtypes

)



# ============================================================

# 8. MISSING VALUES

# ============================================================


print("\nMissing Values:")


print(

    df.isnull().sum()

)



# ============================================================

# 9. MISSING VALUE PERCENTAGE

# ============================================================


missing_percentage = (


    df.isnull()

      .mean()

      .mul(100)

      .sort_values(

          ascending=False

      )


)


print("\nMissing Value Percentage:")


print(

    missing_percentage

)



# ============================================================

# 10. DUPLICATE VALUES

# ============================================================


print(

    "\nNumber of Duplicate Rows:",

    df.duplicated().sum()

)



# ============================================================

# 11. REMOVE DUPLICATES

# ============================================================


df = df.drop_duplicates()


print(

    "\nShape After Duplicate Removal:"

)


print(

    df.shape

)



# ============================================================

# 12. STATISTICAL SUMMARY

# ============================================================


print(

    "\nStatistical Summary:"

)


display(

    df.describe()

)



# ============================================================

# 13. TARGET VARIABLE

# ============================================================


target = "MEDV"


print(

    "\nTarget Variable:",

    target

)



# ============================================================

# 14. TARGET DISTRIBUTION

# ============================================================


plt.figure(

    figsize=(9,5)

)


sns.histplot(

    df[target],

    kde=True

)


plt.title(

    "Distribution of Boston House Prices"

)


plt.xlabel(

    "MEDV"

)


plt.ylabel(

    "Frequency"

)


plt.show()



# ============================================================

# 15. OUTLIER DETECTION

# ============================================================


numeric_columns = (


    df.select_dtypes(

        include=np.number

    ).columns


)



plt.figure(

    figsize=(14,6)

)


sns.boxplot(

    data=df[

        numeric_columns

    ]

)


plt.title(

    "Boxplot of Boston Housing Features"

)


plt.xticks(

    rotation=45

)


plt.show()



# ============================================================

# 16. IQR OUTLIER ANALYSIS

# ============================================================


print(

    "\n============================================"

)


print(

    "          IQR OUTLIER ANALYSIS"

)


print(

    "============================================"

)



for column in numeric_columns:


    Q1 = df[column].quantile(

        0.25

    )


    Q3 = df[column].quantile(

        0.75

    )


    IQR = Q3 - Q1


    lower_bound = (

        Q1 - 1.5 * IQR

    )


    upper_bound = (

        Q3 + 1.5 * IQR

    )


    outliers = df[

        (df[column] < lower_bound)

        |

        (df[column] > upper_bound)

    ]


    print(

        column,

        "→",

        len(outliers),

        "outliers"

    )



# ============================================================

# 17. OUTLIER CAPPING

# ============================================================


df_capped = df.copy()


for column in numeric_columns:


    Q1 = df_capped[column].quantile(

        0.25

    )


    Q3 = df_capped[column].quantile(

        0.75

    )


    IQR = Q3 - Q1


    lower_bound = (

        Q1 - 1.5 * IQR

    )


    upper_bound = (

        Q3 + 1.5 * IQR

    )


    df_capped[column] = (


        df_capped[column]

        .clip(

            lower_bound,

            upper_bound

        )


    )



print(

    "\nIQR Outlier Capping Completed!"

)



# ============================================================

# 18. CORRELATION MATRIX

# ============================================================


correlation = df.corr(

    numeric_only=True

)


print(

    "\nCorrelation with MEDV:"

)


print(

    correlation[target]

    .sort_values(

        ascending=False

    )

)



# ============================================================

# 19. CORRELATION HEATMAP

# ============================================================


plt.figure(

    figsize=(12,9)

)


sns.heatmap(

    correlation,

    annot=True,

    fmt=".2f",

    cmap="coolwarm"

)


plt.title(

    "Boston Housing Correlation Matrix"

)


plt.show()



# ============================================================

# 20. CONVERT MEDV INTO CLASS

# ============================================================

#

# Original MEDV is continuous.

#

# KNN classification requires classes.

#

# 0 = Low Price

# 1 = High Price

#

# Median MEDV is used as threshold.

# ============================================================


median_price = df[target].median()


print(

    "\nMedian House Price:",

    median_price

)



df["Price_Class"] = (


    df[target] >= median_price


).astype(int)



# ============================================================

# 21. DISPLAY CLASS DISTRIBUTION

# ============================================================


print(

    "\nPrice Class Distribution:"

)


print(

    df["Price_Class"].value_counts()

)



# ============================================================

# 22. CLASS DISTRIBUTION GRAPH

# ============================================================


plt.figure(

    figsize=(7,5)

)


sns.countplot(

    x="Price_Class",

    data=df

)


plt.title(

    "Low Price vs High Price Houses"

)


plt.xlabel(

    "Price Class"

)


plt.ylabel(

    "Number of Houses"

)


plt.xticks(

    [0,1],

    [

        "Low Price",

        "High Price"

    ]

)


plt.show()



# ============================================================

# 23. FEATURES AND TARGET

# ============================================================


X = df.drop(

    columns=[

        "MEDV",

        "Price_Class"

    ]

)


y = df["Price_Class"]



print(

    "\nFeature Shape:",

    X.shape

)


print(

    "Target Shape:",

    y.shape

)



# ============================================================

# 24. IDENTIFY NUMERICAL FEATURES

# ============================================================


numeric_features = (


    X.select_dtypes(

        include=np.number

    )

    .columns

    .tolist()


)



categorical_features = (


    X.select_dtypes(

        exclude=np.number

    )

    .columns

    .tolist()


)



print(

    "\nNumerical Features:"

)


print(

    numeric_features

)


print(

    "\nCategorical Features:"

)


print(

    categorical_features

)



# ============================================================

# 25. TRAIN TEST SPLIT

# ============================================================


X_train, X_test, y_train, y_test = (


    train_test_split(


        X,

        y,


        test_size=0.20,


        random_state=42,


        stratify=y


    )


)



print(

    "\nTraining Shape:",

    X_train.shape

)


print(

    "Testing Shape:",

    X_test.shape

)



# ============================================================

# 26. NUMERICAL PREPROCESSING

# ============================================================


numeric_pipeline = Pipeline(


    steps=[


        (

            "imputer",


            SimpleImputer(

                strategy="median"

            )

        ),


        (

            "scaler",


            StandardScaler()

        )


    ]


)



# ============================================================

# 27. CATEGORICAL PREPROCESSING

# ============================================================


categorical_pipeline = Pipeline(


    steps=[


        (

            "imputer",


            SimpleImputer(

                strategy="most_frequent"

            )

        ),


        (

            "encoder",


            OneHotEncoder(

                handle_unknown="ignore"

            )

        )


    ]


)



# ============================================================

# 28. COMBINE PREPROCESSING

# ============================================================


preprocessor = ColumnTransformer(


    transformers=[


        (

            "numeric",


            numeric_pipeline,


            numeric_features

        ),


        (

            "categorical",


            categorical_pipeline,


            categorical_features

        )


    ]


)



# ============================================================

# 29. KNN MODEL

# ============================================================


knn_model = Pipeline(


    steps=[


        (

            "preprocessing",


            preprocessor

        ),


        (

            "model",


            KNeighborsClassifier(

                n_neighbors=5

            )

        )


    ]


)



# ============================================================

# 30. TRAIN KNN MODEL

# ============================================================


knn_model.fit(


    X_train,


    y_train


)



print(

    "\nKNN Model Trained Successfully!"

)



# ============================================================

# 31. PREDICTION

# ============================================================


y_pred = knn_model.predict(

    X_test

)



# ============================================================

# 32. PREDICT PROBABILITY

# ============================================================


y_probability = (


    knn_model

    .predict_proba(

        X_test

    )[:, 1]


)



# ============================================================

# 33. ACCURACY

# ============================================================


accuracy = accuracy_score(


    y_test,


    y_pred


)



# ============================================================

# 34. PRECISION

# ============================================================


precision = precision_score(


    y_test,


    y_pred,


    zero_division=0


)



# ============================================================

# 35. RECALL

# ============================================================


recall = recall_score(


    y_test,


    y_pred,


    zero_division=0


)



# ============================================================

# 36. F1 SCORE

# ============================================================


f1 = f1_score(


    y_test,


    y_pred,


    zero_division=0


)



# ============================================================

# 37. ROC-AUC

# ============================================================


roc_auc = roc_auc_score(


    y_test,


    y_probability


)



# ============================================================

# 38. DISPLAY MODEL RESULTS

# ============================================================


print(

    "\n============================================"

)


print(

    "             KNN RESULTS"

)


print(

    "============================================"

)


print(

    "Accuracy  :",

    round(accuracy, 4)

)


print(

    "Precision :",

    round(precision, 4)

)


print(

    "Recall    :",

    round(recall, 4)

)


print(

    "F1 Score  :",

    round(f1, 4)

)


print(

    "ROC-AUC   :",

    round(roc_auc, 4)

)


print(

    "============================================"

)



# ============================================================

# 39. CONFUSION MATRIX

# ============================================================


cm = confusion_matrix(


    y_test,


    y_pred


)



print(

    "\nConfusion Matrix:"

)


print(

    cm

)



# ============================================================

# 40. TRUE NEGATIVE / FALSE POSITIVE /

#     FALSE NEGATIVE / TRUE POSITIVE

# ============================================================


TN, FP, FN, TP = cm.ravel()



print(

    "\nTrue Negative (TN):",

    TN

)


print(

    "False Positive (FP):",

    FP

)


print(

    "False Negative (FN):",

    FN

)


print(

    "True Positive (TP):",

    TP

)



# ============================================================

# 41. CONFUSION MATRIX GRAPH

# ============================================================


plt.figure(

    figsize=(7,6)

)


sns.heatmap(


    cm,


    annot=True,


    fmt="d",


    cmap="Blues",


    xticklabels=[

        "Predicted Low",

        "Predicted High"

    ],


    yticklabels=[

        "Actual Low",

        "Actual High"

    ]


)


plt.title(

    "KNN Confusion Matrix"

)


plt.xlabel(

    "Predicted Class"

)


plt.ylabel(

    "Actual Class"

)


plt.show()



# ============================================================

# 42. CLASSIFICATION REPORT

# ============================================================


print(

    "\n============================================"

)


print(

    "          CLASSIFICATION REPORT"

)


print(

    "============================================"

)


print(


    classification_report(


        y_test,


        y_pred,


        target_names=[

            "Low Price",

            "High Price"

        ],


        zero_division=0


    )


)



# ============================================================

# 43. PERFORMANCE TABLE

# ============================================================


metrics_table = pd.DataFrame({


    "Metric": [


        "Accuracy",


        "Precision",


        "Recall",


        "F1 Score",


        "ROC-AUC"


    ],


    "Score": [


        accuracy,


        precision,


        recall,


        f1,


        roc_auc


    ]


})



metrics_table["Score"] = (


    metrics_table["Score"]

    .round(4)


)



print(

    "\nKNN Performance Metrics:"

)


display(

    metrics_table

)



# ============================================================

# 44. CROSS VALIDATION

# ============================================================


cv_scores = cross_val_score(


    knn_model,


    X,


    y,


    cv=5,


    scoring="accuracy"


)



print(

    "\n5-Fold Cross Validation Accuracy:"

)


print(

    cv_scores

)



print(

    "\nMean Cross Validation Accuracy:"

)


print(

    round(

        cv_scores.mean(),

        4

    )

)



# ============================================================

# 45. TEST DIFFERENT K VALUES

# ============================================================


k_values = range(

    1,

    21

)


k_accuracy = []



for k in k_values:


    temp_knn = Pipeline(


        steps=[


            (

                "preprocessing",


                preprocessor

            ),


            (

                "model",


                KNeighborsClassifier(

                    n_neighbors=k

                )


            )


        ]


    )



    temp_knn.fit(

        X_train,

        y_train

    )



    temp_prediction = temp_knn.predict(

        X_test

    )



    score = accuracy_score(


        y_test,


        temp_prediction


    )



    k_accuracy.append(

        score

    )



# ============================================================

# 46. K VALUE GRAPH

# ============================================================


plt.figure(

    figsize=(10,6)

)


plt.plot(


    list(k_values),


    k_accuracy,


    marker="o"


)


plt.xlabel(

    "Number of Neighbors (K)"

)


plt.ylabel(

    "Accuracy"

)


plt.title(

    "KNN Accuracy for Different K Values"

)


plt.xticks(

    list(k_values)

)


plt.grid(

    True

)


plt.show()



# ============================================================

# 47. BEST K VALUE

# ============================================================


best_k_index = np.argmax(

    k_accuracy

)


best_k = list(

    k_values

)[best_k_index]


best_k_accuracy = k_accuracy[

    best_k_index

]



print(

    "\n============================================"

)


print(

    "             BEST K VALUE"

)


print(

    "============================================"

)


print(

    "Best K:",

    best_k

)


print(

    "Accuracy:",

    round(

        best_k_accuracy,

        4

    )

)



# ============================================================

# 48. TRAIN FINAL KNN USING BEST K

# ============================================================


final_knn = Pipeline(


    steps=[


        (

            "preprocessing",


            preprocessor

        ),


        (

            "model",


            KNeighborsClassifier(


                n_neighbors=best_k


            )


        )


    ]


)



final_knn.fit(


    X_train,


    y_train


)



final_prediction = final_knn.predict(

    X_test

)



final_probability = (


    final_knn

    .predict_proba(

        X_test

    )[:, 1]


)



# ============================================================

# 49. FINAL METRICS

# ============================================================


final_accuracy = accuracy_score(


    y_test,


    final_prediction


)



final_precision = precision_score(


    y_test,


    final_prediction,


    zero_division=0


)



final_recall = recall_score(


    y_test,


    final_prediction,


    zero_division=0


)



final_f1 = f1_score(


    y_test,


    final_prediction,


    zero_division=0


)



final_auc = roc_auc_score(


    y_test,


    final_probability


)



print(

    "\n============================================"

)


print(

    "       FINAL KNN MODEL RESULTS"

)


print(

    "============================================"

)


print(

    "Best K       :",

    best_k

)


print(

    "Accuracy     :",

    round(

        final_accuracy,

        4

    )

)


print(

    "Precision    :",

    round(

        final_precision,

        4

    )

)


print(

    "Recall       :",

    round(

        final_recall,

        4

    )

)


print(

    "F1 Score     :",

    round(

        final_f1,

        4

    )

)


print(

    "ROC-AUC      :",

    round(

        final_auc,

        4

    )

)


print(

    "============================================"

)



# ============================================================

# 50. FINAL CONFUSION MATRIX

# ============================================================


final_cm = confusion_matrix(


    y_test,


    final_prediction


)



plt.figure(

    figsize=(7,6)

)


sns.heatmap(


    final_cm,


    annot=True,


    fmt="d",


    cmap="Blues",


    xticklabels=[

        "Predicted Low",

        "Predicted High"

    ],


    yticklabels=[

        "Actual Low",

        "Actual High"

    ]


)


plt.title(

    f"Final KNN Confusion Matrix (K={best_k})"

)


plt.xlabel(

    "Predicted Class"

)


plt.ylabel(

    "Actual Class"

)


plt.show()



# ============================================================

# 51. FINAL CLASSIFICATION REPORT

# ============================================================


print(

    "\nFinal Classification Report:"

)


print(


    classification_report(


        y_test,


        final_prediction,


        target_names=[

            "Low Price",

            "High Price"

        ],


        zero_division=0


    )


)



# ============================================================

# 52. ROC CURVE

# ============================================================


fpr, tpr, thresholds = roc_curve(


    y_test,


    final_probability


)



plt.figure(

    figsize=(8,6)

)


plt.plot(


    fpr,


    tpr,


    label=f"KNN ROC-AUC = {final_auc:.4f}"


)


plt.plot(


    [0,1],


    [0,1],


    linestyle="--"


)


plt.xlabel(

    "False Positive Rate"

)


plt.ylabel(

    "True Positive Rate"

)


plt.title(

    "KNN ROC Curve"

)


plt.legend()


plt.show()



# ============================================================

# 53. ACTUAL VS PREDICTED

# ============================================================


prediction_table = pd.DataFrame({


    "Actual Class":

        y_test.values,


    "Predicted Class":

        final_prediction,


    "High Price Probability":

        final_probability


})



prediction_table["Actual Class"] = (


    prediction_table[

        "Actual Class"

    ].map({


        0: "Low Price",


        1: "High Price"


    })


)



prediction_table["Predicted Class"] = (


    prediction_table[

        "Predicted Class"

    ].map({


        0: "Low Price",


        1: "High Price"


    })


)



prediction_table[

    "High Price Probability"

] = (


    prediction_table[

        "High Price Probability"

    ].round(4)


)



print(

    "\nActual vs Predicted:"

)


display(

    prediction_table.head(20)

)



# ============================================================

# 54. NEW HOUSE CLASSIFICATION

# ============================================================

#

# This version avoids the X.mean() categorical error.

# Numerical columns → median

# Categorical columns → mode

# ============================================================


new_house = pd.DataFrame(

    index=[0],

    columns=X.columns

)



# Numerical features


for column in numeric_features:


    new_house.loc[

        0,

        column

    ] = X[

        column

    ].median()



# Categorical features


for column in categorical_features:


    new_house.loc[

        0,

        column

    ] = X[

        column

    ].mode()[0]



# Restore category dtype if necessary


for column in categorical_features:


    if str(

        X[column].dtype

    ) == "category":


        new_house[column] = pd.Categorical(


            new_house[column],


            categories=X[

                column

            ].cat.categories


        )



# ============================================================

# 55. PREDICT NEW HOUSE

# ============================================================


new_prediction = final_knn.predict(


    new_house


)[0]



new_probability = final_knn.predict_proba(


    new_house


)[0][1]



if new_prediction == 1:


    new_result = "High Price"


else:


    new_result = "Low Price"



print(

    "\n============================================"

)


print(

    "        NEW HOUSE CLASSIFICATION"

)


print(

    "============================================"

)


print(

    "Predicted Class:",

    new_result

)


print(

    "Probability of High Price:",

    round(

        new_probability,

        4

    )

)


print(

    "Probability of Low Price:",

    round(

        1 - new_probability,

        4

    )

)



# ============================================================

# 56. FINAL KNN SUMMARY

# ============================================================


print(

    "\n============================================"

)


print(

    "             KNN SUMMARY"

)


print(

    "============================================"

)


print(

    "Algorithm: K-Nearest Neighbors"

)


print(

    "Number of Neighbors:",

    best_k

)


print(

    "Accuracy:",

    round(

        final_accuracy,

        4

    )

)


print(

    "Precision:",

    round(

        final_precision,

        4

    )

)


print(

    "Recall:",

    round(

        final_recall,

        4

    )

)


print(

    "F1 Score:",

    round(

        final_f1,

        4

    )

)


print(

    "ROC-AUC:",

    round(

        final_auc,

        4

    )

)


print(

    "Cross Validation Mean:",

    round(

        cv_scores.mean(),

        4

    )

)


print(

    "============================================"

)



# ============================================================

# 57. KNN THEORY

# ============================================================


print("""

============================================================

K-NEAREST NEIGHBORS (KNN) - THEORY

============================================================


KNN is a supervised machine learning algorithm used for

classification and regression.


For classification, KNN looks at the nearest K observations

and assigns the class that receives the majority vote.


Example:


K = 5


Nearest neighbors:


Neighbor 1 → High Price

Neighbor 2 → High Price

Neighbor 3 → Low Price

Neighbor 4 → High Price

Neighbor 5 → Low Price


High Price = 3

Low Price  = 2


Prediction = High Price


============================================================


WHY STANDARDIZATION IS IMPORTANT?


KNN uses distance calculations.


Features with large numerical values can dominate the distance.


StandardScaler converts features approximately to:


Mean = 0

Standard Deviation = 1


Therefore, scaling is particularly important for KNN.


============================================================


K VALUE


Small K:

- Sensitive to noise

- More flexible

- Can overfit


Large K:

- Smoother decision boundary

- Less sensitive to noise

- Can underfit


The code tests K = 1 to K = 20.


============================================================


BOSTON HOUSING CLASSIFICATION


Original target:


MEDV = Continuous house price


Converted target:


0 = Low Price

1 = High Price


Median MEDV is used as the classification threshold.


============================================================

""")

No comments:

Post a Comment