Total Pageviews

Monday, September 28, 2026

kMEDOID

 # ================================================================

# K-MEDOIDS CLUSTERING - BOSTON HOUSING DATASET

# Complete Google Colab Single-Cell Code

# ================================================================


# -----------------------------

# 1. INSTALL REQUIRED LIBRARY

# -----------------------------

!pip install scikit-learn-extra -q


# -----------------------------

# 2. IMPORT LIBRARIES

# -----------------------------

import numpy as np

import pandas as pd

import matplotlib.pyplot as plt


from sklearn.datasets import fetch_openml

from sklearn.preprocessing import StandardScaler

from sklearn.decomposition import PCA

from sklearn.metrics import silhouette_score


from sklearn_extra.cluster import KMedoids


import warnings

warnings.filterwarnings("ignore")


# -----------------------------

# 3. LOAD BOSTON HOUSING DATA

# -----------------------------

print("=" * 70)

print("BOSTON HOUSING DATASET - K-MEDOIDS CLUSTERING")

print("=" * 70)


boston = fetch_openml(

    name="boston",

    version=1,

    as_frame=True

)


df = boston.frame.copy()


print("\nDataset Shape:", df.shape)


# -----------------------------

# 4. DISPLAY DATA

# -----------------------------

print("\nFirst 5 Rows:")

display(df.head())


print("\nDataset Information:")

print(df.info())


# -----------------------------

# 5. CONVERT NUMERIC COLUMNS

# -----------------------------

df = df.apply(pd.to_numeric, errors="coerce")


# -----------------------------

# 6. CHECK MISSING VALUES

# -----------------------------

print("\nMissing Values:")

print(df.isnull().sum())


# -----------------------------

# 7. HANDLE MISSING VALUES

# -----------------------------

df = df.dropna()


print("\nShape After Removing Missing Values:")

print(df.shape)


# -----------------------------

# 8. CHECK DUPLICATES

# -----------------------------

print("\nDuplicate Rows:")

print(df.duplicated().sum())


# Remove duplicates

df = df.drop_duplicates()


print("Shape After Removing Duplicates:")

print(df.shape)


# -----------------------------

# 9. SEPARATE FEATURES

# -----------------------------

# MEDV is the house-price target.

# It will NOT be used for clustering.


X = df.drop("MEDV", axis=1)


y = df["MEDV"]


print("\nFeatures Used for K-Medoids:")

print(X.columns.tolist())


print("\nNumber of Features:", X.shape[1])


# -----------------------------

# 10. STANDARDIZATION

# -----------------------------

scaler = StandardScaler()


X_scaled = scaler.fit_transform(X)


X_scaled = pd.DataFrame(

    X_scaled,

    columns=X.columns

)


print("\nScaled Data:")

display(X_scaled.head())


# -----------------------------

# 11. CHECK SCALED DATA

# -----------------------------

print("\nMean After Scaling:")

print(X_scaled.mean().round(2))


print("\nStandard Deviation After Scaling:")

print(X_scaled.std().round(2))


# -----------------------------

# 12. FIND BEST K

# USING SILHOUETTE SCORE

# -----------------------------

silhouette_scores = []


print("\n" + "=" * 70)

print("SILHOUETTE ANALYSIS")

print("=" * 70)


for k in range(2, 11):


    model = KMedoids(

        n_clusters=k,

        metric="euclidean",

        method="alternate",

        init="k-medoids++",

        max_iter=300,

        random_state=42

    )


    labels = model.fit_predict(X_scaled)


    score = silhouette_score(

        X_scaled,

        labels

    )


    silhouette_scores.append(score)


    print(

        "K =", k,

        "| Silhouette Score =",

        round(score, 4)

    )


# -----------------------------

# 13. PLOT SILHOUETTE SCORE

# -----------------------------

plt.figure(figsize=(9, 5))


plt.plot(

    range(2, 11),

    silhouette_scores,

    marker="o"

)


plt.xlabel("Number of Clusters (K)")

plt.ylabel("Silhouette Score")

plt.title("K-Medoids - Silhouette Score")


plt.xticks(range(2, 11))

plt.grid(True)


plt.show()


# -----------------------------

# 14. AUTOMATICALLY SELECT K

# -----------------------------

best_k = list(range(2, 11))[

    np.argmax(silhouette_scores)

]


print("\nBest K According to Silhouette Score:", best_k)


# -----------------------------

# 15. TRAIN FINAL K-MEDOIDS

# -----------------------------

kmedoids = KMedoids(

    n_clusters=best_k,

    metric="euclidean",

    method="alternate",

    init="k-medoids++",

    max_iter=300,

    random_state=42

)


kmedoids.fit(X_scaled)


# -----------------------------

# 16. GET CLUSTER LABELS

# -----------------------------

cluster_labels = kmedoids.labels_


df_result = df.copy()


df_result["Cluster"] = cluster_labels


print("\n" + "=" * 70)

print("K-MEDOIDS CLUSTERING RESULT")

print("=" * 70)


print("\nFirst 10 Cluster Assignments:")

display(df_result.head(10))


# -----------------------------

# 17. CLUSTER COUNTS

# -----------------------------

cluster_counts = (

    df_result["Cluster"]

    .value_counts()

    .sort_index()

)


print("\nNumber of Observations in Each Cluster:")

print(cluster_counts)


# -----------------------------

# 18. CLUSTER SIZE GRAPH

# -----------------------------

plt.figure(figsize=(8, 5))


cluster_counts.plot(

    kind="bar"

)


plt.xlabel("Cluster")

plt.ylabel("Number of Houses")

plt.title("Number of Houses in Each K-Medoids Cluster")


plt.xticks(rotation=0)

plt.grid(axis="y")


plt.show()


# -----------------------------

# 19. MEDOID INDICES

# -----------------------------

medoid_indices = kmedoids.medoid_indices_


print("\nMedoid Indices:")

print(medoid_indices)


# -----------------------------

# 20. ACTUAL MEDOID OBSERVATIONS

# -----------------------------

medoids = df.iloc[

    medoid_indices

].copy()


medoids["Cluster"] = range(best_k)


print("\nActual Medoid Observations:")

display(medoids)


# -----------------------------

# 21. MEDOID HOUSE VALUES

# -----------------------------

print("\nMEDV of Medoid Observations:")


for cluster_number, index in enumerate(medoid_indices):


    print(

        "Cluster",

        cluster_number,

        "| MEDV =",

        round(df.iloc[index]["MEDV"], 2)

    )


# -----------------------------

# 22. SILHOUETTE SCORE

# -----------------------------

final_silhouette = silhouette_score(

    X_scaled,

    cluster_labels

)


print("\nFinal Silhouette Score:")

print(round(final_silhouette, 4))


# -----------------------------

# 23. K-MEDOIDS INERTIA

# -----------------------------

print("\nK-Medoids Inertia:")

print(round(kmedoids.inertia_, 4))


# -----------------------------

# 24. PCA DIMENSION REDUCTION

# -----------------------------

pca = PCA(

    n_components=2

)


X_pca = pca.fit_transform(

    X_scaled

)


print("\nPCA Explained Variance:")

print(

    pca.explained_variance_ratio_

)


print(

    "\nTotal Explained Variance:",

    round(

        sum(pca.explained_variance_ratio_) * 100,

        2

    ),

    "%"

)


# -----------------------------

# 25. PCA CLUSTER VISUALIZATION

# -----------------------------

plt.figure(figsize=(10, 7))


scatter = plt.scatter(

    X_pca[:, 0],

    X_pca[:, 1],

    c=cluster_labels,

    s=45,

    alpha=0.7

)


# PCA coordinates of actual medoids

medoid_pca = X_pca[

    medoid_indices

]


plt.scatter(

    medoid_pca[:, 0],

    medoid_pca[:, 1],

    marker="X",

    s=250,

    edgecolors="black",

    linewidths=2,

    label="Medoids"

)


plt.xlabel("Principal Component 1")

plt.ylabel("Principal Component 2")


plt.title(

    "K-Medoids Clustering - Boston Housing"

)


plt.colorbar(

    scatter,

    label="Cluster"

)


plt.legend()

plt.grid(True)


plt.show()


# -----------------------------

# 26. CLUSTER-WISE ANALYSIS

# -----------------------------

cluster_analysis = (

    df_result

    .groupby("Cluster")

    .mean(numeric_only=True)

)


print("\n" + "=" * 70)

print("CLUSTER-WISE FEATURE ANALYSIS")

print("=" * 70)


display(

    cluster_analysis.round(2)

)


# -----------------------------

# 27. HOUSE PRICE ANALYSIS

# MEDV WAS NOT USED FOR TRAINING

# -----------------------------

cluster_report = (

    df_result

    .groupby("Cluster")

    .agg(

        Number_of_Houses=("MEDV", "count"),

        Average_MEDV=("MEDV", "mean"),

        Minimum_MEDV=("MEDV", "min"),

        Maximum_MEDV=("MEDV", "max")

    )

)


print("\nCluster House-Value Analysis:")

display(

    cluster_report.round(2)

)


# -----------------------------

# 28. AVERAGE MEDV GRAPH

# -----------------------------

plt.figure(figsize=(8, 5))


cluster_report[

    "Average_MEDV"

].plot(

    kind="bar"

)


plt.xlabel("Cluster")

plt.ylabel("Average MEDV")

plt.title(

    "Average House Value by K-Medoids Cluster"

)


plt.xticks(rotation=0)

plt.grid(axis="y")


plt.show()


# -----------------------------

# 29. FINAL SUMMARY

# -----------------------------

print("\n" + "=" * 70)

print("FINAL K-MEDOIDS SUMMARY")

print("=" * 70)


print("Dataset Size       :", df.shape)

print("Number of Features :", X.shape[1])

print("Selected K          :", best_k)

print("Silhouette Score    :", round(final_silhouette, 4))

print("Inertia             :", round(kmedoids.inertia_, 4))


print("\nCluster Sizes:")

print(cluster_counts)


print("\nMedoid Indices:")

print(medoid_indices)


print("\nK-Medoids completed successfully.")

Logistic Regression

 For Logistic Regression, the important change is that Boston Housing's original MEDV is a continuous house-price target, so we need to convert it into classes. 

  • 0 = Low Price
  • 1 = High Price
  • Median MEDV as the classification threshold


# ============================================================
# BOSTON HOUSING DATASET
# LOGISTIC REGRESSION
# COMPLETE DATA PREPROCESSING
# GOOGLE COLAB VERSION
# ============================================================


# ============================================================
# 1. IMPORT LIBRARIES
# ============================================================

import numpy as np
import pandas as pd

import matplotlib.pyplot as plt
import seaborn as sns

from sklearn.datasets import fetch_openml

from sklearn.model_selection import (
    train_test_split,
    cross_val_score
)

from sklearn.pipeline import Pipeline

from sklearn.compose import ColumnTransformer

from sklearn.impute import SimpleImputer

from sklearn.preprocessing import (
    StandardScaler,
    MinMaxScaler,
    RobustScaler,
    OneHotEncoder
)

from sklearn.feature_selection import (
    SelectKBest,
    f_classif
)

from sklearn.linear_model import LogisticRegression

from sklearn.metrics import (
    accuracy_score,
    precision_score,
    recall_score,
    f1_score,
    confusion_matrix,
    classification_report,
    ConfusionMatrixDisplay,
    roc_curve,
    roc_auc_score
)

import warnings

warnings.filterwarnings("ignore")


# ============================================================
# 2. LOAD BOSTON HOUSING DATASET
# ============================================================

boston = fetch_openml(
    name="boston",
    version=1,
    as_frame=True
)

df = boston.frame.copy()

print("Boston Housing Dataset Loaded Successfully!")


# ============================================================
# 3. DISPLAY FIRST 5 ROWS
# ============================================================

print("\nFirst 5 Rows:")

display(
    df.head()
)


# ============================================================
# 4. DATASET SHAPE
# ============================================================

print("\nDataset Shape:")

print(
    df.shape
)


# ============================================================
# 5. COLUMN NAMES
# ============================================================

print("\nColumn Names:")

print(
    df.columns.tolist()
)


# ============================================================
# 6. DATA INFORMATION
# ============================================================

print("\nDataset Information:")

df.info()


# ============================================================
# 7. DATA TYPES
# ============================================================

print("\nData Types:")

print(
    df.dtypes
)


# ============================================================
# 8. CHECK MISSING VALUES
# ============================================================

print("\nMissing Values:")

print(
    df.isnull().sum()
)


# ============================================================
# 9. MISSING VALUE PERCENTAGE
# ============================================================

missing_percentage = (

    df.isnull()
      .mean()
      .mul(100)
      .sort_values(
          ascending=False
      )

)

print("\nMissing Value Percentage:")

print(
    missing_percentage
)


# ============================================================
# 10. CHECK DUPLICATE RECORDS
# ============================================================

duplicate_count = df.duplicated().sum()

print(
    "\nNumber of Duplicate Rows:",
    duplicate_count
)


# ============================================================
# 11. REMOVE DUPLICATES
# ============================================================

df = df.drop_duplicates()

print(
    "\nShape After Duplicate Removal:"
)

print(
    df.shape
)


# ============================================================
# 12. STATISTICAL SUMMARY
# ============================================================

print(
    "\nStatistical Summary:"
)

display(
    df.describe()
)


# ============================================================
# 13. UNIQUE VALUES
# ============================================================

print(
    "\nNumber of Unique Values:"
)

print(
    df.nunique()
)


# ============================================================
# 14. TARGET VARIABLE
# ============================================================

target = "MEDV"

print(
    "\nOriginal Target Variable:",
    target
)


# ============================================================
# 15. TARGET DISTRIBUTION
# ============================================================

plt.figure(
    figsize=(9,5)
)

sns.histplot(
    df[target],
    kde=True
)

plt.title(
    "Distribution of Boston House Prices"
)

plt.xlabel(
    "MEDV"
)

plt.ylabel(
    "Frequency"
)

plt.show()


# ============================================================
# 16. OUTLIER DETECTION
# ============================================================

numeric_columns = (

    df.select_dtypes(
        include=np.number
    ).columns

)


plt.figure(
    figsize=(14,6)
)

sns.boxplot(
    data=df[
        numeric_columns
    ]
)

plt.title(
    "Boxplot of Boston Housing Features"
)

plt.xticks(
    rotation=45
)

plt.show()


# ============================================================
# 17. IQR OUTLIER DETECTION
# ============================================================

print(
    "\nIQR OUTLIER ANALYSIS"
)

for column in numeric_columns:

    Q1 = df[column].quantile(
        0.25
    )

    Q3 = df[column].quantile(
        0.75
    )

    IQR = Q3 - Q1

    lower_bound = (
        Q1 - 1.5 * IQR
    )

    upper_bound = (
        Q3 + 1.5 * IQR
    )

    outliers = df[
        (df[column] < lower_bound)
        |
        (df[column] > upper_bound)
    ]

    print(
        column,
        "→",
        len(outliers),
        "outliers"
    )


# ============================================================
# 18. OUTLIER CAPPING
# ============================================================
#
# We do not automatically delete unusual observations.
# Instead, this demonstrates IQR-based capping.
# ============================================================

df_capped = df.copy()

for column in numeric_columns:

    Q1 = df_capped[column].quantile(
        0.25
    )

    Q3 = df_capped[column].quantile(
        0.75
    )

    IQR = Q3 - Q1

    lower_bound = (
        Q1 - 1.5 * IQR
    )

    upper_bound = (
        Q3 + 1.5 * IQR
    )

    df_capped[column] = (

        df_capped[column]
        .clip(
            lower_bound,
            upper_bound
        )

    )


print(
    "\nIQR Outlier Capping Completed."
)


# ============================================================
# 19. CORRELATION MATRIX
# ============================================================

correlation = df.corr(
    numeric_only=True
)


print(
    "\nCorrelation with MEDV:"
)

print(
    correlation[target]
    .sort_values(
        ascending=False
    )
)


# ============================================================
# 20. CORRELATION HEATMAP
# ============================================================

plt.figure(
    figsize=(12,9)
)

sns.heatmap(
    correlation,
    annot=True,
    fmt=".2f",
    cmap="coolwarm"
)

plt.title(
    "Boston Housing Correlation Matrix"
)

plt.show()


# ============================================================
# 21. CONVERT REGRESSION TARGET INTO CLASSIFICATION TARGET
# ============================================================
#
# Boston Housing MEDV is continuous.
#
# Logistic Regression requires categorical classes.
#
# Median price is used as the threshold.
#
# MEDV < median → 0 → Low Price
#
# MEDV >= median → 1 → High Price
# ============================================================

median_price = df[target].median()

print(
    "\nMedian House Price (MEDV):",
    median_price
)


df["Price_Class"] = (

    df[target] >= median_price

).astype(int)


# ============================================================
# 22. DISPLAY CLASS DISTRIBUTION
# ============================================================

print(
    "\nPrice Class Distribution:"
)

print(
    df["Price_Class"]
    .value_counts()
)


# ============================================================
# 23. VISUALIZE CLASS DISTRIBUTION
# ============================================================

plt.figure(
    figsize=(7,5)
)

sns.countplot(
    x="Price_Class",
    data=df
)

plt.title(
    "Low Price vs High Price Houses"
)

plt.xlabel(
    "Price Class"
)

plt.ylabel(
    "Number of Houses"
)

plt.xticks(
    [0,1],
    [
        "Low Price",
        "High Price"
    ]
)

plt.show()


# ============================================================
# 24. SEPARATE FEATURES AND TARGET
# ============================================================

X = df.drop(
    columns=[
        "MEDV",
        "Price_Class"
    ]
)

y = df["Price_Class"]


print(
    "\nFeature Shape:"
)

print(
    X.shape
)


print(
    "\nTarget Shape:"
)

print(
    y.shape
)


# ============================================================
# 25. IDENTIFY NUMERICAL FEATURES
# ============================================================

numeric_features = (

    X.select_dtypes(
        include=np.number
    )
    .columns
    .tolist()

)


categorical_features = (

    X.select_dtypes(
        exclude=np.number
    )
    .columns
    .tolist()

)


print(
    "\nNumerical Features:"
)

print(
    numeric_features
)


print(
    "\nCategorical Features:"
)

print(
    categorical_features
)


# ============================================================
# 26. TRAIN-TEST SPLIT
# ============================================================

X_train, X_test, y_train, y_test = (

    train_test_split(

        X,
        y,

        test_size=0.20,

        random_state=42,

        stratify=y

    )

)


print(
    "\nTraining Data:",
    X_train.shape
)

print(
    "Testing Data:",
    X_test.shape
)


# ============================================================
# 27. NUMERICAL PREPROCESSING
# ============================================================

numeric_pipeline = Pipeline(

    steps=[

        (
            "imputer",

            SimpleImputer(
                strategy="median"
            )

        ),

        (
            "scaler",

            StandardScaler()
        )

    ]

)


# ============================================================
# 28. CATEGORICAL PREPROCESSING
# ============================================================

categorical_pipeline = Pipeline(

    steps=[

        (
            "imputer",

            SimpleImputer(
                strategy="most_frequent"
            )

        ),

        (
            "encoder",

            OneHotEncoder(
                handle_unknown="ignore"
            )

        )

    ]

)


# ============================================================
# 29. COMBINE PREPROCESSING
# ============================================================

preprocessor = ColumnTransformer(

    transformers=[

        (
            "numeric",

            numeric_pipeline,

            numeric_features

        ),

        (
            "categorical",

            categorical_pipeline,

            categorical_features

        )

    ]

)


# ============================================================
# 30. LOGISTIC REGRESSION PIPELINE
# ============================================================

logistic_model = Pipeline(

    steps=[

        (
            "preprocessing",

            preprocessor

        ),

        (
            "model",

            LogisticRegression(
                max_iter=2000
            )

        )

    ]

)


# ============================================================
# 31. TRAIN MODEL
# ============================================================

logistic_model.fit(

    X_train,

    y_train

)


print(
    "\nLogistic Regression Model Trained Successfully!"
)


# ============================================================
# 32. MAKE PREDICTIONS
# ============================================================

y_pred = logistic_model.predict(
    X_test
)


# ============================================================
# 33. PREDICT PROBABILITIES
# ============================================================

y_probability = (

    logistic_model
    .predict_proba(
        X_test
    )[:, 1]

)


# ============================================================
# 34. ACCURACY
# ============================================================

accuracy = accuracy_score(

    y_test,

    y_pred

)


# ============================================================
# 35. PRECISION
# ============================================================

precision = precision_score(

    y_test,

    y_pred,

    zero_division=0

)


# ============================================================
# 36. RECALL
# ============================================================

recall = recall_score(

    y_test,

    y_pred,

    zero_division=0

)


# ============================================================
# 37. F1 SCORE
# ============================================================

f1 = f1_score(

    y_test,

    y_pred,

    zero_division=0

)


# ============================================================
# 38. ROC-AUC
# ============================================================

roc_auc = roc_auc_score(

    y_test,

    y_probability

)


# ============================================================
# 39. DISPLAY RESULTS
# ============================================================

print("\n")
print("============================================")
print("       LOGISTIC REGRESSION RESULTS")
print("============================================")

print(
    "Accuracy  :",
    round(accuracy, 4)
)

print(
    "Precision :",
    round(precision, 4)
)

print(
    "Recall    :",
    round(recall, 4)
)

print(
    "F1 Score  :",
    round(f1, 4)
)

print(
    "ROC-AUC   :",
    round(roc_auc, 4)
)

print("============================================")


# ============================================================
# 40. CONFUSION MATRIX
# ============================================================

cm = confusion_matrix(

    y_test,

    y_pred

)


print(
    "\nConfusion Matrix:"
)

print(
    cm
)


# ============================================================
# 41. CONFUSION MATRIX VALUES
# ============================================================

TN, FP, FN, TP = cm.ravel()


print(
    "\nTrue Negative (TN):",
    TN
)

print(
    "False Positive (FP):",
    FP
)

print(
    "False Negative (FN):",
    FN
)

print(
    "True Positive (TP):",
    TP
)


# ============================================================
# 42. CONFUSION MATRIX VISUALIZATION
# ============================================================

plt.figure(
    figsize=(7,6)
)

disp = ConfusionMatrixDisplay(

    confusion_matrix=cm,

    display_labels=[
        "Low Price",
        "High Price"
    ]

)

disp.plot()

plt.title(
    "Boston Housing - Logistic Regression"
)

plt.show()


# ============================================================
# 43. CONFUSION MATRIX HEATMAP
# ============================================================

plt.figure(
    figsize=(7,6)
)

sns.heatmap(

    cm,

    annot=True,

    fmt="d",

    cmap="Blues",

    xticklabels=[
        "Predicted Low",
        "Predicted High"
    ],

    yticklabels=[
        "Actual Low",
        "Actual High"
    ]

)

plt.title(
    "Confusion Matrix Heatmap"
)

plt.xlabel(
    "Predicted Class"
)

plt.ylabel(
    "Actual Class"
)

plt.show()


# ============================================================
# 44. CLASSIFICATION REPORT
# ============================================================

print(
    "\n============================================"
)

print(
    "          CLASSIFICATION REPORT"
)

print(
    "============================================"
)

print(

    classification_report(

        y_test,

        y_pred,

        target_names=[
            "Low Price",
            "High Price"
        ],

        zero_division=0

    )

)


# ============================================================
# 45. PERFORMANCE TABLE
# ============================================================

metrics_table = pd.DataFrame({

    "Metric": [

        "Accuracy",

        "Precision",

        "Recall",

        "F1 Score",

        "ROC-AUC"

    ],

    "Score": [

        accuracy,

        precision,

        recall,

        f1,

        roc_auc

    ]

})


metrics_table["Score"] = (

    metrics_table["Score"]
    .round(4)

)


print(
    "\nPerformance Metrics:"
)

display(
    metrics_table
)


# ============================================================
# 46. CROSS VALIDATION
# ============================================================

cv_scores = cross_val_score(

    logistic_model,

    X,

    y,

    cv=5,

    scoring="accuracy"

)


print(
    "\n5-Fold Cross Validation Accuracy:"
)

print(
    cv_scores
)


print(
    "\nMean Cross Validation Accuracy:"
)

print(
    round(
        cv_scores.mean(),
        4
    )
)


# ============================================================
# 47. ROC CURVE
# ============================================================

fpr, tpr, thresholds = roc_curve(

    y_test,

    y_probability

)


plt.figure(
    figsize=(8,6)
)

plt.plot(

    fpr,

    tpr,

    label=f"ROC-AUC = {roc_auc:.4f}"

)

plt.plot(

    [0,1],

    [0,1],

    linestyle="--"

)

plt.xlabel(
    "False Positive Rate"
)

plt.ylabel(
    "True Positive Rate"
)

plt.title(
    "ROC Curve - Logistic Regression"
)

plt.legend()

plt.show()


# ============================================================
# 48. PRECISION / RECALL / F1 GRAPH
# ============================================================

metric_names = [

    "Accuracy",

    "Precision",

    "Recall",

    "F1 Score"

]


metric_values = [

    accuracy,

    precision,

    recall,

    f1

]


plt.figure(
    figsize=(9,6)
)

plt.bar(

    metric_names,

    metric_values

)

plt.ylim(
    0,
    1
)

plt.ylabel(
    "Score"
)

plt.title(
    "Logistic Regression Performance"
)

plt.show()


# ============================================================
# 49. ACTUAL VS PREDICTED TABLE
# ============================================================

prediction_table = pd.DataFrame({

    "Actual Class": y_test.values,

    "Predicted Class": y_pred,

    "High Price Probability":
        y_probability

})


prediction_table["Actual Class"] = (

    prediction_table["Actual Class"]
    .map({
        0: "Low Price",
        1: "High Price"
    })

)


prediction_table["Predicted Class"] = (

    prediction_table["Predicted Class"]
    .map({
        0: "Low Price",
        1: "High Price"
    })

)


prediction_table[
    "High Price Probability"
] = (

    prediction_table[
        "High Price Probability"
    ].round(4)

)


print(
    "\nActual vs Predicted:"
)

display(
    prediction_table.head(20)
)


# ============================================================
# 50. TEST A NEW HOUSE
# ============================================================

# ============================================================
# 50. TEST A NEW HOUSE - CORRECTED VERSION
# ============================================================

print(
    "\n============================================"
)

print(
    "        NEW HOUSE CLASSIFICATION"
)

print(
    "============================================"
)


# ------------------------------------------------------------
# Create a new house using the MEDIAN/MODE of each feature
# ------------------------------------------------------------

new_house = pd.DataFrame(
    index=[0],
    columns=X.columns
)


# Fill numerical columns with their median
numeric_cols = X.select_dtypes(
    include=np.number
).columns


for column in numeric_cols:

    new_house.loc[0, column] = X[
        column
    ].median()


# Fill categorical columns with their most frequent value
categorical_cols = X.select_dtypes(
    exclude=np.number
).columns


for column in categorical_cols:

    new_house.loc[0, column] = X[
        column
    ].mode()[0]


# Make sure categorical columns retain their proper type
for column in categorical_cols:

    if str(X[column].dtype) == "category":

        new_house[column] = pd.Categorical(
            new_house[column],
            categories=X[column].cat.categories
        )


# ------------------------------------------------------------
# Display the new house data
# ------------------------------------------------------------

print("\nNew House Input:")

display(
    new_house
)


# ------------------------------------------------------------
# Predict class
# ------------------------------------------------------------

new_prediction = logistic_model.predict(
    new_house
)[0]


# ------------------------------------------------------------
# Predict probability
# ------------------------------------------------------------

new_probability = logistic_model.predict_proba(
    new_house
)[0][1]


# ------------------------------------------------------------
# Convert prediction to readable result
# ------------------------------------------------------------

if new_prediction == 1:

    result = "High Price"

else:

    result = "Low Price"


# ------------------------------------------------------------
# Display result
# ------------------------------------------------------------

print(
    "\nPredicted Class:",
    result
)

print(
    "Probability of High Price:",
    round(
        new_probability,
        4
    )
)

print(
    "Probability of Low Price:",
    round(
        1 - new_probability,
        4
    )
)


print(
    "\n============================================"
)
# ============================================================
# 51. FINAL SUMMARY
# ============================================================

print("""
============================================================

BOSTON HOUSING LOGISTIC REGRESSION
COMPLETE WORKFLOW

1. Load Boston Housing Dataset
2. Inspect Dataset
3. Check Data Types
4. Check Missing Values
5. Check Duplicate Records
6. Remove Duplicate Records
7. Statistical Summary
8. Distribution Analysis
9. Outlier Detection
10. IQR Analysis
11. Outlier Capping
12. Correlation Analysis
13. Convert MEDV into Classes
14. Low Price = 0
15. High Price = 1
16. Train-Test Split
17. Missing Value Imputation
18. Standardization
19. Categorical Encoding
20. Logistic Regression
21. Prediction
22. Accuracy
23. Precision
24. Recall
25. F1 Score
26. ROC-AUC
27. Confusion Matrix
28. Classification Report
29. Cross Validation
30. ROC Curve

============================================================
"""
)