Total Pageviews

Monday, September 28, 2026

Logistic Regression

 For Logistic Regression, the important change is that Boston Housing's original MEDV is a continuous house-price target, so we need to convert it into classes. 

  • 0 = Low Price
  • 1 = High Price
  • Median MEDV as the classification threshold


# ============================================================
# BOSTON HOUSING DATASET
# LOGISTIC REGRESSION
# COMPLETE DATA PREPROCESSING
# GOOGLE COLAB VERSION
# ============================================================


# ============================================================
# 1. IMPORT LIBRARIES
# ============================================================

import numpy as np
import pandas as pd

import matplotlib.pyplot as plt
import seaborn as sns

from sklearn.datasets import fetch_openml

from sklearn.model_selection import (
    train_test_split,
    cross_val_score
)

from sklearn.pipeline import Pipeline

from sklearn.compose import ColumnTransformer

from sklearn.impute import SimpleImputer

from sklearn.preprocessing import (
    StandardScaler,
    MinMaxScaler,
    RobustScaler,
    OneHotEncoder
)

from sklearn.feature_selection import (
    SelectKBest,
    f_classif
)

from sklearn.linear_model import LogisticRegression

from sklearn.metrics import (
    accuracy_score,
    precision_score,
    recall_score,
    f1_score,
    confusion_matrix,
    classification_report,
    ConfusionMatrixDisplay,
    roc_curve,
    roc_auc_score
)

import warnings

warnings.filterwarnings("ignore")


# ============================================================
# 2. LOAD BOSTON HOUSING DATASET
# ============================================================

boston = fetch_openml(
    name="boston",
    version=1,
    as_frame=True
)

df = boston.frame.copy()

print("Boston Housing Dataset Loaded Successfully!")


# ============================================================
# 3. DISPLAY FIRST 5 ROWS
# ============================================================

print("\nFirst 5 Rows:")

display(
    df.head()
)


# ============================================================
# 4. DATASET SHAPE
# ============================================================

print("\nDataset Shape:")

print(
    df.shape
)


# ============================================================
# 5. COLUMN NAMES
# ============================================================

print("\nColumn Names:")

print(
    df.columns.tolist()
)


# ============================================================
# 6. DATA INFORMATION
# ============================================================

print("\nDataset Information:")

df.info()


# ============================================================
# 7. DATA TYPES
# ============================================================

print("\nData Types:")

print(
    df.dtypes
)


# ============================================================
# 8. CHECK MISSING VALUES
# ============================================================

print("\nMissing Values:")

print(
    df.isnull().sum()
)


# ============================================================
# 9. MISSING VALUE PERCENTAGE
# ============================================================

missing_percentage = (

    df.isnull()
      .mean()
      .mul(100)
      .sort_values(
          ascending=False
      )

)

print("\nMissing Value Percentage:")

print(
    missing_percentage
)


# ============================================================
# 10. CHECK DUPLICATE RECORDS
# ============================================================

duplicate_count = df.duplicated().sum()

print(
    "\nNumber of Duplicate Rows:",
    duplicate_count
)


# ============================================================
# 11. REMOVE DUPLICATES
# ============================================================

df = df.drop_duplicates()

print(
    "\nShape After Duplicate Removal:"
)

print(
    df.shape
)


# ============================================================
# 12. STATISTICAL SUMMARY
# ============================================================

print(
    "\nStatistical Summary:"
)

display(
    df.describe()
)


# ============================================================
# 13. UNIQUE VALUES
# ============================================================

print(
    "\nNumber of Unique Values:"
)

print(
    df.nunique()
)


# ============================================================
# 14. TARGET VARIABLE
# ============================================================

target = "MEDV"

print(
    "\nOriginal Target Variable:",
    target
)


# ============================================================
# 15. TARGET DISTRIBUTION
# ============================================================

plt.figure(
    figsize=(9,5)
)

sns.histplot(
    df[target],
    kde=True
)

plt.title(
    "Distribution of Boston House Prices"
)

plt.xlabel(
    "MEDV"
)

plt.ylabel(
    "Frequency"
)

plt.show()


# ============================================================
# 16. OUTLIER DETECTION
# ============================================================

numeric_columns = (

    df.select_dtypes(
        include=np.number
    ).columns

)


plt.figure(
    figsize=(14,6)
)

sns.boxplot(
    data=df[
        numeric_columns
    ]
)

plt.title(
    "Boxplot of Boston Housing Features"
)

plt.xticks(
    rotation=45
)

plt.show()


# ============================================================
# 17. IQR OUTLIER DETECTION
# ============================================================

print(
    "\nIQR OUTLIER ANALYSIS"
)

for column in numeric_columns:

    Q1 = df[column].quantile(
        0.25
    )

    Q3 = df[column].quantile(
        0.75
    )

    IQR = Q3 - Q1

    lower_bound = (
        Q1 - 1.5 * IQR
    )

    upper_bound = (
        Q3 + 1.5 * IQR
    )

    outliers = df[
        (df[column] < lower_bound)
        |
        (df[column] > upper_bound)
    ]

    print(
        column,
        "→",
        len(outliers),
        "outliers"
    )


# ============================================================
# 18. OUTLIER CAPPING
# ============================================================
#
# We do not automatically delete unusual observations.
# Instead, this demonstrates IQR-based capping.
# ============================================================

df_capped = df.copy()

for column in numeric_columns:

    Q1 = df_capped[column].quantile(
        0.25
    )

    Q3 = df_capped[column].quantile(
        0.75
    )

    IQR = Q3 - Q1

    lower_bound = (
        Q1 - 1.5 * IQR
    )

    upper_bound = (
        Q3 + 1.5 * IQR
    )

    df_capped[column] = (

        df_capped[column]
        .clip(
            lower_bound,
            upper_bound
        )

    )


print(
    "\nIQR Outlier Capping Completed."
)


# ============================================================
# 19. CORRELATION MATRIX
# ============================================================

correlation = df.corr(
    numeric_only=True
)


print(
    "\nCorrelation with MEDV:"
)

print(
    correlation[target]
    .sort_values(
        ascending=False
    )
)


# ============================================================
# 20. CORRELATION HEATMAP
# ============================================================

plt.figure(
    figsize=(12,9)
)

sns.heatmap(
    correlation,
    annot=True,
    fmt=".2f",
    cmap="coolwarm"
)

plt.title(
    "Boston Housing Correlation Matrix"
)

plt.show()


# ============================================================
# 21. CONVERT REGRESSION TARGET INTO CLASSIFICATION TARGET
# ============================================================
#
# Boston Housing MEDV is continuous.
#
# Logistic Regression requires categorical classes.
#
# Median price is used as the threshold.
#
# MEDV < median → 0 → Low Price
#
# MEDV >= median → 1 → High Price
# ============================================================

median_price = df[target].median()

print(
    "\nMedian House Price (MEDV):",
    median_price
)


df["Price_Class"] = (

    df[target] >= median_price

).astype(int)


# ============================================================
# 22. DISPLAY CLASS DISTRIBUTION
# ============================================================

print(
    "\nPrice Class Distribution:"
)

print(
    df["Price_Class"]
    .value_counts()
)


# ============================================================
# 23. VISUALIZE CLASS DISTRIBUTION
# ============================================================

plt.figure(
    figsize=(7,5)
)

sns.countplot(
    x="Price_Class",
    data=df
)

plt.title(
    "Low Price vs High Price Houses"
)

plt.xlabel(
    "Price Class"
)

plt.ylabel(
    "Number of Houses"
)

plt.xticks(
    [0,1],
    [
        "Low Price",
        "High Price"
    ]
)

plt.show()


# ============================================================
# 24. SEPARATE FEATURES AND TARGET
# ============================================================

X = df.drop(
    columns=[
        "MEDV",
        "Price_Class"
    ]
)

y = df["Price_Class"]


print(
    "\nFeature Shape:"
)

print(
    X.shape
)


print(
    "\nTarget Shape:"
)

print(
    y.shape
)


# ============================================================
# 25. IDENTIFY NUMERICAL FEATURES
# ============================================================

numeric_features = (

    X.select_dtypes(
        include=np.number
    )
    .columns
    .tolist()

)


categorical_features = (

    X.select_dtypes(
        exclude=np.number
    )
    .columns
    .tolist()

)


print(
    "\nNumerical Features:"
)

print(
    numeric_features
)


print(
    "\nCategorical Features:"
)

print(
    categorical_features
)


# ============================================================
# 26. TRAIN-TEST SPLIT
# ============================================================

X_train, X_test, y_train, y_test = (

    train_test_split(

        X,
        y,

        test_size=0.20,

        random_state=42,

        stratify=y

    )

)


print(
    "\nTraining Data:",
    X_train.shape
)

print(
    "Testing Data:",
    X_test.shape
)


# ============================================================
# 27. NUMERICAL PREPROCESSING
# ============================================================

numeric_pipeline = Pipeline(

    steps=[

        (
            "imputer",

            SimpleImputer(
                strategy="median"
            )

        ),

        (
            "scaler",

            StandardScaler()
        )

    ]

)


# ============================================================
# 28. CATEGORICAL PREPROCESSING
# ============================================================

categorical_pipeline = Pipeline(

    steps=[

        (
            "imputer",

            SimpleImputer(
                strategy="most_frequent"
            )

        ),

        (
            "encoder",

            OneHotEncoder(
                handle_unknown="ignore"
            )

        )

    ]

)


# ============================================================
# 29. COMBINE PREPROCESSING
# ============================================================

preprocessor = ColumnTransformer(

    transformers=[

        (
            "numeric",

            numeric_pipeline,

            numeric_features

        ),

        (
            "categorical",

            categorical_pipeline,

            categorical_features

        )

    ]

)


# ============================================================
# 30. LOGISTIC REGRESSION PIPELINE
# ============================================================

logistic_model = Pipeline(

    steps=[

        (
            "preprocessing",

            preprocessor

        ),

        (
            "model",

            LogisticRegression(
                max_iter=2000
            )

        )

    ]

)


# ============================================================
# 31. TRAIN MODEL
# ============================================================

logistic_model.fit(

    X_train,

    y_train

)


print(
    "\nLogistic Regression Model Trained Successfully!"
)


# ============================================================
# 32. MAKE PREDICTIONS
# ============================================================

y_pred = logistic_model.predict(
    X_test
)


# ============================================================
# 33. PREDICT PROBABILITIES
# ============================================================

y_probability = (

    logistic_model
    .predict_proba(
        X_test
    )[:, 1]

)


# ============================================================
# 34. ACCURACY
# ============================================================

accuracy = accuracy_score(

    y_test,

    y_pred

)


# ============================================================
# 35. PRECISION
# ============================================================

precision = precision_score(

    y_test,

    y_pred,

    zero_division=0

)


# ============================================================
# 36. RECALL
# ============================================================

recall = recall_score(

    y_test,

    y_pred,

    zero_division=0

)


# ============================================================
# 37. F1 SCORE
# ============================================================

f1 = f1_score(

    y_test,

    y_pred,

    zero_division=0

)


# ============================================================
# 38. ROC-AUC
# ============================================================

roc_auc = roc_auc_score(

    y_test,

    y_probability

)


# ============================================================
# 39. DISPLAY RESULTS
# ============================================================

print("\n")
print("============================================")
print("       LOGISTIC REGRESSION RESULTS")
print("============================================")

print(
    "Accuracy  :",
    round(accuracy, 4)
)

print(
    "Precision :",
    round(precision, 4)
)

print(
    "Recall    :",
    round(recall, 4)
)

print(
    "F1 Score  :",
    round(f1, 4)
)

print(
    "ROC-AUC   :",
    round(roc_auc, 4)
)

print("============================================")


# ============================================================
# 40. CONFUSION MATRIX
# ============================================================

cm = confusion_matrix(

    y_test,

    y_pred

)


print(
    "\nConfusion Matrix:"
)

print(
    cm
)


# ============================================================
# 41. CONFUSION MATRIX VALUES
# ============================================================

TN, FP, FN, TP = cm.ravel()


print(
    "\nTrue Negative (TN):",
    TN
)

print(
    "False Positive (FP):",
    FP
)

print(
    "False Negative (FN):",
    FN
)

print(
    "True Positive (TP):",
    TP
)


# ============================================================
# 42. CONFUSION MATRIX VISUALIZATION
# ============================================================

plt.figure(
    figsize=(7,6)
)

disp = ConfusionMatrixDisplay(

    confusion_matrix=cm,

    display_labels=[
        "Low Price",
        "High Price"
    ]

)

disp.plot()

plt.title(
    "Boston Housing - Logistic Regression"
)

plt.show()


# ============================================================
# 43. CONFUSION MATRIX HEATMAP
# ============================================================

plt.figure(
    figsize=(7,6)
)

sns.heatmap(

    cm,

    annot=True,

    fmt="d",

    cmap="Blues",

    xticklabels=[
        "Predicted Low",
        "Predicted High"
    ],

    yticklabels=[
        "Actual Low",
        "Actual High"
    ]

)

plt.title(
    "Confusion Matrix Heatmap"
)

plt.xlabel(
    "Predicted Class"
)

plt.ylabel(
    "Actual Class"
)

plt.show()


# ============================================================
# 44. CLASSIFICATION REPORT
# ============================================================

print(
    "\n============================================"
)

print(
    "          CLASSIFICATION REPORT"
)

print(
    "============================================"
)

print(

    classification_report(

        y_test,

        y_pred,

        target_names=[
            "Low Price",
            "High Price"
        ],

        zero_division=0

    )

)


# ============================================================
# 45. PERFORMANCE TABLE
# ============================================================

metrics_table = pd.DataFrame({

    "Metric": [

        "Accuracy",

        "Precision",

        "Recall",

        "F1 Score",

        "ROC-AUC"

    ],

    "Score": [

        accuracy,

        precision,

        recall,

        f1,

        roc_auc

    ]

})


metrics_table["Score"] = (

    metrics_table["Score"]
    .round(4)

)


print(
    "\nPerformance Metrics:"
)

display(
    metrics_table
)


# ============================================================
# 46. CROSS VALIDATION
# ============================================================

cv_scores = cross_val_score(

    logistic_model,

    X,

    y,

    cv=5,

    scoring="accuracy"

)


print(
    "\n5-Fold Cross Validation Accuracy:"
)

print(
    cv_scores
)


print(
    "\nMean Cross Validation Accuracy:"
)

print(
    round(
        cv_scores.mean(),
        4
    )
)


# ============================================================
# 47. ROC CURVE
# ============================================================

fpr, tpr, thresholds = roc_curve(

    y_test,

    y_probability

)


plt.figure(
    figsize=(8,6)
)

plt.plot(

    fpr,

    tpr,

    label=f"ROC-AUC = {roc_auc:.4f}"

)

plt.plot(

    [0,1],

    [0,1],

    linestyle="--"

)

plt.xlabel(
    "False Positive Rate"
)

plt.ylabel(
    "True Positive Rate"
)

plt.title(
    "ROC Curve - Logistic Regression"
)

plt.legend()

plt.show()


# ============================================================
# 48. PRECISION / RECALL / F1 GRAPH
# ============================================================

metric_names = [

    "Accuracy",

    "Precision",

    "Recall",

    "F1 Score"

]


metric_values = [

    accuracy,

    precision,

    recall,

    f1

]


plt.figure(
    figsize=(9,6)
)

plt.bar(

    metric_names,

    metric_values

)

plt.ylim(
    0,
    1
)

plt.ylabel(
    "Score"
)

plt.title(
    "Logistic Regression Performance"
)

plt.show()


# ============================================================
# 49. ACTUAL VS PREDICTED TABLE
# ============================================================

prediction_table = pd.DataFrame({

    "Actual Class": y_test.values,

    "Predicted Class": y_pred,

    "High Price Probability":
        y_probability

})


prediction_table["Actual Class"] = (

    prediction_table["Actual Class"]
    .map({
        0: "Low Price",
        1: "High Price"
    })

)


prediction_table["Predicted Class"] = (

    prediction_table["Predicted Class"]
    .map({
        0: "Low Price",
        1: "High Price"
    })

)


prediction_table[
    "High Price Probability"
] = (

    prediction_table[
        "High Price Probability"
    ].round(4)

)


print(
    "\nActual vs Predicted:"
)

display(
    prediction_table.head(20)
)


# ============================================================
# 50. TEST A NEW HOUSE
# ============================================================

# ============================================================
# 50. TEST A NEW HOUSE - CORRECTED VERSION
# ============================================================

print(
    "\n============================================"
)

print(
    "        NEW HOUSE CLASSIFICATION"
)

print(
    "============================================"
)


# ------------------------------------------------------------
# Create a new house using the MEDIAN/MODE of each feature
# ------------------------------------------------------------

new_house = pd.DataFrame(
    index=[0],
    columns=X.columns
)


# Fill numerical columns with their median
numeric_cols = X.select_dtypes(
    include=np.number
).columns


for column in numeric_cols:

    new_house.loc[0, column] = X[
        column
    ].median()


# Fill categorical columns with their most frequent value
categorical_cols = X.select_dtypes(
    exclude=np.number
).columns


for column in categorical_cols:

    new_house.loc[0, column] = X[
        column
    ].mode()[0]


# Make sure categorical columns retain their proper type
for column in categorical_cols:

    if str(X[column].dtype) == "category":

        new_house[column] = pd.Categorical(
            new_house[column],
            categories=X[column].cat.categories
        )


# ------------------------------------------------------------
# Display the new house data
# ------------------------------------------------------------

print("\nNew House Input:")

display(
    new_house
)


# ------------------------------------------------------------
# Predict class
# ------------------------------------------------------------

new_prediction = logistic_model.predict(
    new_house
)[0]


# ------------------------------------------------------------
# Predict probability
# ------------------------------------------------------------

new_probability = logistic_model.predict_proba(
    new_house
)[0][1]


# ------------------------------------------------------------
# Convert prediction to readable result
# ------------------------------------------------------------

if new_prediction == 1:

    result = "High Price"

else:

    result = "Low Price"


# ------------------------------------------------------------
# Display result
# ------------------------------------------------------------

print(
    "\nPredicted Class:",
    result
)

print(
    "Probability of High Price:",
    round(
        new_probability,
        4
    )
)

print(
    "Probability of Low Price:",
    round(
        1 - new_probability,
        4
    )
)


print(
    "\n============================================"
)
# ============================================================
# 51. FINAL SUMMARY
# ============================================================

print("""
============================================================

BOSTON HOUSING LOGISTIC REGRESSION
COMPLETE WORKFLOW

1. Load Boston Housing Dataset
2. Inspect Dataset
3. Check Data Types
4. Check Missing Values
5. Check Duplicate Records
6. Remove Duplicate Records
7. Statistical Summary
8. Distribution Analysis
9. Outlier Detection
10. IQR Analysis
11. Outlier Capping
12. Correlation Analysis
13. Convert MEDV into Classes
14. Low Price = 0
15. High Price = 1
16. Train-Test Split
17. Missing Value Imputation
18. Standardization
19. Categorical Encoding
20. Logistic Regression
21. Prediction
22. Accuracy
23. Precision
24. Recall
25. F1 Score
26. ROC-AUC
27. Confusion Matrix
28. Classification Report
29. Cross Validation
30. ROC Curve

============================================================
"""
)

No comments:

Post a Comment