Total Pageviews

Monday, September 28, 2026

LINEAR REGRESSION IN BOSTON HOUSING DATASET

 

# ============================================================
# BOSTON HOUSING DATASET
# COMPLETE DATA PREPROCESSING + LINEAR REGRESSION
# Google Colab Version
# ============================================================


# ============================================================
# 1. IMPORT LIBRARIES
# ============================================================

import numpy as np
import pandas as pd

import matplotlib.pyplot as plt
import seaborn as sns

from sklearn.datasets import fetch_openml

from sklearn.model_selection import train_test_split, cross_val_score

from sklearn.compose import ColumnTransformer

from sklearn.pipeline import Pipeline

from sklearn.impute import SimpleImputer

from sklearn.preprocessing import (
    StandardScaler,
    MinMaxScaler,
    RobustScaler,
    OneHotEncoder,
    PolynomialFeatures
)

from sklearn.feature_selection import SelectKBest, f_regression

from sklearn.linear_model import LinearRegression

from sklearn.metrics import (
    mean_absolute_error,
    mean_squared_error,
    r2_score
)

import warnings
warnings.filterwarnings("ignore")


# ============================================================
# 2. LOAD BOSTON HOUSING DATASET
# ============================================================

boston = fetch_openml(
    name="boston",
    version=1,
    as_frame=True
)

df = boston.frame.copy()

print("Dataset Loaded Successfully")


# ============================================================
# 3. DISPLAY DATASET
# ============================================================

print("\nFirst 5 Rows:")
display(df.head())


# ============================================================
# 4. DATASET SHAPE
# ============================================================

print("\nDataset Shape:")
print(df.shape)


# ============================================================
# 5. COLUMN NAMES
# ============================================================

print("\nColumn Names:")
print(df.columns.tolist())


# ============================================================
# 6. DATA INFORMATION
# ============================================================

print("\nDataset Information:")
df.info()


# ============================================================
# 7. DATA TYPES
# ============================================================

print("\nData Types:")
print(df.dtypes)


# ============================================================
# 8. CHECK MISSING VALUES
# ============================================================

print("\nMissing Values:")
print(df.isnull().sum())


# ============================================================
# 9. MISSING VALUE PERCENTAGE
# ============================================================

missing_percentage = (
    df.isnull().mean() * 100
).sort_values(ascending=False)

print("\nMissing Value Percentage:")
print(missing_percentage)


# ============================================================
# 10. DUPLICATE RECORDS
# ============================================================

print("\nNumber of Duplicate Rows:")
print(df.duplicated().sum())


# Remove duplicates if present

df = df.drop_duplicates()

print("\nShape After Duplicate Removal:")
print(df.shape)


# ============================================================
# 11. STATISTICAL SUMMARY
# ============================================================

print("\nStatistical Summary:")
display(df.describe())


# ============================================================
# 12. CHECK UNIQUE VALUES
# ============================================================

print("\nNumber of Unique Values:")
print(df.nunique())


# ============================================================
# 13. TARGET VARIABLE
# ============================================================

# Boston Housing target is MEDV

target = "MEDV"

print("\nTarget Variable:")
print(target)

print("\nTarget Statistics:")
print(df[target].describe())


# ============================================================
# 14. HISTOGRAM OF TARGET
# ============================================================

plt.figure(figsize=(9,5))

sns.histplot(
    df[target],
    kde=True
)

plt.title("Distribution of House Prices")
plt.xlabel("MEDV")
plt.ylabel("Frequency")

plt.show()


# ============================================================
# 15. CHECK OUTLIERS USING BOXPLOT
# ============================================================

plt.figure(figsize=(12,6))

sns.boxplot(
    data=df
)

plt.title("Boxplot of Boston Housing Features")
plt.xticks(rotation=45)

plt.show()


# ============================================================
# 16. OUTLIER DETECTION USING IQR
# ============================================================

numeric_columns = df.select_dtypes(
    include=np.number
).columns


def find_iqr_outliers(data, column):

    Q1 = data[column].quantile(0.25)

    Q3 = data[column].quantile(0.75)

    IQR = Q3 - Q1

    lower_bound = Q1 - 1.5 * IQR

    upper_bound = Q3 + 1.5 * IQR

    outliers = data[
        (data[column] < lower_bound) |
        (data[column] > upper_bound)
    ]

    return outliers


print("\nIQR Outlier Count:")

for column in numeric_columns:

    count = len(
        find_iqr_outliers(df, column)
    )

    print(
        column,
        ":",
        count
    )


# ============================================================
# 17. OUTLIER TREATMENT USING CAPPING
# ============================================================

# We preserve observations instead of deleting rows.

df_capped = df.copy()

for column in numeric_columns:

    Q1 = df_capped[column].quantile(0.25)

    Q3 = df_capped[column].quantile(0.75)

    IQR = Q3 - Q1

    lower_bound = Q1 - 1.5 * IQR

    upper_bound = Q3 + 1.5 * IQR

    df_capped[column] = df_capped[
        column
    ].clip(
        lower_bound,
        upper_bound
    )


print("\nOutlier capping completed.")


# ============================================================
# 18. CORRELATION MATRIX
# ============================================================

correlation = df.corr(
    numeric_only=True
)

print("\nCorrelation with MEDV:")

print(
    correlation[target]
    .sort_values(
        ascending=False
    )
)


# ============================================================
# 19. CORRELATION HEATMAP
# ============================================================

plt.figure(
    figsize=(12,9)
)

sns.heatmap(
    correlation,
    annot=True,
    fmt=".2f",
    cmap="coolwarm"
)

plt.title(
    "Boston Housing Correlation Heatmap"
)

plt.show()


# ============================================================
# 20. FEATURE / TARGET SEPARATION
# ============================================================

X = df.drop(
    columns=[target]
)

y = df[target]


print("\nFeature Shape:")
print(X.shape)

print("\nTarget Shape:")
print(y.shape)


# ============================================================
# 21. IDENTIFY NUMERICAL AND CATEGORICAL FEATURES
# ============================================================

numeric_features = X.select_dtypes(
    include=np.number
).columns.tolist()

categorical_features = X.select_dtypes(
    exclude=np.number
).columns.tolist()


print("\nNumerical Features:")
print(numeric_features)

print("\nCategorical Features:")
print(categorical_features)


# ============================================================
# 22. TRAIN-TEST SPLIT
# ============================================================

X_train, X_test, y_train, y_test = train_test_split(

    X,
    y,

    test_size=0.20,

    random_state=42
)


print("\nTraining Data:")
print(X_train.shape)

print("\nTesting Data:")
print(X_test.shape)


# ============================================================
# 23. NUMERICAL PREPROCESSING
# ============================================================

numeric_pipeline = Pipeline(

    steps=[

        (
            "imputer",
            SimpleImputer(
                strategy="median"
            )
        ),

        (
            "scaler",
            StandardScaler()
        )

    ]
)


# ============================================================
# 24. CATEGORICAL PREPROCESSING
# ============================================================

categorical_pipeline = Pipeline(

    steps=[

        (
            "imputer",
            SimpleImputer(
                strategy="most_frequent"
            )
        ),

        (
            "encoder",
            OneHotEncoder(
                handle_unknown="ignore"
            )
        )

    ]
)


# ============================================================
# 25. COMBINE PREPROCESSING
# ============================================================

preprocessor = ColumnTransformer(

    transformers=[

        (
            "numeric",
            numeric_pipeline,
            numeric_features
        ),

        (
            "categorical",
            categorical_pipeline,
            categorical_features
        )

    ]

)


# ============================================================
# 26. CREATE LINEAR REGRESSION PIPELINE
# ============================================================

linear_regression_model = Pipeline(

    steps=[

        (
            "preprocessing",
            preprocessor
        ),

        (
            "model",
            LinearRegression()
        )

    ]

)


# ============================================================
# 27. TRAIN LINEAR REGRESSION MODEL
# ============================================================

linear_regression_model.fit(
    X_train,
    y_train
)

print(
    "\nLinear Regression Model Trained Successfully!"
)


# ============================================================
# 28. PREDICTION
# ============================================================

y_pred = linear_regression_model.predict(
    X_test
)


# ============================================================
# 29. MODEL EVALUATION
# ============================================================

mae = mean_absolute_error(
    y_test,
    y_pred
)

mse = mean_squared_error(
    y_test,
    y_pred
)

rmse = np.sqrt(mse)

r2 = r2_score(
    y_test,
    y_pred
)


print("\n==============================")
print("LINEAR REGRESSION RESULTS")
print("==============================")

print(
    "Mean Absolute Error (MAE):",
    mae
)

print(
    "Mean Squared Error (MSE):",
    mse
)

print(
    "Root Mean Squared Error (RMSE):",
    rmse
)

print(
    "R² Score:",
    r2
)


# ============================================================
# 30. ACTUAL VS PREDICTED VALUES
# ============================================================

result = pd.DataFrame({

    "Actual Price": y_test.values,

    "Predicted Price": y_pred

})

print("\nActual vs Predicted:")
display(
    result.head(20)
)


# ============================================================
# 31. ACTUAL VS PREDICTED GRAPH
# ============================================================

plt.figure(
    figsize=(8,6)
)

plt.scatter(
    y_test,
    y_pred,
    alpha=0.7
)

plt.xlabel(
    "Actual House Price"
)

plt.ylabel(
    "Predicted House Price"
)

plt.title(
    "Actual vs Predicted House Prices"
)

# Perfect prediction reference line

minimum = min(
    y_test.min(),
    y_pred.min()
)

maximum = max(
    y_test.max(),
    y_pred.max()
)

plt.plot(
    [minimum, maximum],
    [minimum, maximum],
    linestyle="--"
)

plt.show()


# ============================================================
# 32. RESIDUAL ANALYSIS
# ============================================================

residuals = y_test - y_pred

plt.figure(
    figsize=(9,5)
)

sns.scatterplot(
    x=y_pred,
    y=residuals
)

plt.axhline(
    0,
    linestyle="--"
)

plt.xlabel(
    "Predicted Values"
)

plt.ylabel(
    "Residuals"
)

plt.title(
    "Residual Plot"
)

plt.show()


# ============================================================
# 33. RESIDUAL DISTRIBUTION
# ============================================================

plt.figure(
    figsize=(9,5)
)

sns.histplot(
    residuals,
    kde=True
)

plt.title(
    "Distribution of Residuals"
)

plt.xlabel(
    "Residual"
)

plt.show()


# ============================================================
# 34. CROSS VALIDATION
# ============================================================

cv_scores = cross_val_score(

    linear_regression_model,

    X,

    y,

    cv=5,

    scoring="r2"

)

print("\n5-Fold Cross Validation R² Scores:")

print(cv_scores)

print(
    "\nMean Cross Validation R²:",
    cv_scores.mean()
)


# ============================================================
# 35. TRY MIN-MAX SCALING
# ============================================================

minmax_pipeline = Pipeline(

    steps=[

        (
            "imputer",
            SimpleImputer(
                strategy="median"
            )
        ),

        (
            "scaler",
            MinMaxScaler()
        )

    ]
)


minmax_preprocessor = ColumnTransformer(

    transformers=[

        (
            "numeric",
            minmax_pipeline,
            numeric_features
        )

    ]

)


minmax_model = Pipeline(

    steps=[

        (
            "preprocessing",
            minmax_preprocessor
        ),

        (
            "model",
            LinearRegression()
        )

    ]

)


minmax_model.fit(
    X_train,
    y_train
)


minmax_prediction = minmax_model.predict(
    X_test
)


minmax_r2 = r2_score(
    y_test,
    minmax_prediction
)


print(
    "\nR² using Min-Max Scaling:",
    minmax_r2
)


# ============================================================
# 36. TRY ROBUST SCALING
# ============================================================

robust_pipeline = Pipeline(

    steps=[

        (
            "imputer",
            SimpleImputer(
                strategy="median"
            )
        ),

        (
            "scaler",
            RobustScaler()
        )

    ]

)


robust_preprocessor = ColumnTransformer(

    transformers=[

        (
            "numeric",
            robust_pipeline,
            numeric_features
        )

    ]

)


robust_model = Pipeline(

    steps=[

        (
            "preprocessing",
            robust_preprocessor
        ),

        (
            "model",
            LinearRegression()
        )

    ]

)


robust_model.fit(
    X_train,
    y_train
)


robust_prediction = robust_model.predict(
    X_test
)


robust_r2 = r2_score(
    y_test,
    robust_prediction
)


print(
    "\nR² using Robust Scaling:",
    robust_r2
)


# ============================================================
# 37. COMPARE SCALING METHODS
# ============================================================

comparison = pd.DataFrame({

    "Preprocessing": [
        "StandardScaler",
        "MinMaxScaler",
        "RobustScaler"
    ],

    "R2 Score": [
        r2,
        minmax_r2,
        robust_r2
    ]

})


print("\nScaling Comparison:")

display(
    comparison
)


# ============================================================
# 38. FEATURE SELECTION
# ============================================================

# Select top 8 numerical features

feature_selector = Pipeline(

    steps=[

        (
            "imputer",
            SimpleImputer(
                strategy="median"
            )
        ),

        (
            "scaler",
            StandardScaler()
        ),

        (
            "selection",
            SelectKBest(
                score_func=f_regression,
                k=8
            )
        )

    ]

)


feature_selection_model = Pipeline(

    steps=[

        (
            "preprocessing",
            feature_selector
        ),

        (
            "model",
            LinearRegression()
        )

    ]

)


feature_selection_model.fit(
    X_train[numeric_features],
    y_train
)


feature_prediction = (
    feature_selection_model.predict(
        X_test[numeric_features]
    )
)


feature_r2 = r2_score(
    y_test,
    feature_prediction
)


print(
    "\nR² After Feature Selection:",
    feature_r2
)


# ============================================================
# 39. FINAL MODEL SUMMARY
# ============================================================

print("\n")
print("======================================")
print("       FINAL MODEL SUMMARY")
print("======================================")

print(
    "Number of Training Samples:",
    len(X_train)
)

print(
    "Number of Testing Samples:",
    len(X_test)
)

print(
    "MAE:",
    round(mae, 4)
)

print(
    "MSE:",
    round(mse, 4)
)

print(
    "RMSE:",
    round(rmse, 4)
)

print(
    "R²:",
    round(r2, 4)
)

print(
    "Cross Validation Mean R²:",
    round(cv_scores.mean(), 4)
)

print("======================================")


# ============================================================
# 40. COMPLETE PREPROCESSING WORKFLOW
# ============================================================

print("""
============================================================

COMPLETE DATA PREPROCESSING WORKFLOW

1. Load Dataset
2. Inspect Dataset
3. Check Data Types
4. Check Missing Values
5. Handle Missing Values
6. Check Duplicate Records
7. Remove Duplicates
8. Statistical Summary
9. Distribution Analysis
10. Outlier Detection
11. IQR Outlier Analysis
12. Outlier Capping
13. Correlation Analysis
14. Feature / Target Separation
15. Train-Test Split
16. Numerical Imputation
17. Categorical Imputation
18. Categorical Encoding
19. Standardization
20. Min-Max Scaling
21. Robust Scaling
22. Feature Selection
23. Linear Regression
24. Prediction
25. MAE
26. MSE
27. RMSE
28. R² Score
29. Residual Analysis
30. Cross Validation

============================================================
""")


# ============================================================
# 41. CLASSIFICATION VERSION OF BOSTON HOUSING
# ============================================================
#
# IMPORTANT:
# Linear Regression predicts continuous house prices.
# Therefore, Precision, Recall and Confusion Matrix are
# not directly applicable to the original regression target.
#
# For educational purposes, we convert MEDV into two classes:
#
# 0 = Low Price
# 1 = High Price
#
# Median house price is used as the threshold.
# ============================================================


from sklearn.metrics import (
    accuracy_score,
    precision_score,
    recall_score,
    f1_score,
    confusion_matrix,
    classification_report,
    ConfusionMatrixDisplay
)


# ============================================================
# 42. CREATE CLASSIFICATION TARGET
# ============================================================

classification_df = df.copy()

median_price = classification_df["MEDV"].median()

print("Median House Price:", median_price)


classification_df["Price_Class"] = (
    classification_df["MEDV"] >= median_price
).astype(int)


print("\nClassification Target:")
print(classification_df["Price_Class"].value_counts())


# ============================================================
# 43. CREATE FEATURES AND TARGET
# ============================================================

X_class = classification_df.drop(
    columns=["MEDV", "Price_Class"]
)

y_class = classification_df["Price_Class"]


# ============================================================
# 44. TRAIN-TEST SPLIT
# ============================================================

X_train_class, X_test_class, y_train_class, y_test_class = train_test_split(

    X_class,
    y_class,

    test_size=0.20,

    random_state=42,

    stratify=y_class

)


print("\nClassification Training Shape:")
print(X_train_class.shape)

print("\nClassification Testing Shape:")
print(X_test_class.shape)


# ============================================================
# 45. PREPROCESSING PIPELINE
# ============================================================

numeric_features_class = X_class.select_dtypes(
    include=np.number
).columns.tolist()


categorical_features_class = X_class.select_dtypes(
    exclude=np.number
).columns.tolist()


numeric_pipeline_class = Pipeline(

    steps=[

        (
            "imputer",
            SimpleImputer(
                strategy="median"
            )
        ),

        (
            "scaler",
            StandardScaler()
        )

    ]

)


categorical_pipeline_class = Pipeline(

    steps=[

        (
            "imputer",
            SimpleImputer(
                strategy="most_frequent"
            )
        ),

        (
            "encoder",
            OneHotEncoder(
                handle_unknown="ignore"
            )
        )

    ]

)


classification_preprocessor = ColumnTransformer(

    transformers=[

        (
            "numeric",
            numeric_pipeline_class,
            numeric_features_class
        ),

        (
            "categorical",
            categorical_pipeline_class,
            categorical_features_class
        )

    ]

)


# ============================================================
# 46. CLASSIFICATION MODEL
# ============================================================

# Logistic Regression is used because precision,
# recall and confusion matrix are classification metrics.

from sklearn.linear_model import LogisticRegression


classification_model = Pipeline(

    steps=[

        (
            "preprocessing",
            classification_preprocessor
        ),

        (
            "model",
            LogisticRegression(
                max_iter=2000
            )
        )

    ]

)


# ============================================================
# 47. TRAIN CLASSIFICATION MODEL
# ============================================================

classification_model.fit(

    X_train_class,

    y_train_class

)


print(
    "\nClassification Model Trained Successfully!"
)


# ============================================================
# 48. CLASSIFICATION PREDICTIONS
# ============================================================

y_class_pred = classification_model.predict(
    X_test_class
)


# ============================================================
# 49. ACCURACY
# ============================================================

accuracy = accuracy_score(

    y_test_class,

    y_class_pred

)


print("\n==============================")
print("CLASSIFICATION RESULTS")
print("==============================")

print(
    "Accuracy:",
    round(accuracy, 4)
)


# ============================================================
# 50. PRECISION
# ============================================================

precision = precision_score(

    y_test_class,

    y_class_pred,

    zero_division=0

)


print(
    "Precision:",
    round(precision, 4)
)


# ============================================================
# 51. RECALL
# ============================================================

recall = recall_score(

    y_test_class,

    y_class_pred,

    zero_division=0

)


print(
    "Recall:",
    round(recall, 4)
)


# ============================================================
# 52. F1 SCORE
# ============================================================

f1 = f1_score(

    y_test_class,

    y_class_pred,

    zero_division=0

)


print(
    "F1 Score:",
    round(f1, 4)
)


# ============================================================
# 53. CONFUSION MATRIX
# ============================================================

cm = confusion_matrix(

    y_test_class,

    y_class_pred

)


print("\nConfusion Matrix:")
print(cm)


# ============================================================
# 54. DISPLAY CONFUSION MATRIX
# ============================================================

plt.figure(
    figsize=(7,6)
)

disp = ConfusionMatrixDisplay(

    confusion_matrix=cm,

    display_labels=[
        "Low Price",
        "High Price"
    ]

)

disp.plot()

plt.title(
    "Boston Housing Confusion Matrix"
)

plt.show()


# ============================================================
# 55. EXTRACT TP, TN, FP, FN
# ============================================================

TN, FP, FN, TP = cm.ravel()


print("\n==============================")
print("CONFUSION MATRIX VALUES")
print("==============================")

print(
    "True Negative (TN):",
    TN
)

print(
    "False Positive (FP):",
    FP
)

print(
    "False Negative (FN):",
    FN
)

print(
    "True Positive (TP):",
    TP
)


# ============================================================
# 56. MANUAL PRECISION CALCULATION
# ============================================================

manual_precision = TP / (
    TP + FP
) if (TP + FP) != 0 else 0


print(
    "\nManual Precision:",
    round(manual_precision, 4)
)


# ============================================================
# 57. MANUAL RECALL CALCULATION
# ============================================================

manual_recall = TP / (
    TP + FN
) if (TP + FN) != 0 else 0


print(
    "Manual Recall:",
    round(manual_recall, 4)
)


# ============================================================
# 58. MANUAL F1 SCORE
# ============================================================

manual_f1 = (

    2 * manual_precision * manual_recall

    / (manual_precision + manual_recall)

    if (manual_precision + manual_recall) != 0

    else 0

)


print(
    "Manual F1 Score:",
    round(manual_f1, 4)
)


# ============================================================
# 59. CLASSIFICATION REPORT
# ============================================================

print("\n==============================")
print("CLASSIFICATION REPORT")
print("==============================")

print(

    classification_report(

        y_test_class,

        y_class_pred,

        target_names=[
            "Low Price",
            "High Price"
        ],

        zero_division=0

    )

)


# ============================================================
# 60. METRICS SUMMARY TABLE
# ============================================================

metrics_table = pd.DataFrame({

    "Metric": [

        "Accuracy",

        "Precision",

        "Recall",

        "F1 Score"

    ],

    "Score": [

        accuracy,

        precision,

        recall,

        f1

    ]

})


metrics_table["Score"] = metrics_table[
    "Score"
].round(4)


print("\n==============================")
print("MODEL PERFORMANCE TABLE")
print("==============================")

display(
    metrics_table
)


# ============================================================
# 61. CONFUSION MATRIX TABLE
# ============================================================

confusion_table = pd.DataFrame(

    cm,

    index=[
        "Actual Low Price",
        "Actual High Price"
    ],

    columns=[
        "Predicted Low Price",
        "Predicted High Price"
    ]

)


print("\n==============================")
print("CONFUSION MATRIX TABLE")
print("==============================")

display(
    confusion_table
)


# ============================================================
# 62. COMPLETE METRICS SUMMARY
# ============================================================

final_metrics = pd.DataFrame({

    "Metric": [

        "MAE",
        "MSE",
        "RMSE",
        "R² Score",
        "Classification Accuracy",
        "Classification Precision",
        "Classification Recall",
        "Classification F1 Score"

    ],

    "Value": [

        mae,
        mse,
        rmse,
        r2,
        accuracy,
        precision,
        recall,
        f1

    ]

})


final_metrics["Value"] = final_metrics[
    "Value"
].round(4)


print("\n======================================")
print("COMPLETE BOSTON HOUSING MODEL METRICS")
print("======================================")

display(
    final_metrics
)


# ============================================================
# 63. VISUALIZE PRECISION, RECALL AND F1
# ============================================================

classification_metrics = {

    "Precision": precision,

    "Recall": recall,

    "F1 Score": f1,

    "Accuracy": accuracy

}


plt.figure(
    figsize=(9,6)
)

plt.bar(

    classification_metrics.keys(),

    classification_metrics.values()

)

plt.ylim(
    0,
    1
)

plt.ylabel(
    "Score"
)

plt.xlabel(
    "Metric"
)

plt.title(
    "Classification Performance Metrics"
)

plt.show()


# ============================================================
# 64. CONFUSION MATRIX HEATMAP
# ============================================================

plt.figure(
    figsize=(7,6)
)

sns.heatmap(

    cm,

    annot=True,

    fmt="d",

    cmap="Blues",

    xticklabels=[
        "Predicted Low",
        "Predicted High"
    ],

    yticklabels=[
        "Actual Low",
        "Actual High"
    ]

)

plt.title(
    "Confusion Matrix Heatmap"
)

plt.xlabel(
    "Predicted Class"
)

plt.ylabel(
    "Actual Class"
)

plt.show()


# ============================================================
# 65. INTERPRETATION
# ============================================================

print("""
============================================================
INTERPRETATION OF CLASSIFICATION METRICS
============================================================

Accuracy:
Measures the proportion of all predictions that are correct.

Precision:
Among observations predicted as High Price, precision tells us
how many were actually High Price.

Recall:
Among observations that were actually High Price, recall tells
us how many were correctly identified as High Price.

F1 Score:
Combines Precision and Recall into a single measure.

True Positive (TP):
Actual High Price and predicted High Price.

True Negative (TN):
Actual Low Price and predicted Low Price.

False Positive (FP):
Actual Low Price but predicted High Price.

False Negative (FN):
Actual High Price but predicted Low Price.

============================================================
IMPORTANT:
============================================================

The original Boston Housing problem is a REGRESSION problem.

Therefore:

MAE, MSE, RMSE and R²
        ↓
are the appropriate regression metrics.

Precision, Recall, F1 Score and Confusion Matrix
        ↓
require a CLASSIFICATION problem.

For teaching purposes, the continuous MEDV target was converted
into:

0 = Low Price
1 = High Price

using the median house price as the threshold.

============================================================
""")

No comments:

Post a Comment