"""Implementation of GradientBoostingRegressor in sklearn using the
diabetes dataset, a popular regression problem used to predict
disease progression one year after baseline.

Note: this example previously used the Boston house-price dataset,
which was removed from scikit-learn (>=1.2) for ethical reasons.
``load_diabetes`` is a drop-in bundled alternative that ships with
scikit-learn, so the example runs offline.
"""

import matplotlib.pyplot as plt
import pandas as pd
from sklearn.datasets import load_diabetes
from sklearn.ensemble import GradientBoostingRegressor
from sklearn.metrics import mean_squared_error, r2_score
from sklearn.model_selection import train_test_split


def main() -> None:
    # loading the dataset from sklearn
    df = load_diabetes()
    print(df.keys())
    # now let's construct a data frame
    df_data = pd.DataFrame(df.data, columns=df.feature_names)
    # let's add the target to the dataframe
    df_data["Target"] = df.target
    # print the first five rows using the head function
    print(df_data.head())
    # Summary statistics
    print(df_data.describe().T)
    # Feature selection

    x = df_data.iloc[:, :-1]
    y = df_data.iloc[:, -1]  # target variable
    # split the data with 75% train and 25% test sets.
    x_train, x_test, y_train, y_test = train_test_split(
        x, y, random_state=0, test_size=0.25
    )

    model = GradientBoostingRegressor(
        n_estimators=500, max_depth=5, min_samples_split=4, learning_rate=0.01
    )
    # training the model
    model.fit(x_train, y_train)
    # to see how good the model fit the data
    training_score = model.score(x_train, y_train).round(3)
    test_score = model.score(x_test, y_test).round(3)
    print("Training score of GradientBoosting is :", training_score)
    print("The test score of GradientBoosting is :", test_score)
    # Let us evaluate the model by finding the errors
    y_pred = model.predict(x_test)

    # The mean squared error
    print(f"Mean squared error: {mean_squared_error(y_test, y_pred):.2f}")
    # Explained variance score: 1 is perfect prediction
    print(f"Test Variance score: {r2_score(y_test, y_pred):.2f}")

    # So let's run the model against the test data
    _fig, ax = plt.subplots()
    ax.scatter(y_test, y_pred, edgecolors=(0, 0, 0))
    ax.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], "k--", lw=4)
    ax.set_xlabel("Actual")
    ax.set_ylabel("Predicted")
    ax.set_title("Truth vs Predicted")
    # this show function will display the plotting
    plt.show()


if __name__ == "__main__":
    main()
