{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:#2e4053 ; border-radius:5px; font-size:450%; text-align:center;padding:3.0px; background: #C57804; border-bottom: 3px solid #2c3e50   ; border-top: 8px solid k; border-top: 8px solid k\" > Loan Regression with an Insurance Dataset </center>\n    \n<center style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:#2e4053  ; border-radius:5px; font-size:180%; text-align:center;padding:3.0px; background: silver; border-bottom: 6px solid #2c3e50   ; border-top: 8px solid k\" > Playground Series - Season 4, Episode 12 </center>","metadata":{}},{"cell_type":"markdown","source":"<center>\n<img src=\"https://www.kaggle.com/competitions/84896/images/header\" width=\"500\"/>\n</center>","metadata":{}},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:#2e4053; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: #C57804; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 1. Introduction </p> ","metadata":{}},{"cell_type":"markdown","source":"Can a software decode on our insurance? Yes it can and has since been doing that. How good can we build a model that predict insurance premium amount base on some information fed to it.\nThat will be the purpose of this project.","metadata":{}},{"cell_type":"markdown","source":"### Let's run the numbers\n\n<center>\n<img src=\"https://encrypted-tbn0.gstatic.com/images?q=tbn:ANd9GcSh3vUTO34GvpyOihsVD7XQkExwJrZ_ZI-cFA&s\" width=\"500\"/>\n</center>","metadata":{}},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:#2e4053; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: #C57804; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 2. Load the Tools and Datasets </p> ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom scipy.stats import iqr\nfrom datetime import datetime\nimport warnings\nwarnings.filterwarnings('ignore')\nplt.style.use('ggplot')\n# change default colormap\nplt.rcParams['image.cmap'] = 'Dark2'\n\n# Import the various sklear tools\nfrom sklearn.base import BaseEstimator, TransformerMixin, RegressorMixin\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.compose import make_column_transformer\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.metrics import mean_squared_log_error, accuracy_score, roc_auc_score, roc_curve, r2_score\n\n# from mlxtend.feature_selection import SequentialFeatureSelector as SFS\n# from sklearn.feature_selection import SequentialFeatureSelector as sk_sfs\nfrom sklearn.model_selection import (train_test_split, GridSearchCV, KFold, RepeatedKFold,\n                                     RepeatedStratifiedKFold, RandomizedSearchCV, cross_val_score,\n                                     StratifiedKFold)\nfrom sklearn.ensemble import (RandomForestRegressor, HistGradientBoostingRegressor,\n                              GradientBoostingRegressor, ExtraTreesRegressor, \n                              StackingRegressor, BaggingRegressor,VotingRegressor)\nimport xgboost as xgb\nfrom xgboost import XGBRegressor, XGBClassifier, plot_importance, cv\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import Sequential\nfrom keras import layers\n\nfrom sklearn.svm import LinearSVC\nfrom sklearn.naive_bayes import GaussianNB\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.linear_model import Ridge\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.preprocessing import (MaxAbsScaler, MinMaxScaler, Normalizer,\n                                   PowerTransformer, QuantileTransformer, LabelEncoder,\n                                   RobustScaler, StandardScaler, minmax_scale,\n                                   OneHotEncoder, FunctionTransformer, OrdinalEncoder)\n\nimport yellowbrick\nfrom yellowbrick.classifier import ClassificationReport, DiscriminationThreshold, confusion_matrix\nfrom yellowbrick.regressor import PredictionError\nfrom imblearn.over_sampling import RandomOverSampler\nfrom imblearn.under_sampling import RandomUnderSampler\nfrom yellowbrick.regressor import ResidualsPlot, CooksDistance\nfrom yellowbrick.cluster import KElbowVisualizer, intercluster_distance\n\nimport optuna\nfrom optuna.samplers import TPESampler\nimport plotly.express as px\n\n# Set the color scheme \nmy_scheem = 'flare_r'\nsns.set_palette(my_scheem)\n# sns.color_palette('\"blend:#7AB,#EDA\", as_cmap=True')\n\npd.set_option('display.max_columns', 100)\n# verify the versions\nprint(f'pandas version: {pd.__version__}')\nprint(f'numpy version: {np.__version__}')\nprint(f'seaborn version: {sns.__version__}')\nprint(f'optuna version : {optuna.__version__}')\nprint(f'yellowbrick version: {yellowbrick.__version__}')\n\n# Show all or minimal\nshow_all = False","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-07T23:26:39.471207Z","iopub.execute_input":"2024-12-07T23:26:39.471575Z","iopub.status.idle":"2024-12-07T23:26:39.488039Z","shell.execute_reply.started":"2024-12-07T23:26:39.471543Z","shell.execute_reply":"2024-12-07T23:26:39.486767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the root_mean_squared_log_error score\ndef rmsle_scorer(y_true, y_hat):\n    rmsle = np.sqrt(mean_squared_log_error(y_true, y_hat))\n    return rmsle","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:00.38119Z","iopub.execute_input":"2024-12-07T23:17:00.381919Z","iopub.status.idle":"2024-12-07T23:17:00.387258Z","shell.execute_reply.started":"2024-12-07T23:17:00.381882Z","shell.execute_reply":"2024-12-07T23:17:00.386145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the datasets\n\norig_00 = pd.read_csv('/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv', parse_dates=['Policy Start Date'])\ntrain_00 = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', index_col='id', parse_dates=['Policy Start Date'])\ntest_00 = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', index_col='id', parse_dates=['Policy Start Date'])\nsubmission_00 = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n\n\nprint(f'Shapes of the datasets:\\nTrain: {train_00.shape}\\nTest: {test_00.shape}\\nOriginal: {orig_00.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:00.388511Z","iopub.execute_input":"2024-12-07T23:17:00.388931Z","iopub.status.idle":"2024-12-07T23:17:13.407792Z","shell.execute_reply.started":"2024-12-07T23:17:00.388875Z","shell.execute_reply":"2024-12-07T23:17:13.406706Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Target and features","metadata":{}},{"cell_type":"code","source":"# Define the target\ntarget = 'Premium Amount'\n\n# What are the numerical features?\nnum_cols = test_00.select_dtypes('number').columns.tolist()\n\n# What are the categorical features?\ncat_cols = test_00.select_dtypes(exclude='number').columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:13.410579Z","iopub.execute_input":"2024-12-07T23:17:13.41106Z","iopub.status.idle":"2024-12-07T23:17:13.520435Z","shell.execute_reply.started":"2024-12-07T23:17:13.411012Z","shell.execute_reply":"2024-12-07T23:17:13.519336Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:#2e4053; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: #C57804; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 3. Preview and Examine the Datasets </p> ","metadata":{}},{"cell_type":"markdown","source":"## What does the data say?\n\n<center>\n<img src=\"https://beyondtheory.co.uk/storage/images/other/2016/08/Beyond-Theory-Data-Analysis-Landing-Page-graphic.png\" width=\"400\"/>\n</center>","metadata":{}},{"cell_type":"code","source":"# Preview the train dataset\ntrain_00.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:13.521427Z","iopub.execute_input":"2024-12-07T23:17:13.521787Z","iopub.status.idle":"2024-12-07T23:17:13.549363Z","shell.execute_reply.started":"2024-12-07T23:17:13.521745Z","shell.execute_reply":"2024-12-07T23:17:13.548332Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Count duplicates in the datasets","metadata":{}},{"cell_type":"code","source":"# Check if there are duplicates in the datasets\nfor df_name, df in [('train', train_00), ('test', test_00), ('original', orig_00)]:\n    nunb_of_duplicates = df.duplicated().sum()\n    if nunb_of_duplicates != 0:\n        print(f'{df_name} dataset has {nunb_of_duplicates} duplicates.')\n    else:\n        print(f'The {df_name} dataset has no duplicates')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:13.550545Z","iopub.execute_input":"2024-12-07T23:17:13.550867Z","iopub.status.idle":"2024-12-07T23:17:16.482013Z","shell.execute_reply.started":"2024-12-07T23:17:13.550836Z","shell.execute_reply":"2024-12-07T23:17:16.480746Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Count missing values in the datasets","metadata":{}},{"cell_type":"code","source":"# Count the missing values in the datasets\nnull_count = pd.DataFrame({'NaN in train': train_00.isna().sum(), \n                           'NaN in test': test_00.isna().sum(), \n                           'NaN in original': orig_00.isna().sum(),\n                           '% NaN in train': train_00.isna().mean()*100, \n                           '% NaN in test': test_00.isna().mean()*100, \n                           '% NaN in original': orig_00.isna().mean()*100\n                          }\n                         ).drop(index=[target]).astype('int')\n# pickup only the features with missing values\nnull_count.sort_values(by='NaN in train', ascending=False).head(11).style.background_gradient(cmap='Reds')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:16.483164Z","iopub.execute_input":"2024-12-07T23:17:16.483487Z","iopub.status.idle":"2024-12-07T23:17:18.810543Z","shell.execute_reply.started":"2024-12-07T23:17:16.483455Z","shell.execute_reply":"2024-12-07T23:17:18.809371Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There are lot of missing values in the train, test and original datasets. The missing proportions are same for the three datasets. Some of the features, ­­`­Previous Claims` and `Occupation` are missing close to 30% of their data in both sets.","metadata":{}},{"cell_type":"markdown","source":"### Combine the train and original datasets","metadata":{}},{"cell_type":"code","source":"train_comb = pd.concat([train_00, orig_00], ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:18.811821Z","iopub.execute_input":"2024-12-07T23:17:18.812155Z","iopub.status.idle":"2024-12-07T23:17:18.971836Z","shell.execute_reply.started":"2024-12-07T23:17:18.812122Z","shell.execute_reply":"2024-12-07T23:17:18.97062Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Overview","metadata":{}},{"cell_type":"code","source":"for feat in cat_cols:\n    try:\n        print('\\n{} has {} unique values and has {} missing values which accounts for {:.2f} % of the data.'\n              .format(feat, train_00[feat].nunique(), train_00[feat].isna().sum(), train_00[feat].isna().mean()*100))\n    except:\n        pass","metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:18.973249Z","iopub.execute_input":"2024-12-07T23:17:18.973606Z","iopub.status.idle":"2024-12-07T23:17:20.798581Z","shell.execute_reply.started":"2024-12-07T23:17:18.973572Z","shell.execute_reply":"2024-12-07T23:17:20.797314Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Are the three datasets from the same distribution? Adversarial Validation","metadata":{}},{"cell_type":"code","source":"# Define a function to perform the adversarial validation of two datasets\ndef adversarial_validation(df_1, df_2, name_1, name_2):\n    adv_df_1 = df_1[num_features].copy()\n    adv_df_2 = df_2[num_features].copy()\n\n\n    # label the test and train data with 0 and 1 (it doesn't really matter which is which)\n    adv_df_1 = adv_df_1.assign(adv=1)\n    adv_df_2 = adv_df_2.assign(adv=0)\n\n\n    # combine the training and test data into one big dataset\n    combined = pd.concat([adv_df_1, adv_df_2], axis=0)\n\n    # Shuffle\n    combined = combined.sample(frac=1, random_state=64)\n\n    # perform the binary classification, for example using XGboost\n    X_combined = combined.drop('adv', axis=1)\n    y_combined = combined.adv\n\n    # Define cv spliter\n    cv = StratifiedKFold(n_splits = 5,\n                        shuffle = True,\n                        random_state = 64)\n    \n    # Define the classifier\n    xgb_model = XGBClassifier(max_depth=3,\n                              learning_rate = 0.1,\n                              n_estimators = 100,\n                              objective = 'binary:logistic',\n                              random_state = 64)\n\n    # Get the cross validation scores\n    adv_scores = []\n    for i, _ in enumerate(cv.split(X_combined, y_combined)):\n        X_train, X_valid, y_train, y_valid = train_test_split(X_combined, \n                                                              y_combined, \n                                                              test_size=0.3)\n        xgb_model.fit(X_train, y_train)\n        y_pred = xgb_model.predict_proba(X_valid)[:,1]\n        score = roc_auc_score(y_valid, y_pred)\n        adv_scores.append(score)\n\n#         print(f\"Fold {i+1} AUC Score: {score:.5f}\")\n\n    #Plot the roc_curve\n    mean_auc = np.mean(adv_scores)\n    fpr, tpr, _ = roc_curve(y_valid, y_pred)\n    plt.plot(fpr, tpr, label = 'roc_curve (AUC = %0.4f)' % mean_auc)\n    plt.plot([0,1], [0,1], linestyle = '--', color = 'gray', label = 'Random Guess')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title(f'roc_curve {name_1} vs {name_2}', weight='bold')\n    plt.legend()","metadata":{"trusted":true,"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:20.802346Z","iopub.execute_input":"2024-12-07T23:17:20.802751Z","iopub.status.idle":"2024-12-07T23:17:20.813708Z","shell.execute_reply.started":"2024-12-07T23:17:20.802713Z","shell.execute_reply":"2024-12-07T23:17:20.812333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if show_all:\n    num_features = list(test_00.select_dtypes('number'))\n    plt.figure(figsize=(18,5))\n    plt.subplot(1,3,1)\n    adversarial_validation(train_00, test_00, 'train', 'test')\n    plt.subplot(1,3,2)\n    adversarial_validation(test_00, orig_00, 'train', 'original')\n    plt.subplot(1,3,3)\n    adversarial_validation(train_comb, test_00, 'train_comb', 'test')\nelse:\n    pass","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:20.815308Z","iopub.execute_input":"2024-12-07T23:17:20.815748Z","iopub.status.idle":"2024-12-07T23:17:20.83328Z","shell.execute_reply.started":"2024-12-07T23:17:20.815595Z","shell.execute_reply":"2024-12-07T23:17:20.832017Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As said the feature distributions in train and test datasets are close to, but not exactly the same, as the original.","metadata":{}},{"cell_type":"markdown","source":"### How are the numerical features distributed?","metadata":{}},{"cell_type":"code","source":"def num_cat_distribution(df):\n    fig, axes = plt.subplots(ncols=len(num_cols), nrows=2, figsize=(len(num_cols) * 4, 8))\n    for i, feat in enumerate(num_cols):\n        # Boxen plot\n        sns.boxenplot(data=df, x=feat, ax=axes[0, i])\n        axes[0, i].set_xlabel('')\n        axes[0, i].set_title(f'Plots of {feat}')\n    \n        # Scatter plot\n        sns.violinplot(data=df, x=feat, ax=axes[1, i])\n        axes[1, i].set_xlabel('')\n    \n    plt.tight_layout()\n    plt.show()\n\nnum_cat_distribution(train_00)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:20.834802Z","iopub.execute_input":"2024-12-07T23:17:20.835143Z","iopub.status.idle":"2024-12-07T23:17:43.564523Z","shell.execute_reply.started":"2024-12-07T23:17:20.835109Z","shell.execute_reply":"2024-12-07T23:17:43.563318Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### What are the counts within the various categorical features?","metadata":{}},{"cell_type":"code","source":"for feat in cat_cols:\n    if train_00[feat].nunique() < 10:\n        display(train_00[feat].value_counts().to_frame('number of individuals')\n                .T.style.background_gradient(cmap=my_scheem)\n                .set_properties(**{'font-size': '10pt'})\n               )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:43.566218Z","iopub.execute_input":"2024-12-07T23:17:43.566691Z","iopub.status.idle":"2024-12-07T23:17:45.679139Z","shell.execute_reply.started":"2024-12-07T23:17:43.566616Z","shell.execute_reply":"2024-12-07T23:17:45.677961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### What is the distribution of the target?","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nplt.subplot(211)\nsns.histplot(train_00, x=target, bins=25)\nplt.xlabel('')\nplt.subplot(212)\nsns.boxenplot(train_00, x=target)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:45.680377Z","iopub.execute_input":"2024-12-07T23:17:45.680682Z","iopub.status.idle":"2024-12-07T23:17:46.79041Z","shell.execute_reply.started":"2024-12-07T23:17:45.680635Z","shell.execute_reply":"2024-12-07T23:17:46.789226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(5, 5))\nplt.subplot(211)\nsns.histplot(orig_00, x=target, bins=25)\nplt.xlabel('')\nplt.subplot(212)\nsns.boxenplot(orig_00, x=target)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:46.791797Z","iopub.execute_input":"2024-12-07T23:17:46.792153Z","iopub.status.idle":"2024-12-07T23:17:47.332846Z","shell.execute_reply.started":"2024-12-07T23:17:46.79212Z","shell.execute_reply":"2024-12-07T23:17:47.33162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(train_00, x=target)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:47.334434Z","iopub.execute_input":"2024-12-07T23:17:47.334912Z","iopub.status.idle":"2024-12-07T23:17:47.68963Z","shell.execute_reply.started":"2024-12-07T23:17:47.334864Z","shell.execute_reply":"2024-12-07T23:17:47.68835Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We need to remove outliers from the target.","metadata":{}},{"cell_type":"code","source":"def drop_outliers(df, feature):\n    Q1 = df[feature].quantile(0.25)\n    Q3 = df[feature].quantile(0.75)\n    IQR = Q3 - Q1\n    \n    lower_bound = Q1 - 1.5 * IQR\n    upper_bound = Q3 + 1.5 * IQR\n    \n    filtered_df = df[(df[feature] >= lower_bound) & (df[feature] <= upper_bound)]\n    \n    return filtered_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:47.6914Z","iopub.execute_input":"2024-12-07T23:17:47.691955Z","iopub.status.idle":"2024-12-07T23:17:47.698891Z","shell.execute_reply.started":"2024-12-07T23:17:47.691899Z","shell.execute_reply":"2024-12-07T23:17:47.697734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"drop_outliers(train_00, target)[target].plot.kde()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:17:47.700544Z","iopub.execute_input":"2024-12-07T23:17:47.701088Z","iopub.status.idle":"2024-12-07T23:18:10.864635Z","shell.execute_reply.started":"2024-12-07T23:17:47.701042Z","shell.execute_reply":"2024-12-07T23:18:10.863477Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:#2e4053; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: #C57804; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 4. Preprocessor </p> ","metadata":{}},{"cell_type":"markdown","source":"### Fill the missing values","metadata":{}},{"cell_type":"code","source":"def missing_handler(df):\n    for feat in cat_cols:\n        df[feat] = df[feat].fillna('unknown')\n    for feat in num_cols:\n        df[feat] = df[feat].fillna(df[feat].median())\n        # df[feat] = df[feat].fillna(df[feat].min())\n    return df\n\n\ntrain_01 = missing_handler(train_00)\ntrain_comb_01 = missing_handler(train_comb)\ntest_01 = missing_handler(test_00)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:10.865998Z","iopub.execute_input":"2024-12-07T23:18:10.866337Z","iopub.status.idle":"2024-12-07T23:18:14.133211Z","shell.execute_reply.started":"2024-12-07T23:18:10.866306Z","shell.execute_reply":"2024-12-07T23:18:14.131942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f'There are {train_01.isna().sum().sum()} missing values left')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:14.13456Z","iopub.execute_input":"2024-12-07T23:18:14.134953Z","iopub.status.idle":"2024-12-07T23:18:14.744109Z","shell.execute_reply.started":"2024-12-07T23:18:14.13492Z","shell.execute_reply":"2024-12-07T23:18:14.742764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for feat in ['Number of Dependents', 'Previous Claims', 'Vehicle Age', 'Insurance Duration']:\n    display(train_01[feat].value_counts().to_frame().style.background_gradient())","metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:14.74574Z","iopub.execute_input":"2024-12-07T23:18:14.746173Z","iopub.status.idle":"2024-12-07T23:18:14.868Z","shell.execute_reply.started":"2024-12-07T23:18:14.746133Z","shell.execute_reply":"2024-12-07T23:18:14.866609Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(16, 12))\nfor f, feat in enumerate(['Number of Dependents', 'Previous Claims', 'Vehicle Age', 'Insurance Duration'], start=1):\n    plt.subplot(1, 4, f)\n    pd.Series({' ': 1}).plot.pie(colors=['grey'], radius=0.2, shadow=False)\n    train_01[feat].value_counts().plot.pie(autopct='%.1f%%', radius=1.2, pctdistance=0.44, shadow=False, \n                                         textprops={'color':'black', 'rotation':True, 'weight':'bold', 'size': 8},\n                                         startangle=90 , rotatelabels=True,\n                                         labeldistance=0.7, wedgeprops={'width':0.9}, frame=True)\n    plt.ylabel('')\n    plt.title(feat, color='black', fontsize=10, weight='bold')\n    plt.tight_layout()","metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:14.869524Z","iopub.execute_input":"2024-12-07T23:18:14.869984Z","iopub.status.idle":"2024-12-07T23:18:16.444202Z","shell.execute_reply.started":"2024-12-07T23:18:14.869935Z","shell.execute_reply":"2024-12-07T23:18:16.442591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Generate a mask for the upper triangle \nmask = np.triu(np.ones_like(train_01[num_cols].corr(method='spearman'), dtype=bool))\n# Generate a custom diverging colormap \ncmap = my_scheem[:-2]\n\nsns.heatmap(train_01[num_cols].corr(), annot=True, cmap=cmap, mask=mask, fmt='.3f')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:16.445827Z","iopub.execute_input":"2024-12-07T23:18:16.446286Z","iopub.status.idle":"2024-12-07T23:18:19.047297Z","shell.execute_reply.started":"2024-12-07T23:18:16.446238Z","shell.execute_reply":"2024-12-07T23:18:19.04613Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We cannot identify any pair of features with high correlation.","metadata":{}},{"cell_type":"code","source":"# Get the year, month and age of the account\ndef get_the_month_year(df):\n    # df['month'] = pd.to_datetime(df['Policy Start Date']).dt.month\n    df['year'] = pd.to_datetime(df['Policy Start Date']).dt.year.astype('string')\n    current_date = datetime.now()\n    df['account_age'] = df['Policy Start Date'].apply(lambda x: current_date.year - x.year - ((current_date.month, current_date.day) < (x.month, x.day)))\n    df = df.drop(columns = ['Policy Start Date'])\n    return df\n\n\ntrain_02 = get_the_month_year(train_01)\ntrain_comb_02 = get_the_month_year(train_comb)\ntest_02 = get_the_month_year(test_01)\ntrain_02.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:19.049414Z","iopub.execute_input":"2024-12-07T23:18:19.049888Z","iopub.status.idle":"2024-12-07T23:18:31.157342Z","shell.execute_reply.started":"2024-12-07T23:18:19.049826Z","shell.execute_reply":"2024-12-07T23:18:31.156145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define X and y\nuse_comb=False\n\nif use_comb:\n    X = train_comb_02.copy()\n    y = X.pop(target)\nelse:\n    X = train_02.copy()\n    y = X.pop(target)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:31.158496Z","iopub.execute_input":"2024-12-07T23:18:31.158841Z","iopub.status.idle":"2024-12-07T23:18:31.866439Z","shell.execute_reply.started":"2024-12-07T23:18:31.158808Z","shell.execute_reply":"2024-12-07T23:18:31.864881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# What are the categorical features?\ncat_cols = X.select_dtypes(exclude='number').columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:31.867768Z","iopub.execute_input":"2024-12-07T23:18:31.868215Z","iopub.status.idle":"2024-12-07T23:18:32.417225Z","shell.execute_reply.started":"2024-12-07T23:18:31.86815Z","shell.execute_reply":"2024-12-07T23:18:32.415954Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# feat_to_scale = X_ts.select_dtypes(include='number').columns.tolist()\nfeatures_trans = make_column_transformer(\n    # (label_encoder, 'Gender'),\n    (OrdinalEncoder(), cat_cols),\n    (MinMaxScaler(), num_cols),\n    remainder='passthrough', \n    sparse_threshold=0\n)\n\nfeatures_trans","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:32.41845Z","iopub.execute_input":"2024-12-07T23:18:32.418803Z","iopub.status.idle":"2024-12-07T23:18:32.432695Z","shell.execute_reply.started":"2024-12-07T23:18:32.418769Z","shell.execute_reply":"2024-12-07T23:18:32.431481Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### How does tha features_trans tranform the datasets?","metadata":{}},{"cell_type":"code","source":"X_p = X.copy()\n\nX_p = features_trans.fit_transform(X_p)\n\npd.DataFrame(X_p, columns=X.columns).head(3)","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:32.43998Z","iopub.execute_input":"2024-12-07T23:18:32.440335Z","iopub.status.idle":"2024-12-07T23:18:36.603966Z","shell.execute_reply.started":"2024-12-07T23:18:32.440304Z","shell.execute_reply":"2024-12-07T23:18:36.602783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = KMeans(n_init='auto')\n\nX_num = X[num_cols]\n\nviz = KElbowVisualizer(model, k=(4,16))\nviz.fit(X_num)\nviz.show()\nplt.show()","metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-07T23:18:36.60522Z","iopub.execute_input":"2024-12-07T23:18:36.605549Z","iopub.status.idle":"2024-12-07T23:19:09.359318Z","shell.execute_reply.started":"2024-12-07T23:18:36.605518Z","shell.execute_reply":"2024-12-07T23:19:09.358064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"intercluster_distance(KMeans(6, random_state=42), X_num)","metadata":{"trusted":true,"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:09.360567Z","iopub.execute_input":"2024-12-07T23:19:09.360919Z","iopub.status.idle":"2024-12-07T23:19:20.538479Z","shell.execute_reply.started":"2024-12-07T23:19:09.360884Z","shell.execute_reply":"2024-12-07T23:19:20.537348Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert target to log scale\ny_log = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:20.542391Z","iopub.execute_input":"2024-12-07T23:19:20.542788Z","iopub.status.idle":"2024-12-07T23:19:20.572976Z","shell.execute_reply.started":"2024-12-07T23:19:20.542751Z","shell.execute_reply":"2024-12-07T23:19:20.571789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_tr, X_ts, y_tr_log, y_ts_log = train_test_split(X, y_log, test_size=0.25, random_state=42)\n\n[d.shape for d in [X_tr, X_ts, y_tr_log, y_ts_log]]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:20.574283Z","iopub.execute_input":"2024-12-07T23:19:20.574728Z","iopub.status.idle":"2024-12-07T23:19:21.840495Z","shell.execute_reply.started":"2024-12-07T23:19:20.57465Z","shell.execute_reply":"2024-12-07T23:19:21.838992Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_tr, y_ts = np.expm1(y_tr_log), np.expm1(y_ts_log)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:21.841799Z","iopub.execute_input":"2024-12-07T23:19:21.842135Z","iopub.status.idle":"2024-12-07T23:19:21.874945Z","shell.execute_reply.started":"2024-12-07T23:19:21.842101Z","shell.execute_reply":"2024-12-07T23:19:21.873801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.DataFrame(X_tr).head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:21.876268Z","iopub.execute_input":"2024-12-07T23:19:21.876609Z","iopub.status.idle":"2024-12-07T23:19:21.897496Z","shell.execute_reply.started":"2024-12-07T23:19:21.876575Z","shell.execute_reply":"2024-12-07T23:19:21.896319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# viz = CooksDistance()\n# viz.fit(X_tr[num_cols], y_tr_log)\n# #viz.score(X_test, y_test)\n# viz.show()\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:21.898949Z","iopub.execute_input":"2024-12-07T23:19:21.899302Z","iopub.status.idle":"2024-12-07T23:19:21.910223Z","shell.execute_reply.started":"2024-12-07T23:19:21.89927Z","shell.execute_reply":"2024-12-07T23:19:21.909163Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:#2e4053; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: #C57804; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 5. Modeling </p> ","metadata":{}},{"cell_type":"markdown","source":"<center>\n<img src=\"https://media.geeksforgeeks.org/wp-content/uploads/20240215112547/Data-Modeling-in-Analysis.webp\" width=\"500\"/>\n</center>","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\n\ndef objective(trial): \n    # Define LightGBM parameters\n    params = {\n        \"objective\": 'regression',\n        # \"metric\": \"l2\",\n        \"lambda_l1\": trial.suggest_float(\"lambda_l1\", 1e-8, 10.0, log=True),\n        \"lambda_l2\": trial.suggest_float(\"lambda_l2\", 1e-8, 10.0, log=True),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 1e-3, 0.5, log=True),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 2, 256),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 2, 10),\n        \"feature_fraction\": trial.suggest_float(\"feature_fraction\", 0.4, 1.0),\n        \"bagging_fraction\": trial.suggest_float(\"bagging_fraction\", 0.4, 1.0),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 1, 7),\n        \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 5, 100),\n    }\n    \n    # Create LightGBM dataset\n    model_pipe = make_pipeline(features_trans,\n                               LGBMRegressor(**params, verbose=-1))\n    \n    # Train the model\n    model_pipe.fit(X_tr, y_tr_log)\n    \n    # Make predictions\n    preds_log = model_pipe.predict(X_ts)\n    preds = np.expm1(preds_log)\n\n    # Calculate the mean squared error\n    score = rmsle_scorer(np.expm1(y_ts_log), preds)\n    ## score = r2_score(np.expm1(y_ts), preds)\n    # score = np.mean(np.abs(np.expm1(y_ts) - preds))\n    return score\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:21.914168Z","iopub.execute_input":"2024-12-07T23:19:21.914679Z","iopub.status.idle":"2024-12-07T23:19:21.925739Z","shell.execute_reply.started":"2024-12-07T23:19:21.914616Z","shell.execute_reply":"2024-12-07T23:19:21.924628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the optimization function\nseed= 42\ndef run_optuna(n_trials=1):\n    if n_trials > 1:\n        # Define the sampler\n        sampler = TPESampler(seed=seed)\n        # Create and optimize the optuna study\n        study = optuna.create_study(direction='minimize', sampler=sampler, study_name='R2_Study')\n        study.optimize(lambda trial: objective(trial), n_trials=n_trials, show_progress_bar=True)\n\n        best_study_params = study.best_params\n    else:\n        best_study_params = {'lambda_l1': 0.007787038174438944, \n                            'lambda_l2': 0.007475379286754892, \n                            'learning_rate': 0.06469248141029064, \n                            'num_leaves': 101, \n                            'max_depth': 9, \n                            'feature_fraction': 0.9838734620932038, \n                            'bagging_fraction': 0.871730858029871, \n                            'bagging_freq': 3, \n                            'min_child_samples': 50}\n    \n    print(f'\\nThe best lgbm hyperparameters: \\n{best_study_params}')\n    print('')\n    return best_study_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:21.927224Z","iopub.execute_input":"2024-12-07T23:19:21.927525Z","iopub.status.idle":"2024-12-07T23:19:21.94498Z","shell.execute_reply.started":"2024-12-07T23:19:21.927496Z","shell.execute_reply":"2024-12-07T23:19:21.943601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_lgbm_params = run_optuna(n_trials=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:21.94633Z","iopub.execute_input":"2024-12-07T23:19:21.946717Z","iopub.status.idle":"2024-12-07T23:19:21.966442Z","shell.execute_reply.started":"2024-12-07T23:19:21.946632Z","shell.execute_reply":"2024-12-07T23:19:21.965237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kfold = StratifiedKFold(n_splits=2, shuffle=True, random_state=42)\n\nfor f, (train_ind, val_ind) in enumerate(kfold.split(X, y), start=1):\n\n    model = make_pipeline(features_trans, LGBMRegressor(**best_lgbm_params, verbose=-1))\n    X_train, X_val = X.iloc[train_ind], X.iloc[val_ind]\n    y_train_log, y_val_log = y_log.iloc[train_ind], y_log.iloc[val_ind]\n\n    model.fit(X_train,  y_train_log)\n\n    y_pred_log = model.predict(X_val)\n    y_pred = np.expm1(y_pred_log)\n    y_val = np.expm1(y_val_log)\n    \n    score = rmsle_scorer(y_val, y_pred)\n    \n    print('Fold_{} score: {:.6f}'.format(f, score))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:19:21.96761Z","iopub.execute_input":"2024-12-07T23:19:21.96798Z","iopub.status.idle":"2024-12-07T23:20:10.704486Z","shell.execute_reply.started":"2024-12-07T23:19:21.967944Z","shell.execute_reply":"2024-12-07T23:20:10.703341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a wrapper class\nclass CatBoostWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self):\n        self.model = CatBoostRegressor(verbose=0)\n    \n    def fit(self, X, y):\n        self.model.fit(X, y)\n        return self\n    \n    def predict(self, X):\n        return self.model.predict(X)\n\n# Initialize the wrapped model with best parameters\ncat_reg_wrapped = CatBoostWrapper()\n\n# # Fit the model\n# model.fit(X_train, y_train)\n\n# # Visualize with Yellowbrick\n# visualizer = PredictionError(model)\n# visualizer.fit(X_train, y_train)\n# visualizer.score(X_val, y_val)\n# visualizer.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:27:47.609155Z","iopub.execute_input":"2024-12-07T23:27:47.609566Z","iopub.status.idle":"2024-12-07T23:27:47.616368Z","shell.execute_reply.started":"2024-12-07T23:27:47.609529Z","shell.execute_reply":"2024-12-07T23:27:47.615102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Choice of model\ndef choose_model(use = 'hist'):\n    # define the model\n    if use == 'lgb':\n        model = LGBMRegressor(**best_lgbm_params, verbose=-1)\n    elif use == 'rfr':\n        model = RandomForestRegressor(n_estimators=250, max_depth=8)\n    elif use == 'hist':\n        model = HistGradientBoostingRegressor()\n    elif use == 'cat':\n        # model = CatBoostRegressor(n_estimators=400, eval_fraction=0.2, verbose=0)\n        model = cat_reg_wrapped\n    elif use == 'ridge':\n        model = Ridge()\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:28:06.495938Z","iopub.execute_input":"2024-12-07T23:28:06.496325Z","iopub.status.idle":"2024-12-07T23:28:06.50257Z","shell.execute_reply.started":"2024-12-07T23:28:06.496291Z","shell.execute_reply":"2024-12-07T23:28:06.50177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = choose_model(use = 'cat')\n\nmodel_pipe = make_pipeline(features_trans, model)\n\nmodel_pipe.fit(X_tr, y_tr_log)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:28:08.879209Z","iopub.execute_input":"2024-12-07T23:28:08.879612Z","iopub.status.idle":"2024-12-07T23:29:32.719847Z","shell.execute_reply.started":"2024-12-07T23:28:08.879572Z","shell.execute_reply":"2024-12-07T23:29:32.718582Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_ts = np.expm1(y_ts_log)\n\n# Prediction on the ts set\npred_ts_log = model_pipe.predict(X_ts)\npred_ts = np.expm1(pred_ts_log)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:29:37.344075Z","iopub.execute_input":"2024-12-07T23:29:37.344463Z","iopub.status.idle":"2024-12-07T23:29:40.433595Z","shell.execute_reply.started":"2024-12-07T23:29:37.344428Z","shell.execute_reply":"2024-12-07T23:29:40.432305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rmsle_scorer(y_ts, pred_ts)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:29:44.467061Z","iopub.execute_input":"2024-12-07T23:29:44.467455Z","iopub.status.idle":"2024-12-07T23:29:44.490611Z","shell.execute_reply.started":"2024-12-07T23:29:44.467419Z","shell.execute_reply":"2024-12-07T23:29:44.489431Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The residual is the difference between the predicted and actual values. Ideal there shouldn't be any residual i.e residual = 0. Hence on the graph of residuals we want our points to be closer to the `zero line`.","metadata":{}},{"cell_type":"code","source":"# Instantiate the linear model and visualizer\nmodel_pipe = make_pipeline(features_trans, model)\n\nvisualizer = ResidualsPlot(model_pipe)\n\nvisualizer.fit(X_tr, y_tr_log)  # Fit the training data to the visualizer\nvisualizer.score(X_ts, y_ts_log)  # Evaluate the model on the test data\nvisualizer.poof()                 # Finalize and render the figure\nprint('not the best')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:29:48.932277Z","iopub.execute_input":"2024-12-07T23:29:48.933283Z","iopub.status.idle":"2024-12-07T23:30:27.369015Z","shell.execute_reply.started":"2024-12-07T23:29:48.933239Z","shell.execute_reply":"2024-12-07T23:30:27.367936Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The Residual chart show that the model is not performing well:\n * The residuals range from about `-5000 to 2000`. which is a very wide range for values that are in the range `0, 5000`.\n * The R2 score are too low and suggest a random model.","metadata":{}},{"cell_type":"code","source":"pred_ts = np.expm1(model_pipe.predict(X_ts))\npred_ts","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:30:37.899757Z","iopub.execute_input":"2024-12-07T23:30:37.9002Z","iopub.status.idle":"2024-12-07T23:30:41.054752Z","shell.execute_reply.started":"2024-12-07T23:30:37.900163Z","shell.execute_reply":"2024-12-07T23:30:41.053701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ax = pd.DataFrame(pred_ts).plot.hist(bins=40)\ny_ts.plot.hist(bins=40, ax=ax, alpha=0.5)\nplt.title('Distribution true vs predicted target in validation set')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:30:41.058467Z","iopub.execute_input":"2024-12-07T23:30:41.058887Z","iopub.status.idle":"2024-12-07T23:30:41.469848Z","shell.execute_reply.started":"2024-12-07T23:30:41.058849Z","shell.execute_reply":"2024-12-07T23:30:41.468719Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As opposed to our expectation, the predicted values fall in a norrower range compared to that of the train data.","metadata":{}},{"cell_type":"code","source":"viz = PredictionError(model_pipe, line_colors='green')\nviz.fit(X_tr, y_tr_log)\nviz.score(X_ts, y_ts_log)\nviz.show()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:30:41.471174Z","iopub.execute_input":"2024-12-07T23:30:41.471497Z","iopub.status.idle":"2024-12-07T23:30:51.55629Z","shell.execute_reply.started":"2024-12-07T23:30:41.471465Z","shell.execute_reply":"2024-12-07T23:30:51.555001Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"It looks more like a random prediction.","metadata":{}},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:#2e4053; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: #C57804; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 6. Prediction and submission of test </p> ","metadata":{}},{"cell_type":"code","source":"# Prediction on test set\npreds = model_pipe.predict(test_02)\n\nsubmission_00[target] = np.expm1(preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:30:51.55907Z","iopub.execute_input":"2024-12-07T23:30:51.559721Z","iopub.status.idle":"2024-12-07T23:30:59.739806Z","shell.execute_reply.started":"2024-12-07T23:30:51.559641Z","shell.execute_reply":"2024-12-07T23:30:59.738545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_00","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:30:59.741045Z","iopub.execute_input":"2024-12-07T23:30:59.74138Z","iopub.status.idle":"2024-12-07T23:30:59.753562Z","shell.execute_reply.started":"2024-12-07T23:30:59.741348Z","shell.execute_reply":"2024-12-07T23:30:59.752489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ax = submission_00[target].plot.hist(bins=40)\ny_ts.plot.hist(bins=20, ax=ax, alpha=0.5)\nplt.title('Distribution true vs predicted target in validation set')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:30:59.75515Z","iopub.execute_input":"2024-12-07T23:30:59.755599Z","iopub.status.idle":"2024-12-07T23:31:00.27925Z","shell.execute_reply.started":"2024-12-07T23:30:59.75555Z","shell.execute_reply":"2024-12-07T23:31:00.277937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_00.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T23:31:00.280961Z","iopub.execute_input":"2024-12-07T23:31:00.2814Z","iopub.status.idle":"2024-12-07T23:31:01.986361Z","shell.execute_reply.started":"2024-12-07T23:31:00.281352Z","shell.execute_reply":"2024-12-07T23:31:01.984946Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:#2e4053; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: #C57804; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > We don't expect this atempt to give a good result. More is still to come... </p> ","metadata":{}}]}