{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":9178166,"sourceType":"datasetVersion","datasetId":5547076}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color: orange ; border-radius:5px; font-size:450%; text-align:center;padding:3.0px; background: darkgreen; border-bottom: 3px solid #2c3e50   ; border-top: 8px solid k; border-top: 8px solid k\" > Loan Regression with an Insurance Dataset </center>\n    \n<center style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:darkgreen  ; border-radius:5px; font-size:180%; text-align:center;padding:3.0px; background: orange; border-bottom: 6px solid #2c3e50   ; border-top: 8px solid k\" > Playground Series - Season 4, Episode 12 </center>","metadata":{}},{"cell_type":"markdown","source":"<center>\n<img src=\"https://www.kaggle.com/competitions/84896/images/header\" width=\"500\"/>\n</center>","metadata":{}},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 1. Introduction </p> \n\n\nCan a software decode on our insurance? Yes it can and has since been doing that. How good can we build a model that predict insurance premium amount base on some information fed to it.\nThat will be the purpose of this project.\n\n\n### Let's run the numbers\n\n<center>\n<img src=\"https://encrypted-tbn0.gstatic.com/images?q=tbn:ANd9GcSh3vUTO34GvpyOihsVD7XQkExwJrZ_ZI-cFA&s\" width=\"500\"/>\n</center>","metadata":{}},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 2. Load tools and datasets </p> ","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom scipy.stats import iqr\nfrom datetime import datetime\nimport warnings\nwarnings.filterwarnings('ignore')\nplt.style.use('ggplot')\n# change default colormap\nplt.rcParams['image.cmap'] = 'Dark2'\n\n# Import the various sklear tools\nfrom sklearn.base import BaseEstimator, TransformerMixin, RegressorMixin\nimport category_encoders as ce\nfrom sklearn.pipeline import make_pipeline, Pipeline\nfrom sklearn.compose import make_column_transformer\nfrom sklearn.decomposition import PCA\nfrom sklearn.cluster import KMeans\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.metrics import mean_squared_log_error, accuracy_score, roc_auc_score, roc_curve, r2_score\n\n# from mlxtend.feature_selection import SequentialFeatureSelector as SFS\n# from sklearn.feature_selection import SequentialFeatureSelector as sk_sfs\nfrom sklearn.model_selection import (train_test_split, GridSearchCV, KFold, RepeatedKFold,\n                                     RepeatedStratifiedKFold, RandomizedSearchCV, cross_val_score,\n                                     StratifiedKFold)\nfrom sklearn.ensemble import (RandomForestRegressor, HistGradientBoostingRegressor,\n                              GradientBoostingRegressor, ExtraTreesRegressor, \n                              StackingRegressor, BaggingRegressor,VotingRegressor)\nimport xgboost as xgb\nfrom xgboost import XGBRegressor, XGBClassifier, plot_importance, cv\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import Sequential\nfrom keras import layers\n\nfrom sklearn.svm import LinearSVC\nfrom sklearn.naive_bayes import GaussianNB\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor, Pool\nfrom sklearn.linear_model import Ridge\nfrom sklearn.neighbors import KNeighborsRegressor\nfrom sklearn.preprocessing import (MaxAbsScaler, MinMaxScaler, Normalizer,\n                                   PowerTransformer, QuantileTransformer, LabelEncoder,\n                                   RobustScaler, StandardScaler, minmax_scale,\n                                   OneHotEncoder, FunctionTransformer, OrdinalEncoder)\n\nimport yellowbrick\nfrom yellowbrick.classifier import ClassificationReport, DiscriminationThreshold, confusion_matrix\nfrom yellowbrick.regressor import PredictionError\nfrom imblearn.over_sampling import RandomOverSampler\nfrom imblearn.under_sampling import RandomUnderSampler\nfrom yellowbrick.regressor import ResidualsPlot, CooksDistance\nfrom yellowbrick.cluster import KElbowVisualizer, intercluster_distance\nfrom sklearn.metrics import make_scorer\n\nimport optuna\nfrom optuna.samplers import TPESampler\nimport plotly.express as px\n\n# Set the color scheme \nmy_scheem = 'viridis_r'\nsns.set_palette(my_scheem)\n# sns.color_palette('\"blend:#7AB,#EDA\", as_cmap=True')\n\npd.set_option('display.max_columns', 100)\n# verify the versions\nprint(f'pandas version: {pd.__version__}')\nprint(f'numpy version: {np.__version__}')\nprint(f'seaborn version: {sns.__version__}')\nprint(f'optuna version : {optuna.__version__}')\nprint(f'yellowbrick version: {yellowbrick.__version__}')\n\n# Show all or minimal\nshow_all = False","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:02.318942Z","iopub.execute_input":"2024-12-10T01:37:02.319417Z","iopub.status.idle":"2024-12-10T01:37:23.245861Z","shell.execute_reply.started":"2024-12-10T01:37:02.319376Z","shell.execute_reply":"2024-12-10T01:37:23.244639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target = 'Premium Amount'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:23.247567Z","iopub.execute_input":"2024-12-10T01:37:23.248288Z","iopub.status.idle":"2024-12-10T01:37:23.253088Z","shell.execute_reply.started":"2024-12-10T01:37:23.248251Z","shell.execute_reply":"2024-12-10T01:37:23.252012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the root_mean_squared_log_error score\ndef rmsle_scorer(y_true, y_hat):\n    rmsle = np.sqrt(mean_squared_log_error(y_true, y_hat))\n    return rmsle\n\n# Create a scorer from the custom metric\ncustom_scorer = make_scorer(rmsle_scorer, greater_is_better=False)\ncustom_scorer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:23.25442Z","iopub.execute_input":"2024-12-10T01:37:23.25476Z","iopub.status.idle":"2024-12-10T01:37:23.286133Z","shell.execute_reply.started":"2024-12-10T01:37:23.254726Z","shell.execute_reply":"2024-12-10T01:37:23.284911Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the datasets\n\norig_00 = pd.read_csv('/kaggle/input/insurance-premium-prediction/Insurance Premium Prediction Dataset.csv', parse_dates=['Policy Start Date'])\norig_00 = orig_00[-orig_00[target].isna()]\ntrain_00 = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', index_col='id', parse_dates=['Policy Start Date'])\ntest_00 = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', index_col='id', parse_dates=['Policy Start Date'])\nsubmission_00 = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\n\n\nprint(f'Shapes of the datasets:\\nTrain: {train_00.shape}\\nTest: {test_00.shape}\\nOriginal: {orig_00.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:23.288564Z","iopub.execute_input":"2024-12-10T01:37:23.288927Z","iopub.status.idle":"2024-12-10T01:37:36.2921Z","shell.execute_reply.started":"2024-12-10T01:37:23.288893Z","shell.execute_reply":"2024-12-10T01:37:36.29088Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 3. How do the data look like? </p> ","metadata":{}},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 3.1. Dataset Preview </p> ","metadata":{}},{"cell_type":"code","source":"train_00.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:36.293592Z","iopub.execute_input":"2024-12-10T01:37:36.293957Z","iopub.status.idle":"2024-12-10T01:37:36.322079Z","shell.execute_reply.started":"2024-12-10T01:37:36.29392Z","shell.execute_reply":"2024-12-10T01:37:36.320997Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 3.2. Are there duplicates in the datasets? </p> ","metadata":{}},{"cell_type":"code","source":"# Check if there are duplicates in the datasets\nfor df_name, df in [('train', train_00), ('test', test_00), ('original', orig_00)]:\n    nunb_of_duplicates = df.duplicated().sum()\n    if nunb_of_duplicates != 0:\n        print(f'{df_name} dataset has {nunb_of_duplicates} duplicates.')\n    else:\n        print(f'The {df_name} dataset has no duplicates')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:36.32336Z","iopub.execute_input":"2024-12-10T01:37:36.323749Z","iopub.status.idle":"2024-12-10T01:37:39.053233Z","shell.execute_reply.started":"2024-12-10T01:37:36.323714Z","shell.execute_reply":"2024-12-10T01:37:39.05189Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 3.3. Are there missing values in the datasets? Let's count</p>","metadata":{}},{"cell_type":"code","source":"# Count the missing values in the datasets\nnull_count = pd.DataFrame({'NaN in train': train_00.isna().sum(), \n                           'NaN in test': test_00.isna().sum(), \n                           'NaN in original': orig_00.isna().sum(),\n                           '% NaN in train': train_00.isna().mean()*100, \n                           '% NaN in test': test_00.isna().mean()*100, \n                           '% NaN in original': orig_00.isna().mean()*100\n                          }\n                         ).drop(index=[target]).astype('int')\n# pickup only the features with missing values\nnull_count.sort_values(by='NaN in train', ascending=False).head(11).style.background_gradient(cmap='Oranges')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:39.05458Z","iopub.execute_input":"2024-12-10T01:37:39.054947Z","iopub.status.idle":"2024-12-10T01:37:41.306517Z","shell.execute_reply.started":"2024-12-10T01:37:39.054913Z","shell.execute_reply":"2024-12-10T01:37:41.305301Z"},"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 3.4. Let the Policy Start Date Speak </p> ","metadata":{}},{"cell_type":"code","source":"# Get the year, month and age of the account\ndef get_the_month_year(df):\n    df['month'] = 'month_' + pd.to_datetime(df['Policy Start Date']).dt.month.astype('string')\n    df['year'] = 'year_' + pd.to_datetime(df['Policy Start Date']).dt.year.astype('string')\n    # current_date = datetime.now()\n    # df['account_age'] = df['Policy Start Date'].apply(lambda x: current_date.year - x.year - ((current_date.month, current_date.day) < (x.month, x.day)))\n    df = df.drop(columns = ['Policy Start Date'])\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:41.308001Z","iopub.execute_input":"2024-12-10T01:37:41.308336Z","iopub.status.idle":"2024-12-10T01:37:41.314268Z","shell.execute_reply.started":"2024-12-10T01:37:41.308303Z","shell.execute_reply":"2024-12-10T01:37:41.313246Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_00 = get_the_month_year(train_00)\norig_00 = get_the_month_year(orig_00)\n# train_orig_01 = get_the_month_year(train_orig_00)\ntest_00 = get_the_month_year(test_00)\ntrain_00.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:41.315696Z","iopub.execute_input":"2024-12-10T01:37:41.316047Z","iopub.status.idle":"2024-12-10T01:37:44.378434Z","shell.execute_reply.started":"2024-12-10T01:37:41.316015Z","shell.execute_reply":"2024-12-10T01:37:44.377205Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 3.5. Group features as categorical or numerival </p> ","metadata":{}},{"cell_type":"code","source":"num_features = test_00.select_dtypes('number').columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:44.382973Z","iopub.execute_input":"2024-12-10T01:37:44.383361Z","iopub.status.idle":"2024-12-10T01:37:44.411963Z","shell.execute_reply.started":"2024-12-10T01:37:44.383325Z","shell.execute_reply":"2024-12-10T01:37:44.410805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_features = test_00.select_dtypes(exclude='number').columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:44.413418Z","iopub.execute_input":"2024-12-10T01:37:44.413897Z","iopub.status.idle":"2024-12-10T01:37:44.568305Z","shell.execute_reply.started":"2024-12-10T01:37:44.413848Z","shell.execute_reply":"2024-12-10T01:37:44.566872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# count uniques for each cat_feature\ntrain_00[cat_features].nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:44.570117Z","iopub.execute_input":"2024-12-10T01:37:44.570607Z","iopub.status.idle":"2024-12-10T01:37:45.62313Z","shell.execute_reply.started":"2024-12-10T01:37:44.570566Z","shell.execute_reply":"2024-12-10T01:37:45.622019Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 4. What do the dat say? </p> ","metadata":{}},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 4.1. How does the target distribution in the train and orig sets compare? </p>","metadata":{}},{"cell_type":"code","source":"# choice of colors\ntrain_color, orig_color = 'darkgreen', 'darkorange'\n\n# plot the charts\nplt.figure(figsize=(8, 4))\nplt.subplot(221)\nsns.kdeplot(train_00, x=target, fill=True, color=train_color)\nplt.xlabel('')\nplt.title('Train Dataset', fontsize=12, color=train_color, weight='bold')\nplt.subplot(222)\nsns.kdeplot(orig_00, x=target, fill=True, color=orig_color)\nplt.xlabel('')\nplt.title('Original Dataset', fontsize=12, color=orig_color, weight='bold')\nplt.subplot(223)\nsns.boxenplot(train_00, x=target, color=train_color)\nplt.subplot(224)\nsns.boxenplot(orig_00, x=target, color=orig_color)\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:45.624569Z","iopub.execute_input":"2024-12-10T01:37:45.624947Z","iopub.status.idle":"2024-12-10T01:37:53.098161Z","shell.execute_reply.started":"2024-12-10T01:37:45.624913Z","shell.execute_reply":"2024-12-10T01:37:53.097057Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The distribution of the Premium Amount in train and test sets are almost the same. Both are right skewed, with a high likelywood for outliers in the highest ends. Though this is the target it might be interesting to explore the possibility of handling these outliers.","metadata":{}},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 4.2. Distribution numeric features </p> ","metadata":{}},{"cell_type":"code","source":"def num_cat_distribution(df):\n    fig, axes = plt.subplots(nrows=len(num_features), ncols=3, figsize=(10, 18))\n    for i, feat in enumerate(num_features):\n        # Boxen plot\n        sns.boxenplot(data=df, x=feat, ax=axes[i, 0])\n        axes[i, 0].set_xlabel('')\n        axes[i, 0].set_title(feat, color='maroon', fontsize=10)\n        # violin plot\n        sns.violinplot(data=df, x=feat, ax=axes[i, 1])\n        axes[i, 1].set_xlabel('')\n        axes[i, 1].set_title(feat, color='maroon', fontsize=10)\n        # scatter plot num_feat vs target\n        sns.scatterplot(data=df, y=feat, x=target, ax=axes[i, 2])\n        axes[i, 2].set_title(f'Premium vs {feat}', color='maroon', fontsize=10)\n        axes[i, 2].set_ylabel(f'{feat}', fontsize=9)\n        if i<len(num_features)-1:\n            axes[i, 2].set_xlabel('')\n    \n    plt.tight_layout()\n    plt.show()\n\nnum_cat_distribution(train_00)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:37:53.099657Z","iopub.execute_input":"2024-12-10T01:37:53.100021Z","iopub.status.idle":"2024-12-10T01:38:39.378647Z","shell.execute_reply.started":"2024-12-10T01:37:53.099987Z","shell.execute_reply":"2024-12-10T01:38:39.377506Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 4.3. Distribution target vs cat_features </p> ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20,16))\nfor f, cat_feat in enumerate(cat_features, start=1):\n    plt.subplot(4,3,f)\n    # ax = sns.boxenplot(train_00, x=target, y=cat_feat)\n    sns.violinplot(train_00, x=target, y=cat_feat, fill=True)\n    plt.ylabel(f'{cat_feat}', fontsize=14, weight='bold', color='maroon')\n    if f not in [10, 11, 12]:\n        plt.xlabel('')\n        # plt.ylabel('')\n    else:\n        plt.xlabel('Premium Amount', fontsize=14, weight='bold', color='maroon')\n    #     plt.ylabel(f'{feat}', fontsize=16, weight='bold')\n    # plt.title('{}'.format(cat_feat), fontsize=13, fontdict={'color': 'maroon'}, pad=20)\n    plt.suptitle('Premium distribution by:', fontsize=16, color='maroon', weight='bold')\nplt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:38:39.37999Z","iopub.execute_input":"2024-12-10T01:38:39.380356Z","iopub.status.idle":"2024-12-10T01:39:19.463465Z","shell.execute_reply.started":"2024-12-10T01:38:39.38032Z","shell.execute_reply":"2024-12-10T01:39:19.462281Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As shown by the violinplots above the categorical features have no direct incident on the distribution of the Premium Amount.","metadata":{}},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 4.4. Proportion in cat_features </p> ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16, 12))\nfor f, feat in enumerate(cat_features, start=1):\n    plt.subplot(3, 4, f)\n    pd.Series({' ': 1}).plot.pie(colors=['grey'], radius=0.2, shadow=False)\n    train_00[feat].value_counts().plot.pie(autopct='%.1f%%', radius=1.2, pctdistance=0.44, shadow=False, \n                                         textprops={'color':'black', 'rotation':True, 'weight':'bold', 'size': 8},\n                                         startangle=90 , rotatelabels=True,\n                                         labeldistance=0.7, wedgeprops={'width':0.9}, frame=True)\n    plt.ylabel('')\n    plt.title(f'proportions in {feat}', color='maroon', fontsize=12, weight='bold')\n    plt.tight_layout()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:39:19.465262Z","iopub.execute_input":"2024-12-10T01:39:19.465841Z","iopub.status.idle":"2024-12-10T01:39:24.4286Z","shell.execute_reply.started":"2024-12-10T01:39:19.465786Z","shell.execute_reply":"2024-12-10T01:39:24.42726Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 4.5. How do the features correlate? </p>","metadata":{}},{"cell_type":"code","source":"# Generate a mask for the upper triangle \nmask = np.triu(np.ones_like(train_00[num_features].corr(method='spearman'), dtype=bool))\n# Generate a custom diverging colormap \ncmap = my_scheem[:-2]\n\nsns.heatmap(train_00[num_features].corr(), annot=True, cmap=cmap, mask=mask, fmt='.3f')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:39:24.430065Z","iopub.execute_input":"2024-12-10T01:39:24.430465Z","iopub.status.idle":"2024-12-10T01:39:35.042162Z","shell.execute_reply.started":"2024-12-10T01:39:24.430428Z","shell.execute_reply":"2024-12-10T01:39:35.041064Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"There appears to be no correlations among the features.","metadata":{}},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 5. Let's prep the data </p> ","metadata":{}},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 5.1. Missing values treatment </p> ","metadata":{}},{"cell_type":"code","source":"train_orig_00 = pd.concat([train_00, orig_00], ignore_index=True)\n\ntrain_01 = train_00.copy()\norig_01 = orig_00.copy()\ntest_01 = test_00.copy()\ntrain_orig_01 = train_orig_00.copy()\n\ntrain_01[num_features] = train_00[num_features].fillna(train_00[num_features].median())\norig_01[num_features] = orig_00[num_features].fillna(orig_00[num_features].median())\ntest_01[num_features] = test_00[num_features].fillna(test_00[num_features].median())\ntrain_orig_01[num_features] = train_orig_00[num_features].fillna(train_orig_00[num_features].median())\n\ntrain_orig_01.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:39:35.043595Z","iopub.execute_input":"2024-12-10T01:39:35.043953Z","iopub.status.idle":"2024-12-10T01:39:37.950298Z","shell.execute_reply.started":"2024-12-10T01:39:35.043917Z","shell.execute_reply":"2024-12-10T01:39:37.94913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_orig_01 = pd.concat([train_01, orig_01], ignore_index=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:39:37.951572Z","iopub.execute_input":"2024-12-10T01:39:37.951892Z","iopub.status.idle":"2024-12-10T01:39:38.327063Z","shell.execute_reply.started":"2024-12-10T01:39:37.95186Z","shell.execute_reply":"2024-12-10T01:39:38.326095Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 5.2. Features Engineering </p> ","metadata":{}},{"cell_type":"code","source":"# Get the year, month and age of the account\ndef features_eng(df):\n    df['claims_to_duration'] = df['Previous Claims']/df['Insurance Duration']\n    df['income_to_dependents'] = df['Annual Income']/df['Number of Dependents']\n    df['income_to_age'] = df['Annual Income']/df['Age']\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:39:38.328332Z","iopub.execute_input":"2024-12-10T01:39:38.328661Z","iopub.status.idle":"2024-12-10T01:39:38.33437Z","shell.execute_reply.started":"2024-12-10T01:39:38.328629Z","shell.execute_reply":"2024-12-10T01:39:38.333261Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_01 = features_eng(train_01)\norig_01 = features_eng(orig_01)\ntrain_orig_01 = features_eng(train_orig_01)\ntest_01 = features_eng(test_01)\ntrain_01.head(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:39:38.335556Z","iopub.execute_input":"2024-12-10T01:39:38.335878Z","iopub.status.idle":"2024-12-10T01:39:38.453854Z","shell.execute_reply.started":"2024-12-10T01:39:38.335847Z","shell.execute_reply":"2024-12-10T01:39:38.452611Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 5.3. Build our preprocessing tool </p> ","metadata":{}},{"cell_type":"code","source":"# scaler = MinMaxScaler()\nscaler = PowerTransformer(method='yeo-johnson', standardize=True, copy=True)\n\nfeat_transformer = make_column_transformer(\n    (ce.CatBoostEncoder(), cat_features),\n    (scaler, num_features)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:39:38.45516Z","iopub.execute_input":"2024-12-10T01:39:38.455538Z","iopub.status.idle":"2024-12-10T01:39:38.461143Z","shell.execute_reply.started":"2024-12-10T01:39:38.455501Z","shell.execute_reply":"2024-12-10T01:39:38.45991Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 5.4. Split data and target </p> ","metadata":{}},{"cell_type":"code","source":"# data vs target\nX = train_01.copy()\ny = X.pop(target)\n\n# get the log of the target\ny_log = np.log(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:39:38.463093Z","iopub.execute_input":"2024-12-10T01:39:38.463524Z","iopub.status.idle":"2024-12-10T01:39:38.840912Z","shell.execute_reply.started":"2024-12-10T01:39:38.46348Z","shell.execute_reply":"2024-12-10T01:39:38.839914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 6. Modeling </p> ","metadata":{}},{"cell_type":"code","source":"# defnine the number of splits\nn_splits=2\n# define the kfold spliter\nkfold = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:42:15.309519Z","iopub.execute_input":"2024-12-10T01:42:15.309909Z","iopub.status.idle":"2024-12-10T01:42:15.315411Z","shell.execute_reply.started":"2024-12-10T01:42:15.309874Z","shell.execute_reply":"2024-12-10T01:42:15.314047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"color_1 = 'lightgreen'\ncolor_2 = 'darkgreen'\ncolor_3 = 'green'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:42:15.317201Z","iopub.execute_input":"2024-12-10T01:42:15.317569Z","iopub.status.idle":"2024-12-10T01:42:15.333538Z","shell.execute_reply.started":"2024-12-10T01:42:15.317534Z","shell.execute_reply":"2024-12-10T01:42:15.332078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def oof_regression_modeling(model):\n    for f, (tr_ind, ts_ind) in enumerate(kfold.split(X, y), start=1):\n        # instanciate the regressor\n        reg = model\n        # build the regression pipeline\n        reg_pipe = make_pipeline(feat_transformer, reg)\n        # get the train and test sets\n        X_tr, X_ts = X.iloc[tr_ind], X.iloc[ts_ind]\n        y_tr_log, y_ts_log = y_log.iloc[tr_ind], y_log.iloc[ts_ind]\n        # fit the model\n        reg_pipe.fit(X_tr, y_tr_log)\n        # prediction on the test set\n        pred_log = reg_pipe.predict(X_ts)\n        # convert the predictions from log\n        y_ts,  pred = np.exp(y_ts_log), np.exp(pred_log)\n        # score both log and non log predictions\n        score_log = rmsle_scorer(y_ts_log, pred_log)\n        score = rmsle_scorer(y_ts, pred)\n        # view the performance of the model\n        plt.figure(figsize=(8, 3))\n        plt.subplot(1, 2, 1)\n        ax = sns.histplot(y_ts, bins=25)\n        plt.title('log_score: {:.6} || score: {:.6}'.format(score_log, score), fontsize=11)\n        pd.Series(pred).plot.hist(color=color_2, bins=25, ax=ax)\n        plt.subplot(1, 2, 2)\n        residue = y_ts - pred\n        sns.histplot(residue, fill=True, alpha=0.8, color=color_3, bins=25)\n        plt.title('Histogram of residues: y_true - y_hat', fontsize=11)\n        plt.suptitle(f'Performances in Fold_{f}', fontsize=14, color='green', weight='bold')\n        \n        plt.tight_layout()\n        plt.show()\n\n        if f==n_splits:\n            print('\\n<<<<<<<<<<<<   Performance of the model with yellowbrick   >>>>>>>>>>>>')\n            # Visualize performance with Yellowbrick\n            plt.subplots(figsize=(9, 4))\n            visualizer = ResidualsPlot(reg_pipe, hist=False, qqplot=True)\n            visualizer.fit(X_tr, y_tr_log)  # Fit the training data to the visualizer\n            visualizer.score(X_ts, y_ts_log)  # Evaluate the model on the test data\n            visualizer.show()                 # Finalize and render the figure\n            \n            # Calculate the residuals in the prediction (prediction - true value)\n            resids = y_ts_log - reg_pipe.predict(X_ts)\n            \n            # Plot the distribution of the residuals (kde)\n            fig, ax = plt.subplots(figsize=(9, 3))\n            sns.histplot(resids, fill=True, color='maroon', bins=50, ax=ax)\n            plt.title('Distribution of the residuals in the predictions')\n            plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:42:15.335614Z","iopub.execute_input":"2024-12-10T01:42:15.335968Z","iopub.status.idle":"2024-12-10T01:42:15.35173Z","shell.execute_reply.started":"2024-12-10T01:42:15.335928Z","shell.execute_reply":"2024-12-10T01:42:15.35062Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:110%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 6.1. Our first model: LGBMRegressor </p> \n\n## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 6.1.1. Optuna Tuning of the model </p>","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n    params = {\n        'n_estimators': trial.suggest_int('n_estimators', 100, 1000, 10),\n        'max_depth': trial.suggest_int('max_depth', 3, 10),\n        'num_leaves': trial.suggest_int('num_leaves', 2, 256),\n        'feature_fraction': trial.suggest_float('feature_fraction', 0.4, 1.0),\n        'bagging_fraction': trial.suggest_float('bagging_fraction', 0.4, 1.0),\n        'bagging_freq': trial.suggest_int('bagging_freq', 1, 7),\n        # 'min_child_samples': trial.suggest_int('min_child_samples', 5, 100),\n    }\n\n    model = LGBMRegressor(**params, verbose=-1)\n    model_pipe = make_pipeline(feat_transformer, model)\n\n    x_tr, x_va, y_tr, y_va = train_test_split(X, y_log, test_size=0.2, random_state=32)\n    model_pipe.fit(x_tr, y_tr)\n    y_pred_log = model_pipe.predict(x_va)\n    score = rmsle_scorer(y_va, y_pred_log)\n    return score\n\ndef get_optuna_best_params(n_trials=1):\n    if n_trials >1:\n        study = optuna.create_study(direction='minimize',sampler=TPESampler(seed=33))\n        study.optimize(lambda trial: objective(trial), n_trials=n_trials,show_progress_bar=True,n_jobs=-1, timeout=12000)\n        best_params = study.best_params\n    else:\n        print('No need to run optuna, we will use the parameters obtained earlier')\n        best_params = {'n_estimators': 160, \n                       'max_depth': 9,\n                       'num_leaves': 95, \n                       'feature_fraction': 0.9927320061735965, \n                       'bagging_fraction': 0.8881137072192649,\n                       'bagging_freq': 2}\n        \n    print('best params: {}'.format(best_params))\n    return best_params","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:42:15.352907Z","iopub.execute_input":"2024-12-10T01:42:15.353247Z","iopub.status.idle":"2024-12-10T01:42:15.368331Z","shell.execute_reply.started":"2024-12-10T01:42:15.353208Z","shell.execute_reply":"2024-12-10T01:42:15.367245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params = get_optuna_best_params(n_trials=50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:42:15.371253Z","iopub.execute_input":"2024-12-10T01:42:15.371744Z","iopub.status.idle":"2024-12-10T01:42:15.384771Z","shell.execute_reply.started":"2024-12-10T01:42:15.371694Z","shell.execute_reply":"2024-12-10T01:42:15.383793Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:100%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 6.1.2. Model exploitation </p>","metadata":{}},{"cell_type":"code","source":"oof_regression_modeling(LGBMRegressor(**best_params, verbose=-1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:42:15.385989Z","iopub.execute_input":"2024-12-10T01:42:15.386353Z","iopub.status.idle":"2024-12-10T01:44:28.9641Z","shell.execute_reply.started":"2024-12-10T01:42:15.386318Z","shell.execute_reply":"2024-12-10T01:44:28.963055Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:110%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 6.2. Our CatBoostRegressor </p> ","metadata":{}},{"cell_type":"code","source":"# CatBoostRegresson needs to to wrapped\n# Create a wrapper class\nclass CatBoostWrapper(BaseEstimator, RegressorMixin):\n    def __init__(self, **best_cat_params):\n        self.model = CatBoostRegressor(**best_cat_params)\n    \n    def fit(self, X, y):\n        self.model.fit(X, y)\n        return self\n    \n    def predict(self, X):\n        return self.model.predict(X)\n\n# define the wrapped catboostrgressor\nCatBoosstWrapped = CatBoostWrapper(eval_metric='R2', eval_fraction=0.2, verbose=250, early_stopping_rounds=500)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:44:28.965335Z","iopub.execute_input":"2024-12-10T01:44:28.96565Z","iopub.status.idle":"2024-12-10T01:44:28.974478Z","shell.execute_reply.started":"2024-12-10T01:44:28.965611Z","shell.execute_reply":"2024-12-10T01:44:28.973311Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"oof_regression_modeling(CatBoosstWrapped)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:44:28.975897Z","iopub.execute_input":"2024-12-10T01:44:28.976375Z","iopub.status.idle":"2024-12-10T01:46:37.487498Z","shell.execute_reply.started":"2024-12-10T01:44:28.976325Z","shell.execute_reply":"2024-12-10T01:46:37.486227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:110%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 6.3. Our XGBRegressor </p> ","metadata":{}},{"cell_type":"code","source":"oof_regression_modeling(XGBRegressor())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:46:37.488702Z","iopub.execute_input":"2024-12-10T01:46:37.489016Z","iopub.status.idle":"2024-12-10T01:48:05.921426Z","shell.execute_reply.started":"2024-12-10T01:46:37.488985Z","shell.execute_reply":"2024-12-10T01:48:05.919993Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:110%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 6.4. Our HistGradientBoostingRegressor </p> ","metadata":{}},{"cell_type":"code","source":"oof_regression_modeling(HistGradientBoostingRegressor())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:48:05.922891Z","iopub.execute_input":"2024-12-10T01:48:05.923353Z","iopub.status.idle":"2024-12-10T01:49:52.832329Z","shell.execute_reply.started":"2024-12-10T01:48:05.923313Z","shell.execute_reply":"2024-12-10T01:49:52.830878Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:120%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 7. Prediction and Submission </p> ","metadata":{}},{"cell_type":"markdown","source":"## <p style= \"font-family: Calibri; font-weight:bold; letter-spacing: 0px; color:orange; border-radius:5px; font-size:110%; text-align:left;padding:3.0px; background: darkgreen; border-bottom: 4px solid #2c3e50; border-top: 4px solid k\" > 7.1. Select the model to use for submission </p>","metadata":{}},{"cell_type":"code","source":"# Choice of model\ndef choose_model(use = 'hist'):\n    # define the model\n    if use == 'lgb':\n        model = LGBMRegressor(**best_params, verbose=-1)\n    elif use == 'rfr':\n        model = RandomForestRegressor()\n    elif use == 'hist':\n        model = HistGradientBoostingRegressor()\n    elif use == 'cat':\n        model = CatBoosstWrapped\n    elif use == 'xgb':\n        model = XGBRegressor()\n    return model\n\n# choose the regressor\nmodel = choose_model(use='lgb')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:49:52.835217Z","iopub.execute_input":"2024-12-10T01:49:52.835591Z","iopub.status.idle":"2024-12-10T01:49:52.842957Z","shell.execute_reply.started":"2024-12-10T01:49:52.835555Z","shell.execute_reply":"2024-12-10T01:49:52.841638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# create a copy of the submission data\nsub_df = submission_00.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:49:52.844258Z","iopub.execute_input":"2024-12-10T01:49:52.844607Z","iopub.status.idle":"2024-12-10T01:49:52.858893Z","shell.execute_reply.started":"2024-12-10T01:49:52.844573Z","shell.execute_reply":"2024-12-10T01:49:52.857749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define the regressor_pipeline\nmodel_pipe = make_pipeline(feat_transformer, model)\n\n# fit the model\nmodel_pipe.fit(X, y_log)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T01:49:52.86079Z","iopub.execute_input":"2024-12-10T01:49:52.861271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# predict the log of the test target\ntest_pred_log = model_pipe.predict(test_01)\n\n# convert the preds from log\ntest_pred = np.expm1(test_pred_log)\n\n# build the sub_df\nsub_df['Premium Amount'] = test_pred\n\n# preview of the sub_df\nsub_df.head(10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# prepare the submission file\nsub_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}