{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport seaborn as sns\nfrom scipy import stats\nimport statsmodels.api as sm\nimport matplotlib.pyplot as plt\n\nfrom IPython.display import display, HTML\n\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor, Pool\n\nfrom sklearn.model_selection import RepeatedKFold, train_test_split\n\ndef display_html(size=3, content=\"content\"):\n    display(HTML(f\"<h{size}>{content}</h{size}>\"))\n\nimport warnings \nwarnings.filterwarnings('ignore')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T15:20:15.65015Z","iopub.execute_input":"2024-12-05T15:20:15.650481Z","iopub.status.idle":"2024-12-05T15:20:20.994355Z","shell.execute_reply.started":"2024-12-05T15:20:15.650447Z","shell.execute_reply":"2024-12-05T15:20:20.99314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T15:55:03.890312Z","iopub.execute_input":"2024-12-05T15:55:03.891132Z","iopub.status.idle":"2024-12-05T15:55:09.016294Z","shell.execute_reply.started":"2024-12-05T15:55:03.891094Z","shell.execute_reply":"2024-12-05T15:55:09.01531Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Helper Function ","metadata":{}},{"cell_type":"code","source":"def calculate_stats(data, columns):\n    if isinstance(columns, str):\n        columns = [columns]\n\n    stats = [] \n    for col in columns:\n        if data[col].dtypes=='object' or data[col].dtypes=='category':\n            counts = data[col].value_counts(dropna=False, sort=False)\n            percentage = data[col].value_counts(dropna=False, sort=False, normalize=True)*100\n            formate = counts.astype(str) + \" (\" + percentage.round(2).astype(str) + \"%)\"\n\n            stats_col = pd.DataFrame({'count %': formate})\n            stats.append(stats_col)\n        else:\n            stats_col = data[col].describe().to_frame().T\n            missing = data[col].isnull().sum()\n            percentage = missing/len(data[col])\n            stats_col['missing'] = missing.astype(str) + \" (\"+ percentage.round(2).astype(str) + \"%)\"\n            stats_col.index.name=col\n            stats.append(stats_col)\n\n    return pd.concat(stats,axis=0)\n\ndef num_univar_plots(data, col, figsize=(15, 4), bins=20):\n    display_html(3, f'Univariate Analysis of {col}')\n\n    fig, ax = plt.subplots(1,3, figsize=figsize)\n\n    # histogram\n    sns.histplot(data=data, x=col, ax=ax[0], kde=True, bins=bins, color='#1973bd')\n    sns.rugplot(data=data, color='black', ax=ax[0], x=col)\n    ax[0].set_title('Histogram')\n\n    # box plot \n    sns.boxplot(data=data, x=col, color='orange', ax=ax[1])\n    ax[1].set_title('Boxplot')\n\n    ## qq plot\n    sm.qqplot(data[col].dropna(), line='45', fit=True, ax=ax[2])\n    ax[2].set_title('QQ Plot')\n\n    plt.tight_layout()\n    plt.show()\n\ndef num_cat_bivar_plots(data, num_col, cat_col, estimator='mean', figsize=(15,4)):\n    agg_data = data.groupby(cat_col).agg(estimator, numeric_only=True).loc[:, num_col].dropna().sort_values().reindex()\n\n    fig, ax = plt.subplots(1,3, figsize=figsize)\n    sns.barplot(x=agg_data.index, y=agg_data.values, palette='Set3', ax=ax[0],edgecolor='black')\n    ax[0].set(title='Bar Plot', xlabel=cat_col, ylabel=num_col)\n\n    sns.boxplot(data, x=cat_col, y=num_col, palette='Set3', order=agg_data.index, ax=ax[1])\n    ax[1].set(title='Box Plot', xlabel=cat_col, ylabel=\"\")\n\n    sns.violinplot(data, x=cat_col, y=num_col, palette='Set3', order=agg_data.index, ax=ax[2])\n    ax[2].set(title='Violin Plot', ylabel=num_col, xlabel=cat_col)\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:12.748532Z","iopub.execute_input":"2024-12-05T13:08:12.748901Z","iopub.status.idle":"2024-12-05T13:08:12.761154Z","shell.execute_reply.started":"2024-12-05T13:08:12.748863Z","shell.execute_reply":"2024-12-05T13:08:12.760322Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Exploration","metadata":{}},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:12.762891Z","iopub.execute_input":"2024-12-05T13:08:12.763163Z","iopub.status.idle":"2024-12-05T13:08:12.809155Z","shell.execute_reply.started":"2024-12-05T13:08:12.763138Z","shell.execute_reply":"2024-12-05T13:08:12.808455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drope the id columns \ntrain_data = train_data.drop(columns=['id'])\ntest_data = test_data.drop(columns=['id'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T15:55:09.319597Z","iopub.execute_input":"2024-12-05T15:55:09.320212Z","iopub.status.idle":"2024-12-05T15:55:09.611016Z","shell.execute_reply.started":"2024-12-05T15:55:09.320179Z","shell.execute_reply":"2024-12-05T15:55:09.610117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Shape of the training data : ', train_data.shape)\nprint('Shape of the test data : ', test_data.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T15:50:37.115084Z","iopub.execute_input":"2024-12-05T15:50:37.115422Z","iopub.status.idle":"2024-12-05T15:50:37.12029Z","shell.execute_reply.started":"2024-12-05T15:50:37.115393Z","shell.execute_reply":"2024-12-05T15:50:37.119241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:13.075967Z","iopub.execute_input":"2024-12-05T13:08:13.076233Z","iopub.status.idle":"2024-12-05T13:08:13.645907Z","shell.execute_reply.started":"2024-12-05T13:08:13.076209Z","shell.execute_reply":"2024-12-05T13:08:13.645055Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- The data type of `Policy Start Date` should be datetime instead of object","metadata":{}},{"cell_type":"code","source":"# duplicate data present or not \ntrain_data.duplicated().sum()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:13.647046Z","iopub.execute_input":"2024-12-05T13:08:13.6474Z","iopub.status.idle":"2024-12-05T13:08:15.029424Z","shell.execute_reply.started":"2024-12-05T13:08:13.64737Z","shell.execute_reply":"2024-12-05T13:08:15.028557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check missing value \ntrain_data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:15.030353Z","iopub.execute_input":"2024-12-05T13:08:15.03058Z","iopub.status.idle":"2024-12-05T13:08:15.572321Z","shell.execute_reply.started":"2024-12-05T13:08:15.030558Z","shell.execute_reply":"2024-12-05T13:08:15.571369Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"## Age ","metadata":{}},{"cell_type":"code","source":"train = train_data.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:15.57694Z","iopub.execute_input":"2024-12-05T13:08:15.577281Z","iopub.status.idle":"2024-12-05T13:08:15.734248Z","shell.execute_reply.started":"2024-12-05T13:08:15.577241Z","shell.execute_reply":"2024-12-05T13:08:15.733216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:15.735478Z","iopub.execute_input":"2024-12-05T13:08:15.735786Z","iopub.status.idle":"2024-12-05T13:08:15.808152Z","shell.execute_reply.started":"2024-12-05T13:08:15.735758Z","shell.execute_reply":"2024-12-05T13:08:15.807343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Age')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:15.809383Z","iopub.execute_input":"2024-12-05T13:08:15.810126Z","iopub.status.idle":"2024-12-05T13:08:31.386713Z","shell.execute_reply.started":"2024-12-05T13:08:15.810081Z","shell.execute_reply":"2024-12-05T13:08:31.385801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Age Group'] = pd.cut(train['Age'].dropna(), bins=[17,26,48,64],\n                           labels=['Entry-level Age', 'Experience Age', 'Pre-retirement Age'])\ntrain['Premium Amount Group'] = pd.cut(train['Premium Amount'], bins=[0, 1000, 2000, 3000, 4000, 5000], \n                                       labels=['0-1k', '1k-2k', '2k-3k', '3k-4k', '4k-5k'])\n\ncalculate_stats(train, 'Age Group')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:31.387804Z","iopub.execute_input":"2024-12-05T13:08:31.388069Z","iopub.status.idle":"2024-12-05T13:08:31.513167Z","shell.execute_reply.started":"2024-12-05T13:08:31.388043Z","shell.execute_reply":"2024-12-05T13:08:31.512355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"heatmap_data = pd.pivot_table(\n    train, \n    values='Premium Amount', \n    index='Premium Amount Group', \n    columns='Age Group', \n    aggfunc='count', \n    fill_value=0\n)\nfig, ax = plt.subplots(1,2, figsize=(14,5))\nsns.heatmap(heatmap_data, cmap='Blues', ax=ax[0], cbar=False)\nax[0].set_title(\"Heatmap of Premium Amounts Across Age Groups\")\nax[0].set_xlabel(\"Age Group\")\nax[0].set_ylabel(\"Premium Amount\")\n\nsns.boxplot(x='Age Group', y='Premium Amount', data=train, ax=ax[1])\nplt.title('Premium Amount v/s Age Group')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:31.514335Z","iopub.execute_input":"2024-12-05T13:08:31.514799Z","iopub.status.idle":"2024-12-05T13:08:32.12799Z","shell.execute_reply.started":"2024-12-05T13:08:31.514758Z","shell.execute_reply":"2024-12-05T13:08:32.127088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train[train['Premium Amount']>3000])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:32.129027Z","iopub.execute_input":"2024-12-05T13:08:32.129293Z","iopub.status.idle":"2024-12-05T13:08:32.159848Z","shell.execute_reply.started":"2024-12-05T13:08:32.129266Z","shell.execute_reply":"2024-12-05T13:08:32.159097Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- average premium amount for all age group are equal. and every group have nearly same number of potential outlier . only 49347 (4%) people have premium amount `>3000`\n- premium amount for `Experience Age` and `Pre-retirement Age` groups are concentrated in range of 0-1k. with the maximum premium amount also belonging to this category.**","metadata":{"execution":{"iopub.status.busy":"2024-12-01T05:56:09.869085Z","iopub.execute_input":"2024-12-01T05:56:09.8695Z","iopub.status.idle":"2024-12-01T05:56:09.87724Z","shell.execute_reply.started":"2024-12-01T05:56:09.869463Z","shell.execute_reply":"2024-12-01T05:56:09.875764Z"}}},{"cell_type":"markdown","source":"## Gender","metadata":{}},{"cell_type":"code","source":"calculate_stats(train, 'Gender')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:32.16072Z","iopub.execute_input":"2024-12-05T13:08:32.160941Z","iopub.status.idle":"2024-12-05T13:08:32.24519Z","shell.execute_reply.started":"2024-12-05T13:08:32.160919Z","shell.execute_reply":"2024-12-05T13:08:32.244365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 2, figsize=(15, 4))\nsns.histplot(data=train, hue='Gender', x='Premium Amount', palette='Set3', multiple='stack',bins=20, ax=ax[0])\nsns.countplot(data=train, hue='Gender', x='Premium Amount Group', palette='Set3',ax=ax[1])\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:32.246119Z","iopub.execute_input":"2024-12-05T13:08:32.246362Z","iopub.status.idle":"2024-12-05T13:08:34.508919Z","shell.execute_reply.started":"2024-12-05T13:08:32.246338Z","shell.execute_reply":"2024-12-05T13:08:34.508046Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- There is an equal number of men and women, with most people (abount 80%) paying premiums between 0 and 2,000 rupees. However, a significat portion (around 60%) pay even less, between 0 and 1,000 repees.","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:34.509974Z","iopub.execute_input":"2024-12-05T13:08:34.510245Z","iopub.status.idle":"2024-12-05T13:08:34.529485Z","shell.execute_reply.started":"2024-12-05T13:08:34.510219Z","shell.execute_reply":"2024-12-05T13:08:34.528691Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Annual Income","metadata":{"execution":{"iopub.status.busy":"2024-12-01T06:53:25.549004Z","iopub.execute_input":"2024-12-01T06:53:25.549647Z","iopub.status.idle":"2024-12-01T06:53:25.637989Z","shell.execute_reply.started":"2024-12-01T06:53:25.549604Z","shell.execute_reply":"2024-12-01T06:53:25.636897Z"}}},{"cell_type":"code","source":"calculate_stats(train, 'Annual Income')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:34.530509Z","iopub.execute_input":"2024-12-05T13:08:34.530813Z","iopub.status.idle":"2024-12-05T13:08:34.606609Z","shell.execute_reply.started":"2024-12-05T13:08:34.530787Z","shell.execute_reply":"2024-12-05T13:08:34.605803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Annual Income')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:34.607569Z","iopub.execute_input":"2024-12-05T13:08:34.607845Z","iopub.status.idle":"2024-12-05T13:08:50.459474Z","shell.execute_reply.started":"2024-12-05T13:08:34.607818Z","shell.execute_reply":"2024-12-05T13:08:50.458584Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train[train['Annual Income']>100000])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:50.460579Z","iopub.execute_input":"2024-12-05T13:08:50.460885Z","iopub.status.idle":"2024-12-05T13:08:50.494881Z","shell.execute_reply.started":"2024-12-05T13:08:50.460859Z","shell.execute_reply":"2024-12-05T13:08:50.494046Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Martial Status","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:50.495965Z","iopub.execute_input":"2024-12-05T13:08:50.496234Z","iopub.status.idle":"2024-12-05T13:08:50.517621Z","shell.execute_reply.started":"2024-12-05T13:08:50.496208Z","shell.execute_reply":"2024-12-05T13:08:50.516681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'Marital Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:50.518925Z","iopub.execute_input":"2024-12-05T13:08:50.519197Z","iopub.status.idle":"2024-12-05T13:08:50.627417Z","shell.execute_reply.started":"2024-12-05T13:08:50.519173Z","shell.execute_reply":"2024-12-05T13:08:50.626585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cat_bivar_plots(train, 'Premium Amount', 'Marital Status' )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:50.628601Z","iopub.execute_input":"2024-12-05T13:08:50.628975Z","iopub.status.idle":"2024-12-05T13:08:54.179349Z","shell.execute_reply.started":"2024-12-05T13:08:50.628944Z","shell.execute_reply":"2024-12-05T13:08:54.178443Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- Distribution of `Premium Amount` across different marital status is same. have outliers when premium amount is greater than 3k. majority of data have premium amount less than or equal to 1500","metadata":{"execution":{"iopub.status.busy":"2024-12-01T09:21:33.929381Z","iopub.execute_input":"2024-12-01T09:21:33.929797Z","iopub.status.idle":"2024-12-01T09:21:33.942224Z","shell.execute_reply.started":"2024-12-01T09:21:33.929759Z","shell.execute_reply":"2024-12-01T09:21:33.940876Z"}}},{"cell_type":"markdown","source":"### Number of Dependents","metadata":{}},{"cell_type":"code","source":"num_cat_bivar_plots(train, 'Premium Amount', 'Number of Dependents')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:54.180411Z","iopub.execute_input":"2024-12-05T13:08:54.180691Z","iopub.status.idle":"2024-12-05T13:08:57.08048Z","shell.execute_reply.started":"2024-12-05T13:08:54.180664Z","shell.execute_reply":"2024-12-05T13:08:57.079598Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Occupation ","metadata":{}},{"cell_type":"code","source":"num_cat_bivar_plots(train, 'Premium Amount', 'Occupation')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:57.081683Z","iopub.execute_input":"2024-12-05T13:08:57.082041Z","iopub.status.idle":"2024-12-05T13:08:59.824929Z","shell.execute_reply.started":"2024-12-05T13:08:57.082003Z","shell.execute_reply":"2024-12-05T13:08:59.823796Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Location","metadata":{}},{"cell_type":"code","source":"num_cat_bivar_plots(train, 'Premium Amount', 'Location')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:08:59.826121Z","iopub.execute_input":"2024-12-05T13:08:59.826598Z","iopub.status.idle":"2024-12-05T13:09:03.267466Z","shell.execute_reply.started":"2024-12-05T13:08:59.826557Z","shell.execute_reply":"2024-12-05T13:09:03.26656Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Policy Type","metadata":{}},{"cell_type":"code","source":"calculate_stats(train, 'Policy Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:03.27317Z","iopub.execute_input":"2024-12-05T13:09:03.273517Z","iopub.status.idle":"2024-12-05T13:09:03.359072Z","shell.execute_reply.started":"2024-12-05T13:09:03.27349Z","shell.execute_reply":"2024-12-05T13:09:03.358062Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cat_bivar_plots(train, 'Premium Amount', 'Policy Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:03.360106Z","iopub.execute_input":"2024-12-05T13:09:03.360382Z","iopub.status.idle":"2024-12-05T13:09:06.710152Z","shell.execute_reply.started":"2024-12-05T13:09:03.360357Z","shell.execute_reply":"2024-12-05T13:09:06.709317Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Health Score","metadata":{}},{"cell_type":"code","source":"calculate_stats(train, 'Health Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:06.711614Z","iopub.execute_input":"2024-12-05T13:09:06.711915Z","iopub.status.idle":"2024-12-05T13:09:06.780793Z","shell.execute_reply.started":"2024-12-05T13:09:06.711887Z","shell.execute_reply":"2024-12-05T13:09:06.779884Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Health Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:06.781724Z","iopub.execute_input":"2024-12-05T13:09:06.781993Z","iopub.status.idle":"2024-12-05T13:09:21.83567Z","shell.execute_reply.started":"2024-12-05T13:09:06.781968Z","shell.execute_reply":"2024-12-05T13:09:21.834755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cat_bivar_plots(train, 'Health Score', 'Age Group')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:21.836854Z","iopub.execute_input":"2024-12-05T13:09:21.837757Z","iopub.status.idle":"2024-12-05T13:09:24.347417Z","shell.execute_reply.started":"2024-12-05T13:09:21.837727Z","shell.execute_reply":"2024-12-05T13:09:24.346491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Previous Claims'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:24.348521Z","iopub.execute_input":"2024-12-05T13:09:24.348819Z","iopub.status.idle":"2024-12-05T13:09:24.372899Z","shell.execute_reply.started":"2024-12-05T13:09:24.348791Z","shell.execute_reply":"2024-12-05T13:09:24.372033Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"- We need to drop some unique value like 8.0 and 9.0 ","metadata":{}},{"cell_type":"code","source":"num_cat_bivar_plots(train, 'Premium Amount', 'Previous Claims')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:24.373968Z","iopub.execute_input":"2024-12-05T13:09:24.37447Z","iopub.status.idle":"2024-12-05T13:09:27.152169Z","shell.execute_reply.started":"2024-12-05T13:09:24.374441Z","shell.execute_reply":"2024-12-05T13:09:27.151294Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Credit Score ","metadata":{}},{"cell_type":"code","source":"calculate_stats(train, 'Credit Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:27.153162Z","iopub.execute_input":"2024-12-05T13:09:27.153442Z","iopub.status.idle":"2024-12-05T13:09:27.227427Z","shell.execute_reply.started":"2024-12-05T13:09:27.153414Z","shell.execute_reply":"2024-12-05T13:09:27.226554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Credit Score')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:27.228398Z","iopub.execute_input":"2024-12-05T13:09:27.22873Z","iopub.status.idle":"2024-12-05T13:09:41.632959Z","shell.execute_reply.started":"2024-12-05T13:09:27.228691Z","shell.execute_reply":"2024-12-05T13:09:41.6321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cat_bivar_plots(train, 'Credit Score', 'Age Group')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:41.634077Z","iopub.execute_input":"2024-12-05T13:09:41.634351Z","iopub.status.idle":"2024-12-05T13:09:43.965446Z","shell.execute_reply.started":"2024-12-05T13:09:41.634323Z","shell.execute_reply":"2024-12-05T13:09:43.964646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'Insurance Duration')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:43.966569Z","iopub.execute_input":"2024-12-05T13:09:43.96698Z","iopub.status.idle":"2024-12-05T13:09:44.033193Z","shell.execute_reply.started":"2024-12-05T13:09:43.966923Z","shell.execute_reply":"2024-12-05T13:09:44.032359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Insurance Duration'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.034193Z","iopub.execute_input":"2024-12-05T13:09:44.034504Z","iopub.status.idle":"2024-12-05T13:09:44.052673Z","shell.execute_reply.started":"2024-12-05T13:09:44.034475Z","shell.execute_reply":"2024-12-05T13:09:44.051694Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Policy Start Date","metadata":{}},{"cell_type":"code","source":"# train['Policy Start Date'].dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.053713Z","iopub.execute_input":"2024-12-05T13:09:44.053971Z","iopub.status.idle":"2024-12-05T13:09:44.062324Z","shell.execute_reply.started":"2024-12-05T13:09:44.053947Z","shell.execute_reply":"2024-12-05T13:09:44.061482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'], dayfirst=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.063282Z","iopub.execute_input":"2024-12-05T13:09:44.063537Z","iopub.status.idle":"2024-12-05T13:09:44.406144Z","shell.execute_reply.started":"2024-12-05T13:09:44.063513Z","shell.execute_reply":"2024-12-05T13:09:44.405148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train['Policy Start Date']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.407411Z","iopub.execute_input":"2024-12-05T13:09:44.407923Z","iopub.status.idle":"2024-12-05T13:09:44.417977Z","shell.execute_reply.started":"2024-12-05T13:09:44.407883Z","shell.execute_reply":"2024-12-05T13:09:44.417247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('yes')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.418892Z","iopub.execute_input":"2024-12-05T13:09:44.419216Z","iopub.status.idle":"2024-12-05T13:09:44.431131Z","shell.execute_reply.started":"2024-12-05T13:09:44.419178Z","shell.execute_reply":"2024-12-05T13:09:44.430333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.432005Z","iopub.execute_input":"2024-12-05T13:09:44.432314Z","iopub.status.idle":"2024-12-05T13:09:44.456588Z","shell.execute_reply.started":"2024-12-05T13:09:44.432289Z","shell.execute_reply":"2024-12-05T13:09:44.455801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'Customer Feedback')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.457535Z","iopub.execute_input":"2024-12-05T13:09:44.457931Z","iopub.status.idle":"2024-12-05T13:09:44.546072Z","shell.execute_reply.started":"2024-12-05T13:09:44.457891Z","shell.execute_reply":"2024-12-05T13:09:44.545286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'Smoking Status')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.547134Z","iopub.execute_input":"2024-12-05T13:09:44.547508Z","iopub.status.idle":"2024-12-05T13:09:44.633285Z","shell.execute_reply.started":"2024-12-05T13:09:44.54747Z","shell.execute_reply":"2024-12-05T13:09:44.632351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'Property Type')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.63434Z","iopub.execute_input":"2024-12-05T13:09:44.634646Z","iopub.status.idle":"2024-12-05T13:09:44.725157Z","shell.execute_reply.started":"2024-12-05T13:09:44.634583Z","shell.execute_reply":"2024-12-05T13:09:44.724239Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"calculate_stats(train, 'Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.72641Z","iopub.execute_input":"2024-12-05T13:09:44.726755Z","iopub.status.idle":"2024-12-05T13:09:44.788225Z","shell.execute_reply.started":"2024-12-05T13:09:44.726727Z","shell.execute_reply":"2024-12-05T13:09:44.78733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_univar_plots(train, 'Premium Amount')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:09:44.789087Z","iopub.execute_input":"2024-12-05T13:09:44.789316Z","iopub.status.idle":"2024-12-05T13:10:01.074255Z","shell.execute_reply.started":"2024-12-05T13:09:44.789294Z","shell.execute_reply":"2024-12-05T13:10:01.073436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:13:00.446126Z","iopub.execute_input":"2024-12-05T13:13:00.446831Z","iopub.status.idle":"2024-12-05T13:13:00.46941Z","shell.execute_reply.started":"2024-12-05T13:13:00.446796Z","shell.execute_reply":"2024-12-05T13:13:00.468572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['Gender'] = train_data['Gender'].map({'Female': 0, 'Male':1})\ntrain_data['Smoking Status'] = train_data['Smoking Status'].map({'No':0, 'Yes':1})\ntest_data['Gender'] = test_data['Gender'].map({'Female': 0, 'Male':1})\ntest_data['Smoking Status'] = test_data['Smoking Status'].map({'No':0, 'Yes':1})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:12:00.225845Z","iopub.execute_input":"2024-12-05T13:12:00.226663Z","iopub.status.idle":"2024-12-05T13:12:00.384558Z","shell.execute_reply.started":"2024-12-05T13:12:00.226614Z","shell.execute_reply":"2024-12-05T13:12:00.383883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = train_data[~train_data['Previous Claims'].isin([7.0, 8.0, 9.0])]\n# test_data = test_data[~test_data['Previous Claims'].isin([7.0, 8.0, 9.0])]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:01.321178Z","iopub.execute_input":"2024-12-05T13:10:01.321426Z","iopub.status.idle":"2024-12-05T13:10:01.582527Z","shell.execute_reply.started":"2024-12-05T13:10:01.321403Z","shell.execute_reply":"2024-12-05T13:10:01.58172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols = train_data.select_dtypes('object').columns.to_list()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:01.583397Z","iopub.execute_input":"2024-12-05T13:10:01.583759Z","iopub.status.idle":"2024-12-05T13:10:02.044218Z","shell.execute_reply.started":"2024-12-05T13:10:01.583626Z","shell.execute_reply":"2024-12-05T13:10:02.043474Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#   ------------------------------ Version 1 --------------------------------","metadata":{}},{"cell_type":"code","source":"cat_cols=['Marital Status',\n 'Education Level',\n 'Occupation',\n 'Location',\n 'Policy Type',\n 'Customer Feedback',\n 'Exercise Frequency',\n 'Property Type']\n\nfor col in cat_cols:\n    train_data[col] = train_data[col].astype('str')\n    test_data[col] = test_data[col].astype('str')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:02.045136Z","iopub.execute_input":"2024-12-05T13:10:02.045403Z","iopub.status.idle":"2024-12-05T13:10:02.412066Z","shell.execute_reply.started":"2024-12-05T13:10:02.045377Z","shell.execute_reply":"2024-12-05T13:10:02.41134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_data.drop(columns=['Policy Start Date', 'Premium Amount'], axis=1)\ny = train_data['Premium Amount']\n\nnew_test = test_data.drop(columns=['Policy Start Date'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:02.413003Z","iopub.execute_input":"2024-12-05T13:10:02.413238Z","iopub.status.idle":"2024-12-05T13:10:02.636946Z","shell.execute_reply.started":"2024-12-05T13:10:02.413214Z","shell.execute_reply":"2024-12-05T13:10:02.636226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skf = RepeatedKFold(n_splits=10, n_repeats=1, random_state=42)\ncat_params = {\n    'loss_function': 'MAE',\n    'iterations': 500,\n    'task_type': 'GPU'\n}\n\ntest_pool = Pool(data=new_test, cat_features=cat_cols)\n\nscores, cat_test_preds = [], []\nfor i, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n    \n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n    train_pool = Pool(data=X_train, label=y_train, cat_features=cat_cols)\n    eval_pool = Pool(data=X_val, label=y_val, cat_features=cat_cols)\n\n    model = CatBoostRegressor(**cat_params)\n    model.fit(train_pool, eval_set=eval_pool, verbose=0)\n    y_preds=model.predict(eval_pool)\n\n    score=np.mean((np.log1p(y_val)-np.log1p(y_preds)) ** 2)\n    print(f'The RMSLE Score is {score}')\n    scores.append(score)\n    cat_test_preds.append(model.predict(test_pool))\n\ncat_score = np.mean(scores)\ncat_std = np.std(scores)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:02.637911Z","iopub.execute_input":"2024-12-05T13:10:02.638186Z","iopub.status.idle":"2024-12-05T13:10:02.642406Z","shell.execute_reply.started":"2024-12-05T13:10:02.638161Z","shell.execute_reply":"2024-12-05T13:10:02.641591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nprint(f\"The 10-fold average RMSLE score of the CatBoost model is {cat_score}\")\nprint(f\"The 10-fold std RMSLE score of the CatBoost model is {cat_std}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:16.555587Z","iopub.execute_input":"2024-12-05T13:10:16.556292Z","iopub.status.idle":"2024-12-05T13:10:16.559836Z","shell.execute_reply.started":"2024-12-05T13:10:16.556261Z","shell.execute_reply":"2024-12-05T13:10:16.558895Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ------------------------------ Version 2 --------------------------------¶","metadata":{}},{"cell_type":"code","source":"cat_cols=['Marital Status',\n 'Education Level',\n 'Occupation',\n 'Location',\n 'Policy Type',\n 'Customer Feedback',\n 'Exercise Frequency',\n 'Property Type']\n\nfor col in cat_cols:\n    train_data[col] = train_data[col].astype('category')\n    test_data[col] = test_data[col].astype('category')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T18:11:56.41971Z","iopub.execute_input":"2024-12-05T18:11:56.420486Z","iopub.status.idle":"2024-12-05T18:11:57.342701Z","shell.execute_reply.started":"2024-12-05T18:11:56.420445Z","shell.execute_reply":"2024-12-05T18:11:57.342012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train_data.drop(columns=['Policy Start Date', 'Premium Amount'], axis=1)\ny = train_data['Premium Amount']\n\nnew_test = test_data.drop(columns=['Policy Start Date'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:03.096843Z","iopub.status.idle":"2024-12-05T13:10:03.097225Z","shell.execute_reply.started":"2024-12-05T13:10:03.097053Z","shell.execute_reply":"2024-12-05T13:10:03.097076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from lightgbm import LGBMRegressor\n# from sklearn.model_selection import train_test_split\n# import optuna\n\n\n\n# X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# def objective(trial):\n#     # LGBM parameters\n#     params = {\n#         \"verbosity\": -1,\n#         'objective': 'mae',\n#         'device': 'gpu',  # Use 'cpu' if GPU is not available\n#         'n_jobs': -1,\n#         'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.2),\n#         'max_depth': trial.suggest_int('max_depth', 3, 20),\n#         'n_estimators': trial.suggest_int('n_estimators', 300, 1300),\n#         \"reg_alpha\": trial.suggest_float(\"reg_alpha\", 1e-4, 1.0),\n#         \"reg_lambda\": trial.suggest_float(\"reg_lambda\", 1e-4, 1.0),\n#         'num_leaves': trial.suggest_int('num_leaves', 10, 100),\n#         'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0)\n#     }\n\n#     lgb_model = LGBMRegressor(**params)\n#     lgb_model.fit(X_train, y_train)\n#     y_pred = lgb_model.predict(X_val)\n    \n#     # Ensure predictions are non-negative for MSLE\n#     y_pred = np.maximum(y_pred, 0)\n#     score = np.mean((np.log1p(y_val) - np.log1p(y_pred))**2)\n    \n#     return score\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:03.098667Z","iopub.status.idle":"2024-12-05T13:10:03.099087Z","shell.execute_reply.started":"2024-12-05T13:10:03.098867Z","shell.execute_reply":"2024-12-05T13:10:03.098888Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# study = optuna.create_study(direction='minimize')\n# study.optimize(objective, n_trials=50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:03.100836Z","iopub.status.idle":"2024-12-05T13:10:03.101258Z","shell.execute_reply.started":"2024-12-05T13:10:03.101035Z","shell.execute_reply":"2024-12-05T13:10:03.101059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# {'learning_rate': 0.016709007593793575, 'max_depth': 3, 'n_estimators': 1215, 'reg_alpha': 0.6109919240604866, 'reg_lambda': 0.32911619217061006, 'num_leaves': 82, 'colsample_bytree': 0.502848902421112}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:03.102384Z","iopub.status.idle":"2024-12-05T13:10:03.102818Z","shell.execute_reply.started":"2024-12-05T13:10:03.102583Z","shell.execute_reply":"2024-12-05T13:10:03.102606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skf = RepeatedKFold(n_splits=10, n_repeats=1, random_state=42)\n\nlgb_params={\n            'objective': 'mae',\n            'device':'gpu',\n            'verbose':-1,\n            'n_jobs':-1,\n            'learning_rate': 0.016709007593793575, \n            'max_depth': 3, 'n_estimators': 1215, \n            'reg_alpha': 0.6109919240604866, \n            'reg_lambda': 0.32911619217061006, \n            'num_leaves': 82, \n            'colsample_bytree': 0.502848902421112}\n# \nscores , lgb_test_preds = [], []\nfor i, (train_idx, val_idx) in enumerate(skf.split(X,y)):\n    print(f'--------- Fold {i} -----------')\n    X_train, X_val = X.iloc[train_idx], X.iloc[val_idx]\n    y_train, y_val = y.iloc[train_idx], y.iloc[val_idx]\n\n    lgb_model = LGBMRegressor(**lgb_params)\n    lgb_model.fit(X_train, y_train)\n    y_preds = lgb_model.predict(X_val)\n\n    score = np.mean((np.log1p(y_val)-np.log1p(y_preds))**2)\n    print(f'RMSLE Score : {score}')\n    scores.append(score)\n    lgb_test_preds.append(lgb_model.predict(new_test))\n\nlgb_score = np.mean(scores)  \nlgb_std = np.std(scores)\nprint(f\"The 10-fold average RMSLE score of the LGBM model is {lgb_score}\")\nprint(f\"The 10-fold std RMSLE score of the LGBM model is {lgb_std}\")  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:10:03.104016Z","iopub.status.idle":"2024-12-05T13:10:03.104438Z","shell.execute_reply.started":"2024-12-05T13:10:03.104215Z","shell.execute_reply":"2024-12-05T13:10:03.104236Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ------------------------------ Version 3 --------------------------------","metadata":{}},{"cell_type":"code","source":"train_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T15:21:06.170112Z","iopub.execute_input":"2024-12-05T15:21:06.170444Z","iopub.status.idle":"2024-12-05T15:21:06.194828Z","shell.execute_reply.started":"2024-12-05T15:21:06.170414Z","shell.execute_reply":"2024-12-05T15:21:06.193999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train_data.copy()\ntest = test_data.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T15:55:40.745302Z","iopub.execute_input":"2024-12-05T15:55:40.746082Z","iopub.status.idle":"2024-12-05T15:55:40.99163Z","shell.execute_reply.started":"2024-12-05T15:55:40.746032Z","shell.execute_reply":"2024-12-05T15:55:40.990638Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def data(df):\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Month_name'] = df['Policy Start Date'].dt.month_name()\n    df['Day_of_week'] = df['Policy Start Date'].dt.day_name()\n    df['Week'] = df['Policy Start Date'].dt.isocalendar().week\n    df['Year_sin'] = np.sin(2* np.pi * df['Year'])\n    df['Year_cos'] = np.cos(2* np.pi * df['Year'])\n    df['Month_sin'] =  np.sin(2* np.pi * df['Month']/12)\n    df['Month_cos'] = np.cos(2* np.pi * df['Month']/12)\n    df['Day_sin'] = np.sin(2* np.pi * df['Day']/31)\n    df['Day_cos'] = np.cos(2* np.pi * df['Day']/31)\n\n    df['Group'] = (df['Year']-2020)*48+df['Month']*4+df['Day']//7\n\n    df.drop(columns=['Policy Start Date'], axis=1, inplace=True)\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T15:55:46.370449Z","iopub.execute_input":"2024-12-05T15:55:46.371262Z","iopub.status.idle":"2024-12-05T15:55:46.381157Z","shell.execute_reply.started":"2024-12-05T15:55:46.371213Z","shell.execute_reply":"2024-12-05T15:55:46.380173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = data(train)\ntest_df = data(test)\n\ncat_cols = [col for col in train_df.columns if train_df[col].dtypes=='object']\nfeatures_cols = list(test_df.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T15:55:59.589984Z","iopub.execute_input":"2024-12-05T15:55:59.590813Z","iopub.status.idle":"2024-12-05T15:56:02.240039Z","shell.execute_reply.started":"2024-12-05T15:55:59.590754Z","shell.execute_reply":"2024-12-05T15:56:02.239328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CategoricalEncoder:\n    def __init__(self, train, test):\n        self.train = train\n        self.test = test\n\n    def frequency_encode(self, cat_cols, features_cols, drop_org=False):\n\n        new_cat_cols=[]\n        for col in cat_cols:\n            freq_encoding = self.train[col].value_counts().to_dict()\n\n            self.train[f'{col}_freq'] = self.train[col].map(freq_encoding).astype('float')\n            self.test[f'{col}_freq'] = self.test[col].map(freq_encoding).astype('float')\n\n            new_col_name = f'{col}_freq'\n            new_cat_cols.append(new_col_name)\n            features_cols.append(new_col_name)\n            if drop_org:\n                features_cols.remove(col)\n        return self.train, self.test, new_cat_cols, features_cols\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:12:10.579411Z","iopub.execute_input":"2024-12-05T17:12:10.580281Z","iopub.status.idle":"2024-12-05T17:12:10.586252Z","shell.execute_reply.started":"2024-12-05T17:12:10.580243Z","shell.execute_reply":"2024-12-05T17:12:10.585328Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encoder = CategoricalEncoder(train_df, test_df)\ntrain, test, cat_cols, feature_cols = encoder.frequency_encode(cat_cols, features_cols, drop_org=True)\n\ntrain = train[feature_cols + ['Premium Amount']]\ntest = test[feature_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:15:37.819291Z","iopub.execute_input":"2024-12-05T17:15:37.820115Z","iopub.status.idle":"2024-12-05T17:15:40.547466Z","shell.execute_reply.started":"2024-12-05T17:15:37.82008Z","shell.execute_reply":"2024-12-05T17:15:40.546498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T18:14:49.104709Z","iopub.execute_input":"2024-12-05T18:14:49.105098Z","iopub.status.idle":"2024-12-05T18:14:49.110922Z","shell.execute_reply.started":"2024-12-05T18:14:49.105067Z","shell.execute_reply":"2024-12-05T18:14:49.110037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = train.drop('Premium Amount', axis=1)\ny = train['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:36:30.269375Z","iopub.execute_input":"2024-12-05T17:36:30.269725Z","iopub.status.idle":"2024-12-05T17:36:30.362767Z","shell.execute_reply.started":"2024-12-05T17:36:30.269687Z","shell.execute_reply":"2024-12-05T17:36:30.362127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\ndef objectives(trial):\n    params = {\n        'objective': 'mae',\n        'devic' : 'GPU',\n        'iterations': trial.suggest_int('iterations', 100, 1000),\n        'learning_rate': trial.suggest_float('learning_rate', 0.0001, 0.1),\n        'depth' : trial.suggest_int('depth', 3, 14),\n        'border_count': trial.suggest_int('border_count', 32, 255),\n        'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 0.0001, 0.1),\n        'random_strength': trial.suggest_float('random_strength', 0.0001, 1),\n        'random_seed': 42,\n        'verbose':-1\n    }\n\n    train_pool = Pool(X_train, y_train, cat_features=cat_cols)\n    val_pool = Pool(X_val, y_val, cat_features=cat_cols)\n\n    model = CatBoostRegressor(**params)\n    model.fit(train_pool, eval_set=val_pool, early_stopping_rounds=50)\n    y_pred = model.predict(X_val)\n    score = np.mean((np.log1p(y_val)-np.log1p(y_pred))**2)\n    return score\nstudy = optuna.create_study(direction='minimize')\nstudy.optimize(objectives, n_trials=50)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T18:16:32.670405Z","iopub.execute_input":"2024-12-05T18:16:32.671099Z","iopub.status.idle":"2024-12-05T18:16:33.220939Z","shell.execute_reply.started":"2024-12-05T18:16:32.671063Z","shell.execute_reply":"2024-12-05T18:16:33.21969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\nimport gc\nsubmission = pd.read_csv('../input/playground-series-s4e12/sample_submission.csv')\nsubmission['Premium Amount'] = np.mean(lgb_test_preds, axis=0)\nprint(submission.head())\n\nsubmission.to_csv('model_3_lgbm.csv', index=False)\n\ndel submission\ngc.collect()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T17:31:14.349456Z","iopub.execute_input":"2024-12-05T17:31:14.349836Z","iopub.status.idle":"2024-12-05T17:31:17.774686Z","shell.execute_reply.started":"2024-12-05T17:31:14.349802Z","shell.execute_reply":"2024-12-05T17:31:17.773816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}