{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30805,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-23T06:27:41.827015Z","iopub.execute_input":"2024-12-23T06:27:41.827224Z","iopub.status.idle":"2024-12-23T06:27:42.793006Z","shell.execute_reply.started":"2024-12-23T06:27:41.827201Z","shell.execute_reply":"2024-12-23T06:27:42.79201Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T06:27:46.677168Z","iopub.execute_input":"2024-12-23T06:27:46.677954Z","iopub.status.idle":"2024-12-23T06:27:59.196512Z","shell.execute_reply.started":"2024-12-23T06:27:46.677919Z","shell.execute_reply":"2024-12-23T06:27:59.195591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn\nsklearn.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T06:28:02.212922Z","iopub.execute_input":"2024-12-23T06:28:02.213285Z","iopub.status.idle":"2024-12-23T06:28:02.60682Z","shell.execute_reply.started":"2024-12-23T06:28:02.213252Z","shell.execute_reply":"2024-12-23T06:28:02.605952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.gridspec as gridspec\nfrom sklearn.cluster import KMeans\nfrom sklearn.model_selection import cross_val_score\nfrom xgboost import XGBRegressor\nfrom catboost import Pool,CatBoostRegressor\nimport lightgbm as lgb\nfrom sklearn.impute import SimpleImputer\nimport seaborn as sns\nimport numpy as np\nfrom scipy.stats import norm\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom scipy import stats\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import root_mean_squared_log_error,mean_squared_error, mean_absolute_error, r2_score\nimport optuna\nfrom optuna.samplers import TPESampler\nfrom optuna.samplers import RandomSampler\nimport shap\nimport warnings\nwarnings.filterwarnings('ignore')\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T06:28:06.488935Z","iopub.execute_input":"2024-12-23T06:28:06.489509Z","iopub.status.idle":"2024-12-23T06:28:13.344128Z","shell.execute_reply.started":"2024-12-23T06:28:06.489475Z","shell.execute_reply":"2024-12-23T06:28:13.34343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nids = df_test['id'].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T06:28:18.068799Z","iopub.execute_input":"2024-12-23T06:28:18.069447Z","iopub.status.idle":"2024-12-23T06:28:26.662639Z","shell.execute_reply.started":"2024-12-23T06:28:18.069411Z","shell.execute_reply":"2024-12-23T06:28:26.661864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.float_format', lambda x: '%.2f' % x)\ndf_train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T06:28:35.652764Z","iopub.execute_input":"2024-12-23T06:28:35.653113Z","iopub.status.idle":"2024-12-23T06:28:36.230227Z","shell.execute_reply.started":"2024-12-23T06:28:35.653083Z","shell.execute_reply":"2024-12-23T06:28:36.229392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T06:31:17.898784Z","iopub.execute_input":"2024-12-23T06:31:17.899692Z","iopub.status.idle":"2024-12-23T06:31:17.90791Z","shell.execute_reply.started":"2024-12-23T06:31:17.899639Z","shell.execute_reply":"2024-12-23T06:31:17.906994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['Premium Amount'].describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T07:36:14.93857Z","iopub.execute_input":"2024-12-11T07:36:14.938923Z","iopub.status.idle":"2024-12-11T07:36:14.993601Z","shell.execute_reply.started":"2024-12-11T07:36:14.938888Z","shell.execute_reply":"2024-12-11T07:36:14.992756Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.distplot(df_train['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T07:36:14.994907Z","iopub.execute_input":"2024-12-11T07:36:14.995267Z","iopub.status.idle":"2024-12-11T07:36:19.480727Z","shell.execute_reply.started":"2024-12-11T07:36:14.99522Z","shell.execute_reply":"2024-12-11T07:36:19.479891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-11T07:36:19.484639Z","iopub.execute_input":"2024-12-11T07:36:19.48494Z","iopub.status.idle":"2024-12-11T07:36:20.070087Z","shell.execute_reply.started":"2024-12-11T07:36:19.484913Z","shell.execute_reply":"2024-12-11T07:36:20.069159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:19:06.447348Z","iopub.execute_input":"2024-12-23T07:19:06.448147Z","iopub.status.idle":"2024-12-23T07:19:06.463219Z","shell.execute_reply.started":"2024-12-23T07:19:06.448112Z","shell.execute_reply":"2024-12-23T07:19:06.462316Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def new_feature(df):\n    df['Claims frequency'] = df['Previous Claims']/df['Insurance Duration']\n    df['Income for person'] = df['Annual Income']/(1 + df['Number of Dependents'])\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:18:53.150546Z","iopub.execute_input":"2024-12-23T07:18:53.151402Z","iopub.status.idle":"2024-12-23T07:18:53.155528Z","shell.execute_reply.started":"2024-12-23T07:18:53.15137Z","shell.execute_reply":"2024-12-23T07:18:53.154633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = new_feature(df_train)\ndf_test = new_feature(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:18:57.188071Z","iopub.execute_input":"2024-12-23T07:18:57.188403Z","iopub.status.idle":"2024-12-23T07:18:57.211045Z","shell.execute_reply.started":"2024-12-23T07:18:57.188374Z","shell.execute_reply":"2024-12-23T07:18:57.210111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical = [var for var in df_train.columns if df_train[var].dtype=='object']\nnumerical = [var for var in df_train.columns if df_train[var].dtype != 'object' and (var != 'Premium Amount')]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:20:54.896324Z","iopub.execute_input":"2024-12-23T07:20:54.8972Z","iopub.status.idle":"2024-12-23T07:20:54.904719Z","shell.execute_reply.started":"2024-12-23T07:20:54.897148Z","shell.execute_reply":"2024-12-23T07:20:54.90382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"target_column = 'Premium Amount'\nprint(\"Target Column:\", target_column)\nprint(\"\\nCategorical Columns:\", categorical)\nprint(\"\\nNumerical Columns:\", numerical)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:21:00.901092Z","iopub.execute_input":"2024-12-23T07:21:00.901741Z","iopub.status.idle":"2024-12-23T07:21:00.906342Z","shell.execute_reply.started":"2024-12-23T07:21:00.901705Z","shell.execute_reply":"2024-12-23T07:21:00.905455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filtered_columns = [col for col in categorical if col != 'Policy Start Date']\n\nfig, axes = plt.subplots(len(filtered_columns), 2, figsize=(15, 5 * len(filtered_columns)))\n\nfor i, column in enumerate(filtered_columns):\n \n    sns.countplot(data=df_train, x=column, ax=axes[i, 0], palette='tab10')\n    axes[i, 0].set_title(f'Distribution of {column}', fontsize=14)\n    axes[i, 0].set_xlabel(column, fontsize=12)\n    axes[i, 0].set_ylabel('Count', fontsize=12)\n    sns.despine(ax=axes[i, 0])\n\n  \n    sns.boxplot(data=df_train, x=column, y=target_column, ax=axes[i, 1], palette='tab10')\n    axes[i, 1].set_title(f'{column} vs {target_column}', fontsize=14)\n    axes[i, 1].set_xlabel(column, fontsize=12)\n    axes[i, 1].set_ylabel(target_column, fontsize=12)\n    sns.despine(ax=axes[i, 1])\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:21:14.265995Z","iopub.execute_input":"2024-12-23T07:21:14.266685Z","iopub.status.idle":"2024-12-23T07:21:27.768228Z","shell.execute_reply.started":"2024-12-23T07:21:14.266652Z","shell.execute_reply":"2024-12-23T07:21:27.767375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"palette = sns.color_palette('tab10', len(numerical))\ncolor_dict = dict(zip(numerical, palette))\n\n# Create a grid of subplots for histograms, boxplots, and scatterplots/violin plots\nfig = plt.figure(figsize=(30, 10 * len(numerical)))\ngs = gridspec.GridSpec(2 * len(numerical), 2, figure=fig)\n\ndf_binned = df_train.copy()\n\nfor i, column in enumerate(numerical):\n\n    if df_train[column].nunique() > 50: discrete = False\n    else : discrete = True\n    \n    # Plot histogram with a unique color\n    ax_hist = fig.add_subplot(gs[2 * i, 0])\n    sns.histplot(\n        data=df_train, x=column, fill=True, common_norm=False, alpha=0.6,\n        linewidth=0.8, color=color_dict[column], ax=ax_hist,  discrete = discrete\n    )\n    \n    # Plot boxplot with the same unique color\n    ax_box = fig.add_subplot(gs[2 * i + 1, 0])\n    sns.boxplot(data=df_train, x=column, ax=ax_box, color=color_dict[column])\n    ax_box.set_title(f'{column} vs Target (Boxplot)', fontsize=14)\n    sns.despine(ax=ax_box)\n\n    # Conditional plot: violin plot or barplot based on unique values, fallback to scatterplot\n    ax_conditional = fig.add_subplot(gs[2 * i:2 * i + 2, 1])  # Merges 2 rows\n    if df_train[column].nunique() <= 10:\n        # If the column has 10 or fewer unique values, use a violin plot\n        sns.violinplot(data=df_train, x=column, y=target_column, ax=ax_conditional, color=color_dict[column], alpha=0.6)\n        ax_conditional.set_title(f'{column} vs {target_column} (Violin Plot)', fontsize=14)\n    else:\n        # Bin the column into 10 intervals, but keep original target column values\n        df_binned['Binned Column'] = pd.cut(df_train[column], bins=10)\n        sns.violinplot(data=df_binned, x='Binned Column', y=target_column, ax=ax_conditional, color=color_dict[column], alpha=0.6)\n        ax_conditional.set_title(f'{column} (Binned) vs {target_column} (Violin Plot)', fontsize=14)\n        ax_conditional.set_xlabel(f'{column} (Binned)', fontsize=12)\n\nplt.tight_layout()  # Adjust subplots to fit into the figure area\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:22:02.847264Z","iopub.execute_input":"2024-12-23T07:22:02.847634Z","iopub.status.idle":"2024-12-23T07:22:44.240481Z","shell.execute_reply.started":"2024-12-23T07:22:02.847603Z","shell.execute_reply":"2024-12-23T07:22:44.239299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15,9))\nplt.title(\"Visualizing Missing Values\")\nsns.heatmap(df_train.isnull(), cbar=False, cmap=sns.color_palette('magma'), yticklabels=False);\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:24:12.7375Z","iopub.execute_input":"2024-12-23T07:24:12.738309Z","iopub.status.idle":"2024-12-23T07:24:34.81302Z","shell.execute_reply.started":"2024-12-23T07:24:12.738276Z","shell.execute_reply":"2024-12-23T07:24:34.812185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorrelation_matrix = df_train[numerical].corr()\n\n# Plot the heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, fmt=\".2f\", cmap=\"coolwarm\", cbar=True, linewidths=0.5)\nplt.title(\"Correlation Heatmap of Numerical Variables\", fontsize=16)\nplt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:24:54.154231Z","iopub.execute_input":"2024-12-23T07:24:54.155104Z","iopub.status.idle":"2024-12-23T07:24:55.168086Z","shell.execute_reply.started":"2024-12-23T07:24:54.155058Z","shell.execute_reply":"2024-12-23T07:24:55.167233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def date(df):\n\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Month_name'] = df['Policy Start Date'].dt.month_name()\n    df['Day_of_week'] = df['Policy Start Date'].dt.day_name()\n    df['Week'] = df['Policy Start Date'].dt.isocalendar().week\n    min_year = df['Year'].min()\n    max_year = df['Year'].max()\n    df.drop('Policy Start Date', axis=1, inplace=True)\n\n    return df\n\n# Apply the date function to both datasets\ndf_train = date(df_train)\ndf_test = date(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:25:14.521457Z","iopub.execute_input":"2024-12-23T07:25:14.522298Z","iopub.status.idle":"2024-12-23T07:25:16.847443Z","shell.execute_reply.started":"2024-12-23T07:25:14.522266Z","shell.execute_reply":"2024-12-23T07:25:16.846496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical.remove('Policy Start Date')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:25:31.751451Z","iopub.execute_input":"2024-12-23T07:25:31.752136Z","iopub.status.idle":"2024-12-23T07:25:31.755918Z","shell.execute_reply.started":"2024-12-23T07:25:31.752101Z","shell.execute_reply":"2024-12-23T07:25:31.754936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Shape of training data (num_rows, num_columns)\nprint(df_train[numerical[1:]].shape)\n\n# Number of missing values in each column of training data\nmissing_val_count_by_column = (df_train[numerical[1:]].isnull().sum())\nprint(missing_val_count_by_column[missing_val_count_by_column > 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:25:36.320159Z","iopub.execute_input":"2024-12-23T07:25:36.320497Z","iopub.status.idle":"2024-12-23T07:25:36.416597Z","shell.execute_reply.started":"2024-12-23T07:25:36.320467Z","shell.execute_reply":"2024-12-23T07:25:36.415627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_imputer = SimpleImputer( strategy='median')\ndf_train[numerical[1:]]=num_imputer.fit_transform(df_train[numerical[1:]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:25:52.545372Z","iopub.execute_input":"2024-12-23T07:25:52.545729Z","iopub.status.idle":"2024-12-23T07:25:54.160041Z","shell.execute_reply.started":"2024-12-23T07:25:52.5457Z","shell.execute_reply":"2024-12-23T07:25:54.159095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Shape of training data (cat_rows, cat_columns)\nprint(df_train[categorical].shape)\n\n# Number of missing values in each column of training data\nmissing_val_count_by_column = (df_train[categorical].isnull().sum())\nprint(missing_val_count_by_column[missing_val_count_by_column > 0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:25:57.938952Z","iopub.execute_input":"2024-12-23T07:25:57.939771Z","iopub.status.idle":"2024-12-23T07:25:58.664104Z","shell.execute_reply.started":"2024-12-23T07:25:57.939737Z","shell.execute_reply":"2024-12-23T07:25:58.66334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"obj_imputer = SimpleImputer( strategy='constant', fill_value='Unknown')\ndf_train[categorical] = obj_imputer.fit_transform(df_train[categorical])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:26:03.369686Z","iopub.execute_input":"2024-12-23T07:26:03.370046Z","iopub.status.idle":"2024-12-23T07:26:04.955492Z","shell.execute_reply.started":"2024-12-23T07:26:03.370015Z","shell.execute_reply":"2024-12-23T07:26:04.954466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.drop(columns=['id'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:26:10.458758Z","iopub.execute_input":"2024-12-23T07:26:10.459427Z","iopub.status.idle":"2024-12-23T07:26:10.591382Z","shell.execute_reply.started":"2024-12-23T07:26:10.459396Z","shell.execute_reply":"2024-12-23T07:26:10.590633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test[numerical[1:]]=num_imputer.fit_transform(df_test[numerical[1:]])\ndf_test[categorical]=obj_imputer.fit_transform(df_test[categorical])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:26:14.977759Z","iopub.execute_input":"2024-12-23T07:26:14.978381Z","iopub.status.idle":"2024-12-23T07:26:16.975278Z","shell.execute_reply.started":"2024-12-23T07:26:14.978351Z","shell.execute_reply":"2024-12-23T07:26:16.974236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:26:20.026303Z","iopub.execute_input":"2024-12-23T07:26:20.026703Z","iopub.status.idle":"2024-12-23T07:26:20.450539Z","shell.execute_reply.started":"2024-12-23T07:26:20.026673Z","shell.execute_reply":"2024-12-23T07:26:20.449561Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Scaling Features","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\nscaler.fit(df_train[numerical[1:]])\ndf_train[numerical[1:]] = scaler.transform(df_train[numerical[1:]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:27:06.400196Z","iopub.execute_input":"2024-12-23T07:27:06.400984Z","iopub.status.idle":"2024-12-23T07:27:06.755776Z","shell.execute_reply.started":"2024-12-23T07:27:06.400947Z","shell.execute_reply":"2024-12-23T07:27:06.754793Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler.fit(df_test[numerical[1:]])\ndf_test[numerical[1:]] = scaler.transform(df_test[numerical[1:]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:27:10.518707Z","iopub.execute_input":"2024-12-23T07:27:10.519175Z","iopub.status.idle":"2024-12-23T07:27:10.728401Z","shell.execute_reply.started":"2024-12-23T07:27:10.519141Z","shell.execute_reply":"2024-12-23T07:27:10.727714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical = [var for var in df_train.columns if df_train[var].dtype=='object']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:27:14.278241Z","iopub.execute_input":"2024-12-23T07:27:14.279196Z","iopub.status.idle":"2024-12-23T07:27:14.285446Z","shell.execute_reply.started":"2024-12-23T07:27:14.27915Z","shell.execute_reply":"2024-12-23T07:27:14.284381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"enc = OrdinalEncoder()\ndf_train[categorical] = enc.fit_transform(df_train[categorical])\ndf_test[categorical] = enc.fit_transform(df_test[categorical])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:27:18.504465Z","iopub.execute_input":"2024-12-23T07:27:18.505141Z","iopub.status.idle":"2024-12-23T07:27:23.796488Z","shell.execute_reply.started":"2024-12-23T07:27:18.505107Z","shell.execute_reply":"2024-12-23T07:27:23.79554Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,10))\n\ncorr_matrix = df_train.corr()\nlower = corr_matrix.where(np.tril(np.ones(corr_matrix.shape), k=-1).astype(bool))\n\nsns.heatmap(lower, annot=True, fmt='.2f', cbar=False, center=0);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:27:26.363302Z","iopub.execute_input":"2024-12-23T07:27:26.363922Z","iopub.status.idle":"2024-12-23T07:27:29.805933Z","shell.execute_reply.started":"2024-12-23T07:27:26.363889Z","shell.execute_reply":"2024-12-23T07:27:29.805001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"high_corr = [\n    column for column in lower.columns if any((lower[column] > 0.6)|(lower[column] < -0.6))\n]\nhigh_corr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:27:46.349741Z","iopub.execute_input":"2024-12-23T07:27:46.350424Z","iopub.status.idle":"2024-12-23T07:27:46.364621Z","shell.execute_reply.started":"2024-12-23T07:27:46.350395Z","shell.execute_reply":"2024-12-23T07:27:46.363589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.drop(columns=high_corr, inplace=True)\ndf_test.drop(columns=high_corr, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:28:36.478148Z","iopub.execute_input":"2024-12-23T07:28:36.478754Z","iopub.status.idle":"2024-12-23T07:28:36.718672Z","shell.execute_reply.started":"2024-12-23T07:28:36.478722Z","shell.execute_reply":"2024-12-23T07:28:36.717933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_features = df_train.drop(columns=['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:28:39.969514Z","iopub.execute_input":"2024-12-23T07:28:39.969898Z","iopub.status.idle":"2024-12-23T07:28:40.094444Z","shell.execute_reply.started":"2024-12-23T07:28:39.969869Z","shell.execute_reply":"2024-12-23T07:28:40.092495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def outlier_std(data, col, threshold=3):\n    mean = data[col].mean()\n    std = data[col].std()\n    up_bound = mean + threshold * std\n    low_bound = mean - threshold * std\n    anomalies = pd.concat([data[col]>up_bound, data[col]<low_bound], axis=1).any(axis=1)\n    return anomalies, up_bound, low_bound","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:28:44.212082Z","iopub.execute_input":"2024-12-23T07:28:44.212948Z","iopub.status.idle":"2024-12-23T07:28:44.217879Z","shell.execute_reply.started":"2024-12-23T07:28:44.212902Z","shell.execute_reply":"2024-12-23T07:28:44.21691Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_column_outliers(data, columns=None, function=outlier_std, threshold=3):\n    if columns:\n        columns_to_check = columns\n    else:\n        columns_to_check = data.columns\n\n    outliers = pd.Series(data=[False]*len(data), index=data_features.index, name='is_outlier')\n    comparison_table = {}\n    for column in columns_to_check:\n        anomalies, upper_bound, lower_bound = function(data, column, threshold=threshold)\n        comparison_table[column] = [upper_bound, lower_bound, sum(anomalies), 100*sum(anomalies)/len(anomalies)]\n        outliers[anomalies[anomalies].index] = True\n\n    comparison_table = pd.DataFrame(comparison_table).T\n    comparison_table.columns=['upper_bound', 'lower_bound', 'anomalies_count', 'anomalies_percentage']\n    comparison_table = comparison_table.sort_values(by='anomalies_percentage', ascending=False)\n\n    return comparison_table, outliers\n\ndef anomalies_report(outliers):\n    print(\"Total number of outliers: {}\\nPercentage of outliers:   {:.2f}%\".format(\n            sum(outliers), 100*sum(outliers)/len(outliers)))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:28:47.666452Z","iopub.execute_input":"2024-12-23T07:28:47.667274Z","iopub.status.idle":"2024-12-23T07:28:47.676156Z","shell.execute_reply.started":"2024-12-23T07:28:47.667226Z","shell.execute_reply":"2024-12-23T07:28:47.674982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"comparison_table, std_outliers = get_column_outliers(data_features)\nanomalies_report(std_outliers)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:28:51.623928Z","iopub.execute_input":"2024-12-23T07:28:51.624264Z","iopub.status.idle":"2024-12-23T07:28:56.315518Z","shell.execute_reply.started":"2024-12-23T07:28:51.624234Z","shell.execute_reply":"2024-12-23T07:28:56.314631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"comparison_table","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:29:00.477363Z","iopub.execute_input":"2024-12-23T07:29:00.477728Z","iopub.status.idle":"2024-12-23T07:29:00.488384Z","shell.execute_reply.started":"2024-12-23T07:29:00.477697Z","shell.execute_reply":"2024-12-23T07:29:00.487472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['is_outlier'] = std_outliers\ndf_train = df_train.drop(df_train[df_train.is_outlier == True].index).reset_index(drop=True)\ndf_train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:29:11.417828Z","iopub.execute_input":"2024-12-23T07:29:11.418476Z","iopub.status.idle":"2024-12-23T07:29:11.863289Z","shell.execute_reply.started":"2024-12-23T07:29:11.418445Z","shell.execute_reply":"2024-12-23T07:29:11.862503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split train data into features and target\nX = df_train.drop(columns=[target_column, 'id', 'is_outlier'])\ny = df_train[target_column]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:29:24.809377Z","iopub.execute_input":"2024-12-23T07:29:24.81019Z","iopub.status.idle":"2024-12-23T07:29:24.851231Z","shell.execute_reply.started":"2024-12-23T07:29:24.810158Z","shell.execute_reply":"2024-12-23T07:29:24.850613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:29:29.184693Z","iopub.execute_input":"2024-12-23T07:29:29.18537Z","iopub.status.idle":"2024-12-23T07:29:29.424965Z","shell.execute_reply.started":"2024-12-23T07:29:29.185338Z","shell.execute_reply":"2024-12-23T07:29:29.423941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import math\nclass RMSLE(object):\n    def calc_ders_range(self, approxes, targets, weights):\n        assert len(approxes) == len(targets)\n        if weights is not None:\n            assert len(weights) == len(approxes)\n\n        result = []\n        for index in range(len(targets)):\n            val = max(approxes[index], 0)\n            der1 = math.log1p(targets[index]) - math.log1p(max(0, approxes[index]))\n            der2 = -1 / (max(0, approxes[index]) + 1)\n\n            if weights is not None:\n                der1 *= weights[index]\n                der2 *= weights[index]\n\n            result.append((der1, der2))\n        return result\nclass RMSLE_val(object):\n    def get_final_error(self, error, weight):\n        return np.sqrt(error / (weight + 1e-38))\n\n    def is_max_optimal(self):\n        return False\n\n    def evaluate(self, approxes, target, weight):\n        assert len(approxes) == 1\n        assert len(target) == len(approxes[0])\n\n        approx = approxes[0]\n\n        error_sum = 0.0\n        weight_sum = 0.0\n\n        for i in range(len(approx)):\n            w = 1.0 if weight is None else weight[i]\n            weight_sum += w\n            error_sum += w * ((math.log1p(max(0, approx[i])) - math.log1p(max(0, target[i])))**2)\n\n        return error_sum, weight_sum","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:29:32.896371Z","iopub.execute_input":"2024-12-23T07:29:32.896756Z","iopub.status.idle":"2024-12-23T07:29:32.905431Z","shell.execute_reply.started":"2024-12-23T07:29:32.896723Z","shell.execute_reply":"2024-12-23T07:29:32.904528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cb_model = CatBoostRegressor(iterations=3000,\n                          early_stopping_rounds=100,\n                          grow_policy = 'Depthwise',\n                          depth=8,\n                          loss_function=RMSLE(),\n                          random_state=42,\n                          l2_leaf_reg = 1,\n                          learning_rate=0.03,\n                          verbose=10,\n                          eval_metric=RMSLE_val())\nparams = {'l2_leaf_reg':[1e-1, 1.0],\n          'learning_rate': [0.001, 0.01,0.1],\n          'depth':[8,9,10]\n         }\n#grid_search_res = cb_model.grid_search(params, X=X, y=y, train_size=0.8)\n#grid_search_res","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T07:31:07.734493Z","iopub.execute_input":"2024-12-23T07:31:07.735367Z","iopub.status.idle":"2024-12-23T14:56:31.706334Z","shell.execute_reply.started":"2024-12-23T07:31:07.735333Z","shell.execute_reply":"2024-12-23T14:56:31.70548Z"},"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nxgb_params = {'learning_rate': 0.06435987754796735,\n              'max_depth': 15,\n              'min_child_weight': 844,\n              'max_delta_step': 0, \n              'subsample': 0.6787222525610624,\n              'colsample_bytree': 0.9911185805683849,\n              'reg_lambda': 0.16605803670617628,\n              'reg_alpha': 0.2773706774346394,\n              'gamma': 0.004371913397845945,\n              'n_estimators': 100,\n              'device': 'gpu',\n              'sampling_method': 'gradient_based',\n              'metric': \"rmsle\",\n             'eval_metric': \"rmsle\",\n              'tree_method': 'hist',\n            'verbosity': 0,\n             'seed': 42}\nlgb_params = {'boosting_type': 'dart', \n                       'num_leaves': 506,\n                       'min_child_samples': 88, \n                       'learning_rate': 0.023941956540168206, \n                       'reg_alpha': 5.544302771418543e-07,\n                       'reg_lambda': 7.792319838879878e-06,\n                       'feature_fraction': 0.9785381529571701,\n                       'bagging_fraction': 0.6377113808460491,\n                       'bagging_freq': 10,\n                       'min_data_in_leaf': 45,\n                       'max_depth': 14,\n                       'lambda_l1': 0.07576182964120466,\n                       'lambda_l2': 9.86798320013589,\n                       'colsample_bytree': 0.8908685026291246,\n                       'subsample': 0.5636997605277937,\n                       'subsample_freq': 0,\n                        'seed': 42}\ncb_params = {'depth': 10, \n             'learning_rate': 0.01,\n             'l2_leaf_reg': 0.1,\n            'iterations':3000,\n            'grow_policy': 'Depthwise',\n            'loss_function': RMSLE(),\n            'random_state':42,\n            'verbose': 10,\n            'eval_metric':RMSLE_val()}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T15:19:11.016545Z","iopub.execute_input":"2024-12-23T15:19:11.017213Z","iopub.status.idle":"2024-12-23T15:19:11.023918Z","shell.execute_reply.started":"2024-12-23T15:19:11.01718Z","shell.execute_reply":"2024-12-23T15:19:11.023022Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"grid_search_res","metadata":{}},{"cell_type":"markdown","source":"CatBoostRegressor","metadata":{}},{"cell_type":"code","source":"cb_model = CatBoostRegressor(**cb_params)\ncb_model.fit(X,y)\ny_pred = cb_model.predict(X)\n# Calcul des métriques\nrmsle = root_mean_squared_log_error(y, y_pred)\nrmse = np.sqrt(mean_squared_error(y, y_pred))\nmae = mean_absolute_error(y, y_pred)\nr2 = r2_score(y, y_pred)\nmape = np.mean(np.abs((y - y_pred) / y)) * 100\n\n# Display performance metrics\nprint(f\"\\nPerformance Metrics:\\n{'-'*30}\")\nprint(f\"RMSLE: {rmsle:.4f}\")\nprint(f\"RMSE: {rmse:.4f}\")\nprint(f\"MAE: {mae:.4f}\")\nprint(f\"R²: {r2:.4f}\")\nprint(f\"MAPE: {mape:.2f}%\")\n","metadata":{"execution":{"iopub.status.busy":"2024-12-23T15:19:18.980431Z","iopub.execute_input":"2024-12-23T15:19:18.980781Z","iopub.status.idle":"2024-12-23T15:54:31.330085Z","shell.execute_reply.started":"2024-12-23T15:19:18.980751Z","shell.execute_reply":"2024-12-23T15:54:31.329027Z"},"trusted":true,"_kg_hide-input":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Feature Importance\n\nfeature_important = pd.DataFrame({'feature_importance': cb_model.get_feature_importance(Pool(X_train, y_train)), \n              'feature_names': X_val.columns}).sort_values(by=['feature_importance'], \n                                                       ascending=False)\nfeature_important.sort_values(by=['feature_importance'], ascending=True).plot.barh(x='feature_names', y='feature_importance')","metadata":{"execution":{"iopub.status.busy":"2024-12-23T16:07:42.608207Z","iopub.execute_input":"2024-12-23T16:07:42.608964Z","iopub.status.idle":"2024-12-23T16:09:29.105593Z","shell.execute_reply.started":"2024-12-23T16:07:42.608925Z","shell.execute_reply":"2024-12-23T16:09:29.10489Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Residual Analysis\nresiduals = y - y_pred\n\n# Residuals vs Predicted Values\nplt.figure(figsize=(12, 6))\nsns.scatterplot(x=y_pred, y=residuals, alpha=0.6, color=\"#007acc\")\nplt.axhline(y=0, color='red', linestyle='--', linewidth=1.5)\nplt.title(\"Residuals vs Predicted Values\", fontsize=16, fontweight='bold')\nplt.xlabel(\"Predicted Values\", fontsize=12)\nplt.ylabel(\"Residuals\", fontsize=12)\nplt.tight_layout()\nplt.show()\n\n# Residual Distribution\nplt.figure(figsize=(10, 6))\nsns.histplot(residuals, bins=30, kde=True, color=\"#55a630\")\nplt.axvline(x=0, color='red', linestyle='--', linewidth=1.5)\nplt.title(\"Distribution of Residuals\", fontsize=16, fontweight='bold')\nplt.xlabel(\"Residuals\", fontsize=12)\nplt.ylabel(\"Frequency\", fontsize=12)\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-12-23T16:11:57.242823Z","iopub.execute_input":"2024-12-23T16:11:57.243711Z","iopub.status.idle":"2024-12-23T16:12:04.522208Z","shell.execute_reply.started":"2024-12-23T16:11:57.243663Z","shell.execute_reply":"2024-12-23T16:12:04.521351Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test set\ntest_predictions = cb_model.predict(df_test)\n\n# Prepare submission file\nsubmission = pd.DataFrame({'id': ids, 'Premium Amount': test_predictions})\n\nsubmission.to_csv(\"submission.csv\", index=False)\n\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T16:12:09.289498Z","iopub.execute_input":"2024-12-23T16:12:09.290185Z","iopub.status.idle":"2024-12-23T16:12:51.993994Z","shell.execute_reply.started":"2024-12-23T16:12:09.290154Z","shell.execute_reply":"2024-12-23T16:12:51.993077Z"}},"outputs":[],"execution_count":null}]}