{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"},{"sourceId":10264978,"sourceType":"datasetVersion","datasetId":6350475}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:18:14.712433Z","iopub.execute_input":"2024-12-21T19:18:14.712729Z","iopub.status.idle":"2024-12-21T19:18:15.014632Z","shell.execute_reply.started":"2024-12-21T19:18:14.712706Z","shell.execute_reply":"2024-12-21T19:18:15.013771Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"EDA : in this spet we'll visualize our dataset","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np \nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:18:15.015674Z","iopub.execute_input":"2024-12-21T19:18:15.016081Z","iopub.status.idle":"2024-12-21T19:18:15.695496Z","shell.execute_reply.started":"2024-12-21T19:18:15.016047Z","shell.execute_reply":"2024-12-21T19:18:15.694805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df =  pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:18:15.697296Z","iopub.execute_input":"2024-12-21T19:18:15.697675Z","iopub.status.idle":"2024-12-21T19:18:20.534773Z","shell.execute_reply.started":"2024-12-21T19:18:15.697652Z","shell.execute_reply":"2024-12-21T19:18:20.534048Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_cols = df.select_dtypes(include='number').columns\n\nnum_cols = 3  # Number of columns in the grid\nnum_rows = (len(numerical_cols) + num_cols - 1) // num_cols\n\nfig, axes = plt.subplots(num_rows, num_cols, figsize=(15, 5 * num_rows), constrained_layout=True)\naxes = axes.flatten()\n\nfor i, col in enumerate(numerical_cols):\n    sns.histplot(df[col], ax=axes[i], kde=True)\n    axes[i].set_title(col)\n\n    ax_box = axes[i].inset_axes([0.2, -0.3, 0.6, 0.2])  # [x, y, width, height]\n    sns.boxplot(x=df[col], ax=ax_box, orient='h')\n    ax_box.set(xlabel='')\nfor j in range(len(numerical_cols), len(axes)):\n    fig.delaxes(axes[j])\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:18:20.535909Z","iopub.execute_input":"2024-12-21T19:18:20.536167Z","iopub.status.idle":"2024-12-21T19:19:08.290871Z","shell.execute_reply.started":"2024-12-21T19:18:20.536144Z","shell.execute_reply":"2024-12-21T19:19:08.290049Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As we see this are outlier in Premiun Amount, Annual Incomes","metadata":{}},{"cell_type":"code","source":"df['day'] = pd.to_datetime(df['Policy Start Date']).dt.day\ndf['month'] = pd.to_datetime(df['Policy Start Date']).dt.month\ndf['year'] = pd.to_datetime(df['Policy Start Date']).dt.year\ndf['day_of_week'] = pd.to_datetime(df['Policy Start Date']).dt.day_of_week","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:08.291751Z","iopub.execute_input":"2024-12-21T19:19:08.291964Z","iopub.status.idle":"2024-12-21T19:19:09.868587Z","shell.execute_reply.started":"2024-12-21T19:19:08.291946Z","shell.execute_reply":"2024-12-21T19:19:09.867905Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.lineplot(x='day', y='Premium Amount', data=df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:09.869361Z","iopub.execute_input":"2024-12-21T19:19:09.869692Z","iopub.status.idle":"2024-12-21T19:19:17.571909Z","shell.execute_reply.started":"2024-12-21T19:19:09.86966Z","shell.execute_reply":"2024-12-21T19:19:17.571043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.lineplot(x='month', y='Premium Amount', data=df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:17.572752Z","iopub.execute_input":"2024-12-21T19:19:17.573064Z","iopub.status.idle":"2024-12-21T19:19:25.29229Z","shell.execute_reply.started":"2024-12-21T19:19:17.573034Z","shell.execute_reply":"2024-12-21T19:19:25.291461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.boxplot(x='month', y='Premium Amount', data=df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:25.293136Z","iopub.execute_input":"2024-12-21T19:19:25.293463Z","iopub.status.idle":"2024-12-21T19:19:25.809482Z","shell.execute_reply.started":"2024-12-21T19:19:25.293433Z","shell.execute_reply":"2024-12-21T19:19:25.808531Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.lineplot(x='year', y='Premium Amount', data=df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:25.812368Z","iopub.execute_input":"2024-12-21T19:19:25.812601Z","iopub.status.idle":"2024-12-21T19:19:34.135777Z","shell.execute_reply.started":"2024-12-21T19:19:25.81258Z","shell.execute_reply":"2024-12-21T19:19:34.134933Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Preprocessing : in this step we'll fix missing values","metadata":{}},{"cell_type":"code","source":"df =  pd.read_csv('/kaggle/input/insurance-dataset-with-original-original-dataset/train.csv')\ntest  = pd.read_csv('/kaggle/input/insurance-dataset-with-original-original-dataset/test.csv')\noriginal = pd.read_csv('/kaggle/input/insurance-dataset-with-original-original-dataset/Insurance Premium Prediction Dataset.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:34.13736Z","iopub.execute_input":"2024-12-21T19:19:34.137662Z","iopub.status.idle":"2024-12-21T19:19:42.561669Z","shell.execute_reply.started":"2024-12-21T19:19:34.137631Z","shell.execute_reply":"2024-12-21T19:19:42.560946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original['id']= 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:42.562344Z","iopub.execute_input":"2024-12-21T19:19:42.562552Z","iopub.status.idle":"2024-12-21T19:19:42.566967Z","shell.execute_reply.started":"2024-12-21T19:19:42.562533Z","shell.execute_reply":"2024-12-21T19:19:42.566275Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:42.567769Z","iopub.execute_input":"2024-12-21T19:19:42.568032Z","iopub.status.idle":"2024-12-21T19:19:42.583616Z","shell.execute_reply.started":"2024-12-21T19:19:42.567999Z","shell.execute_reply":"2024-12-21T19:19:42.58282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:42.584435Z","iopub.execute_input":"2024-12-21T19:19:42.584664Z","iopub.status.idle":"2024-12-21T19:19:42.614871Z","shell.execute_reply.started":"2024-12-21T19:19:42.584637Z","shell.execute_reply":"2024-12-21T19:19:42.614167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in df.columns:\n    missing = df[col].isna().sum()\n    if missing != 0:\n        print(f'col : {col}  missing : {missing} {missing/df.shape[0]}% dtypes:{df[col].dtype}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:42.615608Z","iopub.execute_input":"2024-12-21T19:19:42.615802Z","iopub.status.idle":"2024-12-21T19:19:43.128391Z","shell.execute_reply.started":"2024-12-21T19:19:42.615785Z","shell.execute_reply":"2024-12-21T19:19:43.127737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"original['Policy Start Date'] = pd.to_datetime(original['Policy Start Date']).dt.date\noriginal['Policy Start Date'] = pd.to_datetime(original['Policy Start Date']).dt.date","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:43.129115Z","iopub.execute_input":"2024-12-21T19:19:43.12932Z","iopub.status.idle":"2024-12-21T19:19:43.36246Z","shell.execute_reply.started":"2024-12-21T19:19:43.129303Z","shell.execute_reply":"2024-12-21T19:19:43.36178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imputer_methode = {\n    \"Age\" : \"mean\",\n    \"Annual Income\" :\"mean\",\n    \"Marital Status\" : \"mode\",\n    \"Number of Dependents\": 0,\n    \"Occupation\":\"Unemployed\",\n    \"Health Score\":\"mean\",\n    \"Previous Claims\":0,\n    \"Vehicle Age\":\"mean\",\n    \"Credit Score\":\"mean\",\n    \"Insurance Duration\" :\"mean\",\n    \"Customer Feedback\" :  \"Average\"\n}\ndef processing(data):\n    df = data.copy()\n    for col in imputer_methode:\n        df[col] = df[col].fillna(df[col].mean()) if imputer_methode[col]==\"mean\" else df[col].fillna(df[col].mode()[0]) if imputer_methode[col] ==\"mode\" else df[col].fillna(imputer_methode[col])\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date']).dt.date\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:43.363159Z","iopub.execute_input":"2024-12-21T19:19:43.363368Z","iopub.status.idle":"2024-12-21T19:19:43.368663Z","shell.execute_reply.started":"2024-12-21T19:19:43.36335Z","shell.execute_reply":"2024-12-21T19:19:43.367766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df =  pd.concat([df, original])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:43.369688Z","iopub.execute_input":"2024-12-21T19:19:43.370001Z","iopub.status.idle":"2024-12-21T19:19:43.557956Z","shell.execute_reply.started":"2024-12-21T19:19:43.369949Z","shell.execute_reply":"2024-12-21T19:19:43.557289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:43.558716Z","iopub.execute_input":"2024-12-21T19:19:43.559016Z","iopub.status.idle":"2024-12-21T19:19:43.577089Z","shell.execute_reply.started":"2024-12-21T19:19:43.55897Z","shell.execute_reply":"2024-12-21T19:19:43.576214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df2 = processing(df)\ntest2 = processing(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:43.577956Z","iopub.execute_input":"2024-12-21T19:19:43.578247Z","iopub.status.idle":"2024-12-21T19:19:46.226413Z","shell.execute_reply.started":"2024-12-21T19:19:43.578224Z","shell.execute_reply":"2024-12-21T19:19:46.225498Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df2.dropna(inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:46.227202Z","iopub.execute_input":"2024-12-21T19:19:46.227446Z","iopub.status.idle":"2024-12-21T19:19:47.163957Z","shell.execute_reply.started":"2024-12-21T19:19:46.227425Z","shell.execute_reply":"2024-12-21T19:19:47.163037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df2.to_csv('preprocessing_train.csv', index=False)\ntest2.to_csv('preprocessing_test.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:19:47.164974Z","iopub.execute_input":"2024-12-21T19:19:47.165304Z","iopub.status.idle":"2024-12-21T19:20:05.779219Z","shell.execute_reply.started":"2024-12-21T19:19:47.165274Z","shell.execute_reply":"2024-12-21T19:20:05.778563Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Model trainning","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.base import  TransformerMixin\nfrom sklearn.preprocessing import  MinMaxScaler, StandardScaler\nfrom sklearn.preprocessing import OneHotEncoder, LabelEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom collections import defaultdict\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import cross_val_score\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom catboost import CatBoostRegressor\nimport random\nfrom shapely.wkt import loads\nfrom sklearn.metrics import mean_squared_log_error","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:05.779847Z","iopub.execute_input":"2024-12-21T19:20:05.780099Z","iopub.status.idle":"2024-12-21T19:20:09.29628Z","shell.execute_reply.started":"2024-12-21T19:20:05.780078Z","shell.execute_reply":"2024-12-21T19:20:09.295561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pd.set_option('display.max_columns', None)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:09.296991Z","iopub.execute_input":"2024-12-21T19:20:09.297449Z","iopub.status.idle":"2024-12-21T19:20:09.301102Z","shell.execute_reply.started":"2024-12-21T19:20:09.297427Z","shell.execute_reply":"2024-12-21T19:20:09.300079Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Feature engineering","metadata":{}},{"cell_type":"code","source":"df  = pd.read_csv('preprocessing_train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:09.301763Z","iopub.execute_input":"2024-12-21T19:20:09.302112Z","iopub.status.idle":"2024-12-21T19:20:12.981249Z","shell.execute_reply.started":"2024-12-21T19:20:09.302077Z","shell.execute_reply":"2024-12-21T19:20:12.980534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['income_per_age'] =  df['Annual Income'] / df['Age']\ndf['health_score_per_age'] =  df['Health Score'] / df['Age']\ndf['vehicle_age_per_insurance_duration'] =  df['Vehicle Age'] / (df['Insurance Duration']+ 1)\ndf['previous_claims_per_insurance_duration'] = df['Previous Claims'] / (df['Insurance Duration']+ 1)\ndf['health_score_per_income'] = df['Health Score'] / (df['Annual Income']+1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:12.981931Z","iopub.execute_input":"2024-12-21T19:20:12.982164Z","iopub.status.idle":"2024-12-21T19:20:13.018874Z","shell.execute_reply.started":"2024-12-21T19:20:12.982144Z","shell.execute_reply":"2024-12-21T19:20:13.018037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def circular_features_time(df, date_column='Policy Start Date'):\n    df[date_column] = pd.to_datetime(df[date_column])\n    \n    df['day_of_week'] = df[date_column].dt.dayofweek\n    df['day_sin'] = np.sin(2 * np.pi * df['day_of_week'] / 7)\n    df['day_cos'] = np.cos(2 * np.pi * df['day_of_week'] / 7)\n    \n    df['day_of_year'] = df[date_column].dt.dayofyear\n    df['year_day_sin'] = np.sin(2 * np.pi * df['day_of_year'] / 365)\n    df['year_day_cos'] = np.cos(2 * np.pi * df['day_of_year'] / 365)\n    \n    df['month'] = df[date_column].dt.month\n    df['month_sin'] = np.sin(2 * np.pi * df['month'] / 12)\n    df['month_cos'] = np.cos(2 * np.pi * df['month'] / 12)\n    \n    df[\"seconds_since_1970\"] = df[date_column].astype(\"int64\") // 10**9\n\n    return df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:13.019738Z","iopub.execute_input":"2024-12-21T19:20:13.019963Z","iopub.status.idle":"2024-12-21T19:20:13.025021Z","shell.execute_reply.started":"2024-12-21T19:20:13.019942Z","shell.execute_reply":"2024-12-21T19:20:13.024354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = circular_features_time(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:13.028254Z","iopub.execute_input":"2024-12-21T19:20:13.028462Z","iopub.status.idle":"2024-12-21T19:20:13.714716Z","shell.execute_reply.started":"2024-12-21T19:20:13.028445Z","shell.execute_reply":"2024-12-21T19:20:13.714057Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Encoding","metadata":{}},{"cell_type":"code","source":"education_mapping ={\n    \"High School\":0,\n    \"Bachelor's\":1,\n    \"Master's\":2,\n    \"PhD\" : 3\n}\n\npolicy_type_mapping ={\n    \"Basic\" : 0,\n    \"Comprehensive\" : 1,\n    \"Premium\" : 2\n}\n\ncustomer_feedback_mapping={\n    \"Poor\" :-1,\n    \"Average\" :0,\n    \"Good\": 1\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:13.715676Z","iopub.execute_input":"2024-12-21T19:20:13.715886Z","iopub.status.idle":"2024-12-21T19:20:13.71996Z","shell.execute_reply.started":"2024-12-21T19:20:13.715866Z","shell.execute_reply":"2024-12-21T19:20:13.718962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"Education Level\"] = df['Education Level'].apply(lambda x: x if x in  education_mapping else \"High School\")\ndf['Policy Type'] = df['Policy Type'].apply(lambda x :x if x in policy_type_mapping else \"Comprehensive\")\ndf['Customer Feedback'] = df['Customer Feedback'].apply(lambda x : x if x in customer_feedback_mapping else \"Average\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:13.720719Z","iopub.execute_input":"2024-12-21T19:20:13.720912Z","iopub.status.idle":"2024-12-21T19:20:14.288767Z","shell.execute_reply.started":"2024-12-21T19:20:13.720893Z","shell.execute_reply":"2024-12-21T19:20:14.288063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scale_col =[\"Age\", \"Annual Income\", \"Health Score\", \"Vehicle Age\", \"Credit Score\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:14.289629Z","iopub.execute_input":"2024-12-21T19:20:14.28987Z","iopub.status.idle":"2024-12-21T19:20:14.293655Z","shell.execute_reply.started":"2024-12-21T19:20:14.289849Z","shell.execute_reply":"2024-12-21T19:20:14.29283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Custom_Scaler(TransformerMixin):\n    def __init__(self, except_col=[], cols=[], strategy=\"MinMax\"):\n        super().__init__()\n        self.except_col=except_col\n        self.cols = cols if cols else []\n        self.strategy = strategy\n\n    def fit(self, df):\n        numerical_cols = df.select_dtypes(include=[np.number]).columns\n        final_col =  numerical_cols.difference(self.except_col)\n        self.col  =  final_col if not self.cols else self.cols\n        self.scaler = MinMaxScaler().fit(df[self.col]) if self.strategy==\"MinMax\" else StandardScaler().fit(df[self.col])\n        return self\n    \n    def transform(self, data):\n        df =data.copy()\n        scaler_data =  self.scaler.transform(df[self.col])\n        scaler_data_df = pd.DataFrame(scaler_data, columns=self.col, index=df.index)\n        others_cols  =  df.columns.difference(self.col)\n        return pd.concat([scaler_data_df, df[others_cols]], axis='columns')\n\nclass CustomOneHotEncoder(TransformerMixin):\n    def __init__(self, except_col=[], cols=[]):\n        super().__init__()\n        self.except_col=except_col\n        self.cols = cols if cols else []\n\n    def fit(self, data):\n        df =  data.copy()\n        cat_col =  df.select_dtypes(exclude=[np.number]).columns\n        final_col =  cat_col.difference(self.except_col)\n        self.col  =  final_col if not self.cols else self.cols\n        preprocessor = ColumnTransformer(\n            transformers=[\n                ('cat', OneHotEncoder(handle_unknown='infrequent_if_exist'), self.col)\n            ],\n            remainder='passthrough'  # To keep other columns unchanged\n        )\n        self.preprocessor =  preprocessor\n        self.preprocessor.fit(df[self.col])\n        return self\n\n    def transform(self, data):\n        df =  data.copy()\n        final_data_encoded =  self.preprocessor.transform(df[self.col])\n        feature_names = (self.preprocessor\n                        .named_transformers_['cat']\n                        .get_feature_names_out(self.col))\n        final_data_encoded_df = pd.DataFrame(final_data_encoded.toarray() if type(final_data_encoded)!=np.ndarray else final_data_encoded, columns=feature_names, index=df.index)\n        others_col =  df.columns.difference(self.col)\n        final_df  = pd.concat([df[others_col], final_data_encoded_df], axis='columns')\n        return final_df\n\nclass MultiColumnLabelEncoder(TransformerMixin):\n    def __init__(self, except_col=[]):\n        self.except_col = except_col\n        self.label_encoders = defaultdict(LabelEncoder)\n\n    def fit(self,X , y=None):\n        df  = X.copy()\n        cat_col =  df.select_dtypes(exclude=[np.number]).columns\n        final_col =  cat_col.difference(self.except_col)\n        self.columns = final_col\n        for col in self.columns:\n            self.label_encoders[col]\n            self.label_encoders[col].fit(df[col])\n        return self\n\n    def transform(self, X):\n        X_copy = X.copy()  # To avoid modifying the original dataframe\n        for col in self.columns:\n            X_copy[col] = X_copy[col].map(lambda s: '<unknown>' if s not in self.label_encoders[col].classes_ else s)\n            self.label_encoders[col].classes_ = np.append(self.label_encoders[col].classes_, '<unknown>')\n            X_copy[col] = self.label_encoders[col].transform(X_copy[col])\n        return X_copy\n\n    def inverse_transform(self, X):\n        X_copy = X.copy()  # To avoid modifying the original dataframe\n        for col in self.columns:\n            X_copy[col] = self.label_encoders[col].inverse_transform(X_copy[col])\n        return X_copy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:14.294455Z","iopub.execute_input":"2024-12-21T19:20:14.29476Z","iopub.status.idle":"2024-12-21T19:20:14.309243Z","shell.execute_reply.started":"2024-12-21T19:20:14.294731Z","shell.execute_reply":"2024-12-21T19:20:14.308549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_transform  = df.drop(['id', 'Policy Start Date', 'Premium Amount'],axis='columns')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:14.309894Z","iopub.execute_input":"2024-12-21T19:20:14.310131Z","iopub.status.idle":"2024-12-21T19:20:14.558033Z","shell.execute_reply.started":"2024-12-21T19:20:14.310111Z","shell.execute_reply":"2024-12-21T19:20:14.557354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pipe  = Pipeline([('scaler', Custom_Scaler(except_col=['Premium Amount'], strategy=\"Std\"), ),  ('label_encoding', MultiColumnLabelEncoder(except_col={\"Policy Start Date\"}))])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:14.558913Z","iopub.execute_input":"2024-12-21T19:20:14.559175Z","iopub.status.idle":"2024-12-21T19:20:14.563035Z","shell.execute_reply.started":"2024-12-21T19:20:14.559153Z","shell.execute_reply":"2024-12-21T19:20:14.562236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transform_data  = pipe.fit_transform(df_transform)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:20:14.56382Z","iopub.execute_input":"2024-12-21T19:20:14.564106Z","iopub.status.idle":"2024-12-21T19:21:37.412164Z","shell.execute_reply.started":"2024-12-21T19:20:14.564085Z","shell.execute_reply":"2024-12-21T19:21:37.411465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = transform_data\ny = df['Premium Amount']\ny_log =  np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:21:37.412872Z","iopub.execute_input":"2024-12-21T19:21:37.413118Z","iopub.status.idle":"2024-12-21T19:21:37.420057Z","shell.execute_reply.started":"2024-12-21T19:21:37.413096Z","shell.execute_reply":"2024-12-21T19:21:37.419289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import optuna\n# import xgboost as xgb\n# from sklearn.model_selection import train_test_split\n# from sklearn.metrics import mean_squared_error, accuracy_score\n\n\n# def objective(trial):\n#     params = {\n#         \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 150),\n#         \"max_depth\": trial.suggest_int(\"max_depth\", 3, 10),\n#         \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3),\n#         \"n_estimators\": trial.suggest_int(\"n_estimators\", 100, 1000),\n#         \"min_child_samples\": trial.suggest_int(\"min_child_samples\", 5, 50),\n#         \"min_child_weight\": trial.suggest_float(\"min_child_weight\", 1e-3, 1),\n#         \"subsample\": trial.suggest_float(\"subsample\", 0.5, 1.0),\n#         \"colsample_bytree\": trial.suggest_float(\"colsample_bytree\", 0.5, 1.0),\n#         \"lambda_l1\": trial.suggest_float(\"lambda_l1\", 0, 10),\n#         \"lambda_l2\": trial.suggest_float(\"lambda_l2\", 0, 10),\n#         \"verbose\":-1,\n#     }\n#     from sklearn.model_selection import KFold\n#     skf = KFold(n_splits=5, shuffle=True, random_state=42)\n#     scores = []\n#     for train_index, test_index in skf.split(X, y_log):\n#         X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n#         y_train, y_test = y_log.iloc[train_index], y_log.iloc[test_index]\n#         model = LGBMRegressor(**params)\n#         # Entraînement\n#         model.fit(X_train, y_train)\n        \n#         # Prédiction\n#         y_pred = model.predict(X_test)\n        \n#         # Calcul du score\n#         score = np.sqrt(mean_squared_log_error(y_test, y_pred))\n#         print(score)\n#         scores.append(score)\n#     # Afficher les résultats\n#     print(f\"Scores pour chaque fold : {scores}\")\n#     print(f\"Score moyen : {np.mean(scores):.4f}±{np.std(scores)}\")\n#     return np.mean(scores) + np.std(scores)\n\n\n# study = optuna.create_study(direction=\"minimize\")\n# study.optimize(objective, n_trials=10)\n\n# print(study.best_params)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:25:11.538522Z","iopub.execute_input":"2024-12-21T19:25:11.53886Z","iopub.status.idle":"2024-12-21T19:25:11.543406Z","shell.execute_reply.started":"2024-12-21T19:25:11.538829Z","shell.execute_reply":"2024-12-21T19:25:11.542432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"LGBM_params={'num_leaves': 142, 'max_depth': 10, 'learning_rate': 0.05433080564731205, 'n_estimators': 300, 'min_child_samples': 47, 'min_child_weight': 0.5776963031165517, 'subsample': 0.7467693623684808, 'colsample_bytree': 0.8760255315807451, 'lambda_l1': 8.138155357936343, 'lambda_l2': 2.9299722725051716, \"verbose\" : -1}\nparams2 = {'num_leaves': 83, 'max_depth': 8, 'learning_rate': 0.09073000869009397, 'n_estimators': 195, 'min_child_samples': 11, 'min_child_weight': 0.7796942681873954, 'subsample': 0.734370855470486, 'colsample_bytree': 0.6708856171350388, 'lambda_l1': 7.8619472190057955, 'lambda_l2': 3.6901814657933363, 'verbose': -1}\nparams3 ={'num_leaves': 21, 'max_depth': 10, 'learning_rate': 0.03923623452073841, 'n_estimators': 425, 'min_child_samples': 31, 'min_child_weight': 0.7898659573828833, 'subsample': 0.5510442504764471, 'colsample_bytree': 0.7080786590069881, 'lambda_l1': 9.959425048359014, 'lambda_l2': 7.217836840869001, \"verbose\": -1}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:32:38.319273Z","iopub.execute_input":"2024-12-21T19:32:38.319606Z","iopub.status.idle":"2024-12-21T19:32:38.325064Z","shell.execute_reply.started":"2024-12-21T19:32:38.319578Z","shell.execute_reply":"2024-12-21T19:32:38.324084Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"test  = pd.read_csv('preprocessing_test.csv')","metadata":{}},{"cell_type":"code","source":"test['income_per_age'] =  test['Annual Income'] / test['Age']\ntest['health_score_per_age'] =  test['Health Score'] / test['Age']\ntest['vehicle_age_per_insurance_duration'] =  test['Vehicle Age'] / (test['Insurance Duration']+ 1)\ntest['previous_claims_per_insurance_duration'] = test['Previous Claims'] / (test['Insurance Duration']+ 1)\ntest['health_score_per_income'] = test['Health Score'] / (test['Annual Income']+1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:32:38.326327Z","iopub.execute_input":"2024-12-21T19:32:38.326716Z","iopub.status.idle":"2024-12-21T19:32:38.359668Z","shell.execute_reply.started":"2024-12-21T19:32:38.326682Z","shell.execute_reply":"2024-12-21T19:32:38.359033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test[\"Education Level\"] = test['Education Level'].apply(lambda x: x if x in  education_mapping else \"High School\")\ntest['Policy Type'] = test['Policy Type'].apply(lambda x :x if x in policy_type_mapping else \"Comprehensive\")\ntest['Customer Feedback'] = test['Customer Feedback'].apply(lambda x : x if x in customer_feedback_mapping else \"Average\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:32:38.361036Z","iopub.execute_input":"2024-12-21T19:32:38.361257Z","iopub.status.idle":"2024-12-21T19:32:38.697912Z","shell.execute_reply.started":"2024-12-21T19:32:38.361237Z","shell.execute_reply":"2024-12-21T19:32:38.696934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test   = circular_features_time(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:32:38.69917Z","iopub.execute_input":"2024-12-21T19:32:38.699492Z","iopub.status.idle":"2024-12-21T19:32:38.950698Z","shell.execute_reply.started":"2024-12-21T19:32:38.699462Z","shell.execute_reply":"2024-12-21T19:32:38.949815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_transform = pipe.transform(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:33:27.478744Z","iopub.execute_input":"2024-12-21T19:33:27.479097Z","iopub.status.idle":"2024-12-21T19:34:13.353032Z","shell.execute_reply.started":"2024-12-21T19:33:27.479062Z","shell.execute_reply":"2024-12-21T19:34:13.352068Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_transform.drop(['id', 'Policy Start Date'], axis='columns', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:34:13.354223Z","iopub.execute_input":"2024-12-21T19:34:13.354572Z","iopub.status.idle":"2024-12-21T19:34:13.432123Z","shell.execute_reply.started":"2024-12-21T19:34:13.354539Z","shell.execute_reply":"2024-12-21T19:34:13.431236Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test_submit = test_transform","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:48:21.45079Z","iopub.execute_input":"2024-12-21T19:48:21.451098Z","iopub.status.idle":"2024-12-21T19:48:21.454817Z","shell.execute_reply.started":"2024-12-21T19:48:21.451074Z","shell.execute_reply":"2024-12-21T19:48:21.454043Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nskf = KFold(n_splits=5, shuffle=True, random_state=42)\nscores1 = []\nscores2 = []\noof_preds_model1 = np.zeros(len(X))\noof_preds_model2 = np.zeros(len(X))\nprediction1=[]\nprediction2=[]\nfor train_index, test_index in skf.split(X, y_log):\n    X_train, X_test = X.iloc[train_index], X.iloc[test_index]\n    y_train, y_test = y_log.iloc[train_index], y_log.iloc[test_index]\n    model1 = LGBMRegressor(**LGBM_params)\n    model2 = LGBMRegressor(**params3)\n    # Entraînement\n    model1.fit(X_train, y_train)\n    model2.fit(X_train, y_train)\n    # Prédiction\n    y_pred1 = model1.predict(X_test)\n    y_pred2 = model2.predict(X_test)\n    \n    # Calcul du score\n    score1 = np.sqrt(mean_squared_log_error(y_test, y_pred1))\n    print(f\"score model 1 : {score1}\")\n    scores1.append(score1)\n    oof_preds_model1[test_index] = model1.predict(X_test)\n    oof_preds_model2[test_index] = model2.predict(X_test)\n    prediction1.append(model1.predict(X_test_submit))\n    prediction2.append(model2.predict(X_test_submit))\n    score2 = np.sqrt(mean_squared_log_error(y_test, y_pred2))\n    print(f\"score model 2 :  {score2}\")\n    scores2.append(score2)\n# Afficher les résultats\nprint(f\"Scores pour chaque fold model 1 : {scores1}\")\nprint(f\"Score moyen model 2: {np.mean(scores1):.4f}±{np.std(scores1)}\")\n\nprint(f\"Scores pour chaque fold model 2 : {scores2}\")\nprint(f\"Score moyen model 2: {np.mean(scores2):.4f}±{np.std(scores2)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:48:23.362408Z","iopub.execute_input":"2024-12-21T19:48:23.362692Z","iopub.status.idle":"2024-12-21T19:55:04.008505Z","shell.execute_reply.started":"2024-12-21T19:48:23.362668Z","shell.execute_reply":"2024-12-21T19:55:04.007621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stacking= pd.DataFrame({\n    \"model1\" : oof_preds_model1,\n    \"model2\" : oof_preds_model2,\n    'real' : y_log\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:57:42.772778Z","iopub.execute_input":"2024-12-21T19:57:42.773137Z","iopub.status.idle":"2024-12-21T19:57:42.785311Z","shell.execute_reply.started":"2024-12-21T19:57:42.773107Z","shell.execute_reply":"2024-12-21T19:57:42.784583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stacking_test=pd.DataFrame({\n    \"model1\" : np.mean(np.array(prediction1), axis=0),\n    \"model2\" : np.mean(prediction2, axis=0),\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:59:17.917316Z","iopub.execute_input":"2024-12-21T19:59:17.917622Z","iopub.status.idle":"2024-12-21T19:59:17.945765Z","shell.execute_reply.started":"2024-12-21T19:59:17.917598Z","shell.execute_reply":"2024-12-21T19:59:17.945063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import HistGradientBoostingRegressor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:57:51.791646Z","iopub.execute_input":"2024-12-21T19:57:51.791969Z","iopub.status.idle":"2024-12-21T19:57:51.795544Z","shell.execute_reply.started":"2024-12-21T19:57:51.791941Z","shell.execute_reply":"2024-12-21T19:57:51.794572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = HistGradientBoostingRegressor()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:57:53.1931Z","iopub.execute_input":"2024-12-21T19:57:53.193443Z","iopub.status.idle":"2024-12-21T19:57:53.197658Z","shell.execute_reply.started":"2024-12-21T19:57:53.193412Z","shell.execute_reply":"2024-12-21T19:57:53.196725Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(stacking.drop(\"real\", axis='columns'), stacking['real'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:58:03.702289Z","iopub.execute_input":"2024-12-21T19:58:03.702588Z","iopub.status.idle":"2024-12-21T19:58:06.901508Z","shell.execute_reply.started":"2024-12-21T19:58:03.702564Z","shell.execute_reply":"2024-12-21T19:58:06.900477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predict = model.predict(stacking_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:59:22.540813Z","iopub.execute_input":"2024-12-21T19:59:22.541119Z","iopub.status.idle":"2024-12-21T19:59:23.521537Z","shell.execute_reply.started":"2024-12-21T19:59:22.541095Z","shell.execute_reply":"2024-12-21T19:59:23.520769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission  = pd.DataFrame([], columns=['id', 'Premium Amount'])\nsubmission.id  =  test.id\nsubmission['Premium Amount']  =np.expm1(predict)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T19:59:40.011892Z","iopub.execute_input":"2024-12-21T19:59:40.012249Z","iopub.status.idle":"2024-12-21T19:59:40.055292Z","shell.execute_reply.started":"2024-12-21T19:59:40.012218Z","shell.execute_reply":"2024-12-21T19:59:40.054147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-21T20:00:20.915646Z","iopub.execute_input":"2024-12-21T20:00:20.915938Z","iopub.status.idle":"2024-12-21T20:00:22.214312Z","shell.execute_reply.started":"2024-12-21T20:00:20.915914Z","shell.execute_reply":"2024-12-21T20:00:22.213266Z"}},"outputs":[],"execution_count":null}]}