{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Overview","metadata":{}},{"cell_type":"code","source":"# import library\nimport numpy as np\nimport numpy.random as random\nimport scipy as sp\nfrom pandas import Series, DataFrame\nimport pandas as pd\n\n# Visualization Library\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sns\n%matplotlib inline\n\n# Machine Learning Model Library\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nimport xgboost as xgb\nimport lightgbm as lgb\nimport catboost as cb\nfrom catboost import Pool\nfrom catboost import CatBoostRegressor\nfrom catboost import CatBoostClassifier\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.ensemble import RandomForestRegressor\n\n# Evaluation Library\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.metrics import mean_squared_error\n\n# others\nfrom sklearn import base\nfrom sklearn.preprocessing import StandardScaler\nimport category_encoders as ce\nimport warnings\nwarnings.filterwarnings('ignore')\n%precision 3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:14.803357Z","iopub.execute_input":"2024-12-31T14:13:14.803741Z","iopub.status.idle":"2024-12-31T14:13:21.91682Z","shell.execute_reply.started":"2024-12-31T14:13:14.803708Z","shell.execute_reply":"2024-12-31T14:13:21.915711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# directory path\ndir_path = '/kaggle/input/playground-series-s4e12/'\n\n# read train.csv and test.csv\ntrain_df = pd.read_csv(dir_path + 'train.csv')\ntest_df = pd.read_csv(dir_path + 'test.csv')\n\n# connect train and test\ndf = pd.concat([train_df, test_df], ignore_index=True)\n\nprint(f'train data: {train_df.shape}')\nprint(f'test data: {test_df.shape}')\nprint(f'connect data: {df.shape}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:21.917908Z","iopub.execute_input":"2024-12-31T14:13:21.918627Z","iopub.status.idle":"2024-12-31T14:13:33.531297Z","shell.execute_reply.started":"2024-12-31T14:13:21.918557Z","shell.execute_reply":"2024-12-31T14:13:33.530356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#data types\ndf.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:33.532488Z","iopub.execute_input":"2024-12-31T14:13:33.532974Z","iopub.status.idle":"2024-12-31T14:13:33.555546Z","shell.execute_reply.started":"2024-12-31T14:13:33.532935Z","shell.execute_reply":"2024-12-31T14:13:33.554118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert Policy Start Date from object to datetime.\ndf['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:33.556681Z","iopub.execute_input":"2024-12-31T14:13:33.557047Z","iopub.status.idle":"2024-12-31T14:13:34.361872Z","shell.execute_reply.started":"2024-12-31T14:13:33.557019Z","shell.execute_reply":"2024-12-31T14:13:34.360749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Grouping columns of the same data type\nnumber_columns = df.select_dtypes(include='float')\ncategory_columns = df.select_dtypes(include='object')\ndatetime_columns = df.select_dtypes(include='datetime')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:34.362963Z","iopub.execute_input":"2024-12-31T14:13:34.363362Z","iopub.status.idle":"2024-12-31T14:13:35.456113Z","shell.execute_reply.started":"2024-12-31T14:13:34.363332Z","shell.execute_reply":"2024-12-31T14:13:35.455025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"number_columns.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:35.45919Z","iopub.execute_input":"2024-12-31T14:13:35.459516Z","iopub.status.idle":"2024-12-31T14:13:35.483158Z","shell.execute_reply.started":"2024-12-31T14:13:35.459487Z","shell.execute_reply":"2024-12-31T14:13:35.482026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(6,4))\nsns.histplot(df['Premium Amount'], kde=False, bins=50)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:35.484966Z","iopub.execute_input":"2024-12-31T14:13:35.485337Z","iopub.status.idle":"2024-12-31T14:13:37.316227Z","shell.execute_reply.started":"2024-12-31T14:13:35.485303Z","shell.execute_reply":"2024-12-31T14:13:37.314845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"category_columns.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:37.317451Z","iopub.execute_input":"2024-12-31T14:13:37.317898Z","iopub.status.idle":"2024-12-31T14:13:37.331776Z","shell.execute_reply.started":"2024-12-31T14:13:37.317855Z","shell.execute_reply":"2024-12-31T14:13:37.33067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"datetime_columns.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:37.332874Z","iopub.execute_input":"2024-12-31T14:13:37.333209Z","iopub.status.idle":"2024-12-31T14:13:37.353086Z","shell.execute_reply.started":"2024-12-31T14:13:37.333181Z","shell.execute_reply":"2024-12-31T14:13:37.352086Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Missing","metadata":{}},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:37.35407Z","iopub.execute_input":"2024-12-31T14:13:37.354411Z","iopub.status.idle":"2024-12-31T14:13:38.301088Z","shell.execute_reply.started":"2024-12-31T14:13:37.354369Z","shell.execute_reply":"2024-12-31T14:13:38.300093Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Before filling in the missing values, a new feature is created with the data with missing values as 1 and the data without missing values as 0.","metadata":{}},{"cell_type":"code","source":"# create new features\ndf['Annual Income isnull'] = df.isnull()['Annual Income'].map({True: 1, False: 0})\ndf['Customer Feedback isnull'] = df.isnull()['Customer Feedback'].map({True: 1, False: 0})\ndf['Health Score isnull'] = df.isnull()['Health Score'].map({True: 1, False: 0})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:38.301954Z","iopub.execute_input":"2024-12-31T14:13:38.302239Z","iopub.status.idle":"2024-12-31T14:13:41.089341Z","shell.execute_reply.started":"2024-12-31T14:13:38.302216Z","shell.execute_reply":"2024-12-31T14:13:41.088327Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Missing values in categorical columns are filled with Unknown.\nPrevious Claims are filled by 0, and other missing values are complemented by the median.","metadata":{}},{"cell_type":"code","source":"# category columns\nfor cat_col in category_columns:\n  df[cat_col] = df[cat_col].fillna('Unknown')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:41.090186Z","iopub.execute_input":"2024-12-31T14:13:41.090457Z","iopub.status.idle":"2024-12-31T14:13:42.381109Z","shell.execute_reply.started":"2024-12-31T14:13:41.090433Z","shell.execute_reply":"2024-12-31T14:13:42.379937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Previous Claims\ndf['Previous Claims'] = df['Previous Claims'].fillna(0)\n\n# other numeric data\nnum_feature_columns = number_columns.drop('Premium Amount', axis=1)\nfor num_col in num_feature_columns:\n  df[num_col] = df[num_col].fillna(df[num_col].median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:42.382065Z","iopub.execute_input":"2024-12-31T14:13:42.382454Z","iopub.status.idle":"2024-12-31T14:13:42.917463Z","shell.execute_reply.started":"2024-12-31T14:13:42.382416Z","shell.execute_reply":"2024-12-31T14:13:42.916684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:42.918255Z","iopub.execute_input":"2024-12-31T14:13:42.918508Z","iopub.status.idle":"2024-12-31T14:13:43.871351Z","shell.execute_reply.started":"2024-12-31T14:13:42.918487Z","shell.execute_reply":"2024-12-31T14:13:43.869982Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"# datetime\ndf['Policy Start Year'] = df['Policy Start Date'].dt.year\ndf['Policy Start Month'] = df['Policy Start Date'].dt.month\ndf['Policy Start Day'] = df['Policy Start Date'].dt.day\n\ndf = df.drop('Policy Start Date', axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:43.872365Z","iopub.execute_input":"2024-12-31T14:13:43.872733Z","iopub.status.idle":"2024-12-31T14:13:44.630959Z","shell.execute_reply.started":"2024-12-31T14:13:43.872644Z","shell.execute_reply":"2024-12-31T14:13:44.629878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:44.632038Z","iopub.execute_input":"2024-12-31T14:13:44.632389Z","iopub.status.idle":"2024-12-31T14:13:44.658607Z","shell.execute_reply.started":"2024-12-31T14:13:44.632347Z","shell.execute_reply":"2024-12-31T14:13:44.657645Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Train_data and test_data are used for regression models.","metadata":{}},{"cell_type":"code","source":"train_data = df[~df['Premium Amount'].isnull()]\ntest_data = df[df['Premium Amount'].isnull()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:44.659644Z","iopub.execute_input":"2024-12-31T14:13:44.660034Z","iopub.status.idle":"2024-12-31T14:13:45.014614Z","shell.execute_reply.started":"2024-12-31T14:13:44.659992Z","shell.execute_reply":"2024-12-31T14:13:45.013765Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Train_data2 and test_data2 are used for classification models.","metadata":{}},{"cell_type":"code","source":"train_data2 = train_data.copy()\ntrain_data2['Premium Amount flg-500'] = train_data2['Premium Amount'].map(lambda x: 1 if x <= 500 else 0)\n\ntest_data2 = test_data.copy()\ntest_data2['Premium Amount flg-500'] = ''\n\ntrain_data2.groupby('Premium Amount flg-500').size()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:45.015629Z","iopub.execute_input":"2024-12-31T14:13:45.015942Z","iopub.status.idle":"2024-12-31T14:13:46.662763Z","shell.execute_reply.started":"2024-12-31T14:13:45.015918Z","shell.execute_reply":"2024-12-31T14:13:46.66162Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data2.drop('Premium Amount', axis=1, inplace=True)\ntest_data2.drop('Premium Amount', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:46.663531Z","iopub.execute_input":"2024-12-31T14:13:46.663871Z","iopub.status.idle":"2024-12-31T14:13:46.976478Z","shell.execute_reply.started":"2024-12-31T14:13:46.663846Z","shell.execute_reply":"2024-12-31T14:13:46.975352Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Train_data and test_data before target encoding are used for catboost. Make copies of train_data and test_data","metadata":{}},{"cell_type":"code","source":"encode_train_data = train_data.copy()\nencode_test_data = test_data.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:46.982014Z","iopub.execute_input":"2024-12-31T14:13:46.982329Z","iopub.status.idle":"2024-12-31T14:13:48.076907Z","shell.execute_reply.started":"2024-12-31T14:13:46.982304Z","shell.execute_reply":"2024-12-31T14:13:48.075827Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"I used [this code](https://www.kaggle.com/code/anuragbantu/target-encoding-beginner-s-guide) for target encoding. I am very grateful to whoever made [it](https://www.kaggle.com/code/anuragbantu/target-encoding-beginner-s-guide)!","metadata":{}},{"cell_type":"code","source":"class KFoldTargetEncoderTrain(base.BaseEstimator, base.TransformerMixin):\n\n    def __init__(self,colnames,targetName, n_fold=5, verbosity=True, discardOriginal_col=False):\n        self.colnames = colnames\n        self.targetName = targetName\n        self.n_fold = n_fold\n        self.verbosity = verbosity\n        self.discardOriginal_col = discardOriginal_col\n\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self,X):\n        mean_of_target = X[self.targetName].mean()\n        kf = KFold(n_splits = self.n_fold, shuffle = True, random_state=88)\n        col_mean_name = self.colnames + '_' + 'Kfold_Target_Enc'\n        X[col_mean_name] = np.nan\n\n        for tr_ind, val_ind in kf.split(X):\n            X_tr, X_val = X.iloc[tr_ind], X.iloc[val_ind]\n            X.loc[X.index[val_ind], col_mean_name] = X_val[self.colnames].map(X_tr.groupby(self.colnames)[self.targetName].mean())\n            X[col_mean_name].fillna(mean_of_target, inplace = True)\n\n        if self.verbosity:\n            encoded_feature = X[col_mean_name].values\n            print('Correlation between the new feature, {} and, {} is {}.'.format(col_mean_name,self.targetName, np.corrcoef(X[self.targetName].values,encoded_feature)[0][1]))\n        if self.discardOriginal_col:\n            X = X.drop(self.targetName, axis=1)\n        return X\n\n\nclass TargetEncoderTest(base.BaseEstimator, base.TransformerMixin):\n\n    def __init__(self,train,colNames,encodedName):\n\n        self.train = train\n        self.colNames = colNames\n        self.encodedName = encodedName\n\n    def fit(self, X, y=None):\n        return self\n\n    def transform(self,X):\n        mean =  self.train[[self.colNames, self.encodedName]].groupby(self.colNames).mean().reset_index()\n\n        dd = {}\n        for index, row in mean.iterrows():\n            dd[row[self.colNames]] = row[self.encodedName]\n            X[self.encodedName] = X[self.colNames]\n        X = X.replace({self.encodedName: dd})\n        return X","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:48.080505Z","iopub.execute_input":"2024-12-31T14:13:48.080903Z","iopub.status.idle":"2024-12-31T14:13:48.091838Z","shell.execute_reply.started":"2024-12-31T14:13:48.08087Z","shell.execute_reply":"2024-12-31T14:13:48.090692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for cat_col in category_columns:\n  targetc = KFoldTargetEncoderTrain(cat_col, 'Premium Amount', n_fold=5)\n  encode_train_data = targetc.fit_transform(encode_train_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:13:48.092928Z","iopub.execute_input":"2024-12-31T14:13:48.093332Z","iopub.status.idle":"2024-12-31T14:14:07.966133Z","shell.execute_reply.started":"2024-12-31T14:13:48.093285Z","shell.execute_reply":"2024-12-31T14:14:07.964998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for cat_col in category_columns:\n  test_targetc = TargetEncoderTest(encode_train_data, cat_col, f'{cat_col}_Kfold_Target_Enc')\n  encode_test_data = test_targetc.fit_transform(encode_test_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:07.967209Z","iopub.execute_input":"2024-12-31T14:14:07.96762Z","iopub.status.idle":"2024-12-31T14:14:15.677654Z","shell.execute_reply.started":"2024-12-31T14:14:07.967563Z","shell.execute_reply":"2024-12-31T14:14:15.676559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encode_train_data = encode_train_data.drop(encode_train_data[category_columns.columns], axis=1)\nencode_test_data = encode_test_data.drop(encode_test_data[category_columns.columns], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:15.678703Z","iopub.execute_input":"2024-12-31T14:14:15.679017Z","iopub.status.idle":"2024-12-31T14:14:15.987052Z","shell.execute_reply.started":"2024-12-31T14:14:15.678991Z","shell.execute_reply":"2024-12-31T14:14:15.985969Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"encode_train_data2 and encode_test_data2 are used for classification models.","metadata":{}},{"cell_type":"code","source":"encode_train_data2 = encode_train_data.copy()\nencode_train_data2['Premium Amount flg-500'] = encode_train_data2['Premium Amount'].map(lambda x: 1 if x <= 500 else 0)\n\nencode_test_data2 = encode_test_data.copy()\n\nencode_train_data2.groupby('Premium Amount flg-500').size()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:15.988124Z","iopub.execute_input":"2024-12-31T14:14:15.98855Z","iopub.status.idle":"2024-12-31T14:14:16.851071Z","shell.execute_reply.started":"2024-12-31T14:14:15.988513Z","shell.execute_reply":"2024-12-31T14:14:16.850026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encode_train_data2.drop('Premium Amount', axis=1, inplace=True)\nencode_test_data2.drop('Premium Amount', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:16.852092Z","iopub.execute_input":"2024-12-31T14:14:16.852549Z","iopub.status.idle":"2024-12-31T14:14:16.98727Z","shell.execute_reply.started":"2024-12-31T14:14:16.852454Z","shell.execute_reply":"2024-12-31T14:14:16.986206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encode_data = pd.concat([encode_train_data, encode_test_data])\nencode_data.tail()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:16.988426Z","iopub.execute_input":"2024-12-31T14:14:16.988809Z","iopub.status.idle":"2024-12-31T14:14:17.156895Z","shell.execute_reply.started":"2024-12-31T14:14:16.988779Z","shell.execute_reply":"2024-12-31T14:14:17.155908Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Adding new feature values.","metadata":{}},{"cell_type":"code","source":"# number & number\nencode_data['Annual Income_log'] = np.log1p(encode_data['Annual Income'])\nencode_data['Health Score_Income log_Ratio'] = encode_data['Health Score'] / encode_data['Annual Income_log']\nencode_data['Annual Income_Age_Ratio'] = encode_data['Annual Income'] / encode_data['Age']\nencode_data['Age_Previous Claims_Ratio'] = encode_data['Age'] / (encode_data['Previous Claims'] + 1)\nencode_data['Credit Score_Previous Claims_Ratio'] = encode_data['Credit Score'] / (encode_data['Previous Claims'] + 1)\nencode_data['Health Score_Credit Score_Ratio'] = encode_data['Health Score'] / encode_data['Credit Score']\n\n# catecory & category\nencode_data['Feedback_Marital Status'] = encode_data['Customer Feedback_Kfold_Target_Enc'] * encode_data['Marital Status_Kfold_Target_Enc']\nencode_data['Feedback_Occupation'] = encode_data['Customer Feedback_Kfold_Target_Enc'] * encode_data['Occupation_Kfold_Target_Enc']\nencode_data['Marital Status_Occupation'] = encode_data['Marital Status_Kfold_Target_Enc'] * encode_data['Occupation_Kfold_Target_Enc']\nencode_data['Feedback_Policy Type_Ratio'] = encode_data['Customer Feedback_Kfold_Target_Enc'] / encode_data[ 'Policy Type_Kfold_Target_Enc']\nencode_data['Marital Status_Gender_Ratio'] = encode_data['Marital Status_Kfold_Target_Enc'] / encode_data['Gender_Kfold_Target_Enc']\n\n# number & category\nencode_data['Customer Feedback_Credit Score_Ratio'] = encode_data['Customer Feedback_Kfold_Target_Enc'] / encode_data['Credit Score']\n\nencode_train_data3 = encode_data[~encode_data['Premium Amount'].isnull()]\nencode_test_data3 = encode_data[encode_data['Premium Amount'].isnull()]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:17.157993Z","iopub.execute_input":"2024-12-31T14:14:17.158618Z","iopub.status.idle":"2024-12-31T14:14:17.884508Z","shell.execute_reply.started":"2024-12-31T14:14:17.15855Z","shell.execute_reply":"2024-12-31T14:14:17.883467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"encode_test_data3.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:17.885574Z","iopub.execute_input":"2024-12-31T14:14:17.885994Z","iopub.status.idle":"2024-12-31T14:14:17.912662Z","shell.execute_reply.started":"2024-12-31T14:14:17.885947Z","shell.execute_reply":"2024-12-31T14:14:17.911284Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Select Features","metadata":{}},{"cell_type":"markdown","source":"X_a features contains data before target encoding. It is used to build catboost models.","metadata":{}},{"cell_type":"code","source":"# for CatBoost\nX_a = train_data.drop(['id', 'Premium Amount'], axis=1)\nX_test_a = test_data.drop(['id', 'Premium Amount'], axis=1)\n\n# for CatBoost Classifier\nX_a2 = train_data2.drop(['id', 'Premium Amount flg-500'], axis=1)\nX_test_a2 = test_data2.drop(['id', 'Premium Amount flg-500'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:17.913823Z","iopub.execute_input":"2024-12-31T14:14:17.914249Z","iopub.status.idle":"2024-12-31T14:14:18.685815Z","shell.execute_reply.started":"2024-12-31T14:14:17.914206Z","shell.execute_reply":"2024-12-31T14:14:18.684415Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"X_b contains data after target encoding. Select features with correlation coefficients greater than 0.001, because encode_data has many features.","metadata":{}},{"cell_type":"code","source":"# for lightgbm, xgboost, randomforest\nX_b = encode_train_data.drop(['id', 'Premium Amount'], axis=1)\nX_test_b = encode_test_data.drop(['id', 'Premium Amount'], axis=1)\n\n# for lightgbm Classifier\nX_b2 = encode_train_data2.drop(['id', 'Premium Amount flg-500'], axis=1)\nX_test_b2 = encode_test_data2.drop(['id'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:18.687048Z","iopub.execute_input":"2024-12-31T14:14:18.687466Z","iopub.status.idle":"2024-12-31T14:14:19.063384Z","shell.execute_reply.started":"2024-12-31T14:14:18.687426Z","shell.execute_reply":"2024-12-31T14:14:19.062488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# many features\nX_d = encode_train_data3.drop(['id', 'Premium Amount'], axis=1)\nX_test_d = encode_test_data3.drop(['id', 'Premium Amount'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:19.064253Z","iopub.execute_input":"2024-12-31T14:14:19.064527Z","iopub.status.idle":"2024-12-31T14:14:19.447031Z","shell.execute_reply.started":"2024-12-31T14:14:19.064503Z","shell.execute_reply":"2024-12-31T14:14:19.445684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# for LinearRegression\nsc = StandardScaler()\nsc.fit(X_d)\n\nX_d_std = pd.DataFrame(sc.transform(X_d), columns=X_d.columns)\nX_test_d_std = pd.DataFrame(sc.transform(X_test_d), columns=X_d.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:19.448249Z","iopub.execute_input":"2024-12-31T14:14:19.448664Z","iopub.status.idle":"2024-12-31T14:14:20.598946Z","shell.execute_reply.started":"2024-12-31T14:14:19.448624Z","shell.execute_reply":"2024-12-31T14:14:20.597864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# target\ny = encode_train_data['Premium Amount']\ny_log = np.log1p(encode_train_data['Premium Amount'])\n# target(for classifier)\ny_flg = train_data2['Premium Amount flg-500']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:20.600071Z","iopub.execute_input":"2024-12-31T14:14:20.600359Z","iopub.status.idle":"2024-12-31T14:14:20.632884Z","shell.execute_reply.started":"2024-12-31T14:14:20.600334Z","shell.execute_reply":"2024-12-31T14:14:20.631754Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Models","metadata":{}},{"cell_type":"markdown","source":"I created 10 machine learning models.","metadata":{}},{"cell_type":"markdown","source":"1. CatBoost\n2. CatBoost (Classifier)\n3. LightGBM 1  (max_depth:5)\n4. LightGBM 2  (max_depth:9)\n5. LightGBM 3  (max_depth:15)\n6. LightGBM (Classifier)\n7. LightGBM (many features)\n8. XGBoost\n9. RandomForest\n10. LinearRegression","metadata":{}},{"cell_type":"code","source":"# RMSLE of the predicted train data is stored.\nRMSLEs = pd.DataFrame()\nRMSLEs['RMSLE'] = ''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:20.633933Z","iopub.execute_input":"2024-12-31T14:14:20.634309Z","iopub.status.idle":"2024-12-31T14:14:20.642236Z","shell.execute_reply.started":"2024-12-31T14:14:20.634274Z","shell.execute_reply":"2024-12-31T14:14:20.640706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataframe to put in train data predictions\nvalid_preds = pd.DataFrame()\n\n# Dataframe to put the predicted values of test data\ntest_preds = pd.DataFrame()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:20.644174Z","iopub.execute_input":"2024-12-31T14:14:20.645512Z","iopub.status.idle":"2024-12-31T14:14:20.661675Z","shell.execute_reply.started":"2024-12-31T14:14:20.645457Z","shell.execute_reply":"2024-12-31T14:14:20.660663Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. CatBoost","metadata":{}},{"cell_type":"code","source":"cat_valid_preds = pd.DataFrame()\ncat_test_preds = pd.DataFrame()\n\ncat_params = {\n        'iterations': 80,\n        'depth': 8,\n        'random_strength': 62,\n        'bagging_temperature': 1.091,\n        'od_type': 'IncToDec',\n        'early_stopping_rounds': 15,\n        'learning_rate': 0.4\n}\n\ncategorical_features_indices = np.where(X_a.dtypes==object)[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:20.662625Z","iopub.execute_input":"2024-12-31T14:14:20.663049Z","iopub.status.idle":"2024-12-31T14:14:20.686486Z","shell.execute_reply.started":"2024-12-31T14:14:20.663006Z","shell.execute_reply":"2024-12-31T14:14:20.684682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_a)):\n    X_train, X_valid = X_a.iloc[train_idx], X_a.iloc[valid_idx]\n    y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n    \n    train_pool = Pool(X_train, y_train, cat_features=categorical_features_indices)\n    valid_pool = Pool(X_valid, y_valid, cat_features=categorical_features_indices)\n    cat_model = cb.CatBoostRegressor(**cat_params)\n    cat_model.fit(train_pool, eval_set=[valid_pool], use_best_model=True, verbose=False)\n    \n    cat_valid_pred = np.expm1(cat_model.predict(X_valid))\n    cat_valid_pred = pd.DataFrame({'index':y_valid.index, 'cat_model': cat_valid_pred})\n    cat_valid_preds = pd.concat([cat_valid_preds, cat_valid_pred], axis=0)\n\n    cat_test_pred = np.expm1(cat_model.predict(X_test_a))\n    cat_test_pred = pd.DataFrame({f'cat_test_pred_{i+1}': cat_test_pred})\n    cat_test_preds = pd.concat([cat_test_preds, cat_test_pred], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:14:20.687706Z","iopub.execute_input":"2024-12-31T14:14:20.688392Z","iopub.status.idle":"2024-12-31T14:16:24.647623Z","shell.execute_reply.started":"2024-12-31T14:14:20.688354Z","shell.execute_reply":"2024-12-31T14:16:24.646477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_valid_preds = cat_valid_preds.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:16:24.648625Z","iopub.execute_input":"2024-12-31T14:16:24.648937Z","iopub.status.idle":"2024-12-31T14:16:24.744235Z","shell.execute_reply.started":"2024-12-31T14:16:24.648912Z","shell.execute_reply":"2024-12-31T14:16:24.743175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMSLEs.loc['cat_model'] = np.sqrt(mean_squared_log_error(y, cat_valid_preds['cat_model']))\nRMSLEs.loc['cat_model']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:16:24.745366Z","iopub.execute_input":"2024-12-31T14:16:24.745677Z","iopub.status.idle":"2024-12-31T14:16:24.808136Z","shell.execute_reply.started":"2024-12-31T14:16:24.745651Z","shell.execute_reply":"2024-12-31T14:16:24.807166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_test_preds['cat_test_preds_mean'] = cat_test_preds.mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:16:24.809196Z","iopub.execute_input":"2024-12-31T14:16:24.809472Z","iopub.status.idle":"2024-12-31T14:16:24.922969Z","shell.execute_reply.started":"2024-12-31T14:16:24.809448Z","shell.execute_reply":"2024-12-31T14:16:24.921798Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. CatBoost (Classifier)","metadata":{}},{"cell_type":"code","source":"cat_valid_preds_C = pd.DataFrame()\ncat_test_preds_C = pd.DataFrame()\n\ncat_params_C = {\n        'iterations': 80,\n        'depth': 8,\n        'random_strength': 62,\n        'bagging_temperature': 1.091,\n        'od_type': 'IncToDec',\n        'early_stopping_rounds': 15\n}\n\ncategorical_features_indices2 = np.where(X_a2.dtypes==object)[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:16:24.923883Z","iopub.execute_input":"2024-12-31T14:16:24.924202Z","iopub.status.idle":"2024-12-31T14:16:24.931172Z","shell.execute_reply.started":"2024-12-31T14:16:24.924174Z","shell.execute_reply":"2024-12-31T14:16:24.929941Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_a2)):\n    X_train, X_valid = X_a2.iloc[train_idx], X_a2.iloc[valid_idx]\n    y_train, y_valid = y_flg.iloc[train_idx], y_flg.iloc[valid_idx]\n    \n    train_pool = Pool(X_train, y_train, cat_features=categorical_features_indices2)\n    valid_pool = Pool(X_valid, y_valid, cat_features=categorical_features_indices2)\n    cat_model_C = cb.CatBoostClassifier(**cat_params_C)\n    cat_model_C.fit(train_pool, eval_set=[valid_pool], use_best_model=True, verbose=False)\n    \n    cat_valid_pred_C = cat_model_C.predict(X_valid)\n    cat_valid_pred_C = pd.DataFrame({'index':y_valid.index, 'cat_model_C': cat_valid_pred_C})\n    cat_valid_preds_C = pd.concat([cat_valid_preds_C, cat_valid_pred_C], axis=0)\n\n    cat_test_pred_C = cat_model_C.predict(X_test_a2)\n    cat_test_pred_C = pd.DataFrame({f'cat_test_pred_C_{i+1}': cat_test_pred_C})\n    cat_test_preds_C = pd.concat([cat_test_preds_C, cat_test_pred_C], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:16:24.932139Z","iopub.execute_input":"2024-12-31T14:16:24.932401Z","iopub.status.idle":"2024-12-31T14:19:45.808671Z","shell.execute_reply.started":"2024-12-31T14:16:24.932379Z","shell.execute_reply":"2024-12-31T14:19:45.807704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_valid_preds_C = cat_valid_preds_C.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:19:45.810105Z","iopub.execute_input":"2024-12-31T14:19:45.810494Z","iopub.status.idle":"2024-12-31T14:19:45.898124Z","shell.execute_reply.started":"2024-12-31T14:19:45.810458Z","shell.execute_reply":"2024-12-31T14:19:45.897176Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_test_preds_C['cat_test_preds_C_max'] = cat_test_preds_C.max(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:19:45.898952Z","iopub.execute_input":"2024-12-31T14:19:45.89922Z","iopub.status.idle":"2024-12-31T14:19:45.998776Z","shell.execute_reply.started":"2024-12-31T14:19:45.899197Z","shell.execute_reply":"2024-12-31T14:19:45.997701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cat_test_preds_C.groupby('cat_test_preds_C_max').size()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:19:45.999712Z","iopub.execute_input":"2024-12-31T14:19:45.999974Z","iopub.status.idle":"2024-12-31T14:19:46.020223Z","shell.execute_reply.started":"2024-12-31T14:19:45.999948Z","shell.execute_reply":"2024-12-31T14:19:46.018954Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. LightGBM 1  (max_depth:5)","metadata":{}},{"cell_type":"code","source":"lgbm_valid_preds1 = pd.DataFrame()\nlgbm_test_preds1 = pd.DataFrame()\n\nlgbm_params1 = {\n    'objective' : 'regression',\n    'boosting_type': 'gbdt',\n    'metric' : 'rmse',\n    'n_estimators': 10000,\n    'num_leaves' : 50,\n    'max_depth' : 5,\n    'max_bins' : 300,\n    'lambda_l1': 0.712,\n    'lambda_l2': 0.003,\n    'seed': 88,\n    'verbose' : -1}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:19:46.021625Z","iopub.execute_input":"2024-12-31T14:19:46.02204Z","iopub.status.idle":"2024-12-31T14:19:46.028039Z","shell.execute_reply.started":"2024-12-31T14:19:46.022011Z","shell.execute_reply":"2024-12-31T14:19:46.026943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_b)):\n    X_train, X_valid = X_b.iloc[train_idx], X_b.iloc[valid_idx]\n    y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n    \n    train_set = lgb.Dataset(X_train, y_train)\n    valid_sets = lgb.Dataset(X_valid, y_valid, reference=train_set)\n    lgbm_model1 = lgb.train(lgbm_params1, train_set=train_set, valid_sets=valid_sets,\n                           callbacks=[lgb.early_stopping(stopping_rounds=10, verbose=False)])\n    \n    lgbm_valid_pred1 = np.expm1(lgbm_model1.predict(X_valid))\n    lgbm_valid_pred1 = pd.DataFrame({'index':y_valid.index, 'lgbm_model1': lgbm_valid_pred1})\n    lgbm_valid_preds1 = pd.concat([lgbm_valid_preds1, lgbm_valid_pred1], axis=0)\n\n    lgbm_test_pred1 = np.expm1(lgbm_model1.predict(X_test_b))\n    lgbm_test_pred1 = pd.DataFrame({f'lgbm_test_pred1_{i+1}': lgbm_test_pred1})\n    lgbm_test_preds1 = pd.concat([lgbm_test_preds1, lgbm_test_pred1], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:19:46.028851Z","iopub.execute_input":"2024-12-31T14:19:46.02911Z","iopub.status.idle":"2024-12-31T14:21:00.299693Z","shell.execute_reply.started":"2024-12-31T14:19:46.029088Z","shell.execute_reply":"2024-12-31T14:21:00.298708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_valid_preds1 = lgbm_valid_preds1.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:00.30082Z","iopub.execute_input":"2024-12-31T14:21:00.301208Z","iopub.status.idle":"2024-12-31T14:21:00.397104Z","shell.execute_reply.started":"2024-12-31T14:21:00.301172Z","shell.execute_reply":"2024-12-31T14:21:00.395961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMSLEs.loc['lgbm_model1'] = np.sqrt(mean_squared_log_error(y, lgbm_valid_preds1['lgbm_model1']))\nRMSLEs.loc['lgbm_model1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:00.398199Z","iopub.execute_input":"2024-12-31T14:21:00.398468Z","iopub.status.idle":"2024-12-31T14:21:00.46055Z","shell.execute_reply.started":"2024-12-31T14:21:00.398445Z","shell.execute_reply":"2024-12-31T14:21:00.459361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_test_preds1['lgbm_test_preds1_mean'] = lgbm_test_preds1.mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:00.461455Z","iopub.execute_input":"2024-12-31T14:21:00.461763Z","iopub.status.idle":"2024-12-31T14:21:00.57881Z","shell.execute_reply.started":"2024-12-31T14:21:00.461737Z","shell.execute_reply":"2024-12-31T14:21:00.577785Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. LightGBM 2 (max_depth:9)","metadata":{}},{"cell_type":"code","source":"lgbm_valid_preds2 = pd.DataFrame()\nlgbm_test_preds2 = pd.DataFrame()\n\nlgbm_params2 = {\n    'objective' : 'regression',\n    'boosting_type': 'gbdt',\n    'metric' : 'rmse',\n    'n_estimators': 10000,\n    'num_leaves' : 100,\n    'max_depth' : 9,\n    'lambda_l1': 0.712,\n    'lambda_l2': 0.003,\n    'seed': 88,\n    'verbose' : -1}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:00.579773Z","iopub.execute_input":"2024-12-31T14:21:00.580151Z","iopub.status.idle":"2024-12-31T14:21:00.586185Z","shell.execute_reply.started":"2024-12-31T14:21:00.580116Z","shell.execute_reply":"2024-12-31T14:21:00.58503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_b)):\n    X_train, X_valid = X_b.iloc[train_idx], X_b.iloc[valid_idx]\n    y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n    \n    train_set = lgb.Dataset(X_train, y_train)\n    valid_sets = lgb.Dataset(X_valid, y_valid, reference=train_set)\n    lgbm_model2 = lgb.train(lgbm_params2, train_set=train_set, valid_sets=valid_sets,\n                           callbacks=[lgb.early_stopping(stopping_rounds=10, verbose=False)])\n    \n    lgbm_valid_pred2 = np.expm1(lgbm_model2.predict(X_valid))\n    lgbm_valid_pred2 = pd.DataFrame({'index':y_valid.index, 'lgbm_model2': lgbm_valid_pred2})\n    lgbm_valid_preds2 = pd.concat([lgbm_valid_preds2, lgbm_valid_pred2], axis=0)\n\n    lgbm_test_pred2 = np.expm1(lgbm_model2.predict(X_test_b))\n    lgbm_test_pred2 = pd.DataFrame({f'lgbm_test_pred2_{i+1}': lgbm_test_pred2})\n    lgbm_test_preds2 = pd.concat([lgbm_test_preds2, lgbm_test_pred2], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:00.587327Z","iopub.execute_input":"2024-12-31T14:21:00.587775Z","iopub.status.idle":"2024-12-31T14:21:39.896112Z","shell.execute_reply.started":"2024-12-31T14:21:00.587724Z","shell.execute_reply":"2024-12-31T14:21:39.895031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_valid_preds2 = lgbm_valid_preds2.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:39.897132Z","iopub.execute_input":"2024-12-31T14:21:39.897444Z","iopub.status.idle":"2024-12-31T14:21:39.989773Z","shell.execute_reply.started":"2024-12-31T14:21:39.897404Z","shell.execute_reply":"2024-12-31T14:21:39.988647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMSLEs.loc['lgbm_model2'] = np.sqrt(mean_squared_log_error(y, lgbm_valid_preds2['lgbm_model2']))\nRMSLEs.loc['lgbm_model2']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:39.990673Z","iopub.execute_input":"2024-12-31T14:21:39.990939Z","iopub.status.idle":"2024-12-31T14:21:40.05234Z","shell.execute_reply.started":"2024-12-31T14:21:39.990915Z","shell.execute_reply":"2024-12-31T14:21:40.051462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_test_preds2['lgbm_test_preds2_mean'] = lgbm_test_preds2.mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:40.053213Z","iopub.execute_input":"2024-12-31T14:21:40.053473Z","iopub.status.idle":"2024-12-31T14:21:40.161826Z","shell.execute_reply.started":"2024-12-31T14:21:40.053449Z","shell.execute_reply":"2024-12-31T14:21:40.160895Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. LightGBM 3  (max_depth:15)","metadata":{}},{"cell_type":"code","source":"lgbm_valid_preds3 = pd.DataFrame()\nlgbm_test_preds3 = pd.DataFrame()\n\nlgbm_params3 = {\n    'objective' : 'regression',\n    'boosting_type': 'gbdt',\n    'metric' : 'rmse',\n    'n_estimators': 10000,\n    'num_leaves' : 100,\n    'max_depth' : 15,\n    'lambda_l1': 0.712,\n    'lambda_l2': 0.003,\n    'seed': 88,\n    'verbose' : -1}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:40.162741Z","iopub.execute_input":"2024-12-31T14:21:40.163045Z","iopub.status.idle":"2024-12-31T14:21:40.168985Z","shell.execute_reply.started":"2024-12-31T14:21:40.163016Z","shell.execute_reply":"2024-12-31T14:21:40.167732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_b)):\n    X_train, X_valid = X_b.iloc[train_idx], X_b.iloc[valid_idx]\n    y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n    \n    train_set = lgb.Dataset(X_train, y_train)\n    valid_sets = lgb.Dataset(X_valid, y_valid, reference=train_set)\n    lgbm_model3 = lgb.train(lgbm_params3, train_set=train_set, valid_sets=valid_sets,\n                           callbacks=[lgb.early_stopping(stopping_rounds=10, verbose=False)])\n    \n    lgbm_valid_pred3 = np.expm1(lgbm_model3.predict(X_valid))\n    lgbm_valid_pred3 = pd.DataFrame({'index':y_valid.index, 'lgbm_model3': lgbm_valid_pred3})\n    lgbm_valid_preds3 = pd.concat([lgbm_valid_preds3, lgbm_valid_pred3], axis=0)\n\n    lgbm_test_pred3 = np.expm1(lgbm_model3.predict(X_test_b))\n    lgbm_test_pred3 = pd.DataFrame({f'lgbm_test_pred3_{i+1}': lgbm_test_pred3})\n    lgbm_test_preds3 = pd.concat([lgbm_test_preds3, lgbm_test_pred3], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:21:40.170158Z","iopub.execute_input":"2024-12-31T14:21:40.170547Z","iopub.status.idle":"2024-12-31T14:22:12.837853Z","shell.execute_reply.started":"2024-12-31T14:21:40.170507Z","shell.execute_reply":"2024-12-31T14:22:12.83695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_valid_preds3 = lgbm_valid_preds3.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:12.838776Z","iopub.execute_input":"2024-12-31T14:22:12.839053Z","iopub.status.idle":"2024-12-31T14:22:12.93031Z","shell.execute_reply.started":"2024-12-31T14:22:12.839029Z","shell.execute_reply":"2024-12-31T14:22:12.929454Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMSLEs.loc['lgbm_model3'] = np.sqrt(mean_squared_log_error(y, lgbm_valid_preds3['lgbm_model3']))\nRMSLEs.loc['lgbm_model3']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:12.93138Z","iopub.execute_input":"2024-12-31T14:22:12.931757Z","iopub.status.idle":"2024-12-31T14:22:12.992855Z","shell.execute_reply.started":"2024-12-31T14:22:12.931714Z","shell.execute_reply":"2024-12-31T14:22:12.991986Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_test_preds3['lgbm_test_preds3_mean'] = lgbm_test_preds3.mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:12.993751Z","iopub.execute_input":"2024-12-31T14:22:12.994045Z","iopub.status.idle":"2024-12-31T14:22:13.104063Z","shell.execute_reply.started":"2024-12-31T14:22:12.99402Z","shell.execute_reply":"2024-12-31T14:22:13.103078Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. LightGBM (Classifier)","metadata":{}},{"cell_type":"code","source":"lgbm_valid_preds_C = pd.DataFrame()\nlgbm_test_preds_C = pd.DataFrame()\n\nlgbm_params_C = {\n    'objective' : 'binary',\n    'boosting_type': 'gbdt',\n    'metric' : 'binary_logloss',\n    'n_estimators': 10000,\n    'num_leaves' : 100,\n    'max_depth' : 15,\n    'lambda_l1': 0.712,\n    'lambda_l2': 0.003,\n    'seed': 88,\n    'verbose' : -1}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:13.112391Z","iopub.execute_input":"2024-12-31T14:22:13.112741Z","iopub.status.idle":"2024-12-31T14:22:13.118482Z","shell.execute_reply.started":"2024-12-31T14:22:13.112714Z","shell.execute_reply":"2024-12-31T14:22:13.11755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_b2)):\n    X_train, X_valid = X_b2.iloc[train_idx], X_b2.iloc[valid_idx]\n    y_train, y_valid = y_flg.iloc[train_idx], y_flg.iloc[valid_idx]\n    \n    train_set = lgb.Dataset(X_train, y_train)\n    valid_sets = lgb.Dataset(X_valid, y_valid, reference=train_set)\n    lgbm_model_C = lgb.train(lgbm_params_C, train_set=train_set, valid_sets=valid_sets,\n                           callbacks=[lgb.early_stopping(stopping_rounds=10, verbose=False)])\n    \n    lgbm_valid_pred_C = lgbm_model_C.predict(X_valid)\n    lgbm_valid_pred_C = np.where(lgbm_valid_pred_C < 0.5, 0, 1)\n    lgbm_valid_pred_C = pd.DataFrame({'index':y_valid.index, 'lgbm_model_C': lgbm_valid_pred_C})\n    lgbm_valid_preds_C = pd.concat([lgbm_valid_preds_C, lgbm_valid_pred_C], axis=0)\n\n    lgbm_test_pred_C = lgbm_model_C.predict(X_test_b2)\n    lgbm_test_pred_C = np.where(lgbm_test_pred_C < 0.5, 0, 1)\n    lgbm_test_pred_C = pd.DataFrame({f'lgbm_test_pred_C_{i+1}': lgbm_test_pred_C})\n    lgbm_test_preds_C = pd.concat([lgbm_test_preds_C, lgbm_test_pred_C], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:13.120767Z","iopub.execute_input":"2024-12-31T14:22:13.121055Z","iopub.status.idle":"2024-12-31T14:22:56.186886Z","shell.execute_reply.started":"2024-12-31T14:22:13.12103Z","shell.execute_reply":"2024-12-31T14:22:56.185821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_valid_preds_C = lgbm_valid_preds_C.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:56.187909Z","iopub.execute_input":"2024-12-31T14:22:56.188185Z","iopub.status.idle":"2024-12-31T14:22:56.275172Z","shell.execute_reply.started":"2024-12-31T14:22:56.188162Z","shell.execute_reply":"2024-12-31T14:22:56.274233Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_test_preds_C['lgbm_test_preds_C_max'] = lgbm_test_preds_C.max(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:56.276276Z","iopub.execute_input":"2024-12-31T14:22:56.276674Z","iopub.status.idle":"2024-12-31T14:22:56.374309Z","shell.execute_reply.started":"2024-12-31T14:22:56.276638Z","shell.execute_reply":"2024-12-31T14:22:56.373339Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_test_preds_C.groupby('lgbm_test_preds_C_max').size()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:56.375198Z","iopub.execute_input":"2024-12-31T14:22:56.375458Z","iopub.status.idle":"2024-12-31T14:22:56.395789Z","shell.execute_reply.started":"2024-12-31T14:22:56.375434Z","shell.execute_reply":"2024-12-31T14:22:56.394641Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 7. LightGBM (many features)","metadata":{}},{"cell_type":"code","source":"lgbm_valid_preds_M = pd.DataFrame()\nlgbm_test_preds_M = pd.DataFrame()\n\nlgbm_params_M = {\n    'objective' : 'regression',\n    'boosting_type': 'gbdt',\n    'metric' : 'rmse',\n    'n_estimators': 10000,\n    'num_leaves' : 100,\n    'max_depth' : 15,\n    'lambda_l1': 0.712,\n    'lambda_l2': 0.003,\n    'seed': 88,\n    'verbose' : -1}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:56.397168Z","iopub.execute_input":"2024-12-31T14:22:56.397516Z","iopub.status.idle":"2024-12-31T14:22:56.403304Z","shell.execute_reply.started":"2024-12-31T14:22:56.397479Z","shell.execute_reply":"2024-12-31T14:22:56.402186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_d)):\n    X_train, X_valid = X_d.iloc[train_idx], X_d.iloc[valid_idx]\n    y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n    \n    train_set = lgb.Dataset(X_train, y_train)\n    valid_sets = lgb.Dataset(X_valid, y_valid, reference=train_set)\n    lgbm_model_M = lgb.train(lgbm_params_M, train_set=train_set, valid_sets=valid_sets,\n                           callbacks=[lgb.early_stopping(stopping_rounds=10, verbose=False)])\n    \n    lgbm_valid_pred_M = np.expm1(lgbm_model_M.predict(X_valid))\n    lgbm_valid_pred_M = pd.DataFrame({'index':y_valid.index, 'lgbm_model_M': lgbm_valid_pred_M})\n    lgbm_valid_preds_M = pd.concat([lgbm_valid_preds_M, lgbm_valid_pred_M], axis=0)\n\n    lgbm_test_pred_M = np.expm1(lgbm_model_M.predict(X_test_d))\n    lgbm_test_pred_M = pd.DataFrame({f'lgbm_test_pred_M_{i+1}': lgbm_test_pred_M})\n    lgbm_test_preds_M = pd.concat([lgbm_test_preds_M, lgbm_test_pred_M], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:22:56.404444Z","iopub.execute_input":"2024-12-31T14:22:56.404841Z","iopub.status.idle":"2024-12-31T14:23:40.413453Z","shell.execute_reply.started":"2024-12-31T14:22:56.404802Z","shell.execute_reply":"2024-12-31T14:23:40.41223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_valid_preds_M = lgbm_valid_preds_M.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:23:40.414569Z","iopub.execute_input":"2024-12-31T14:23:40.414962Z","iopub.status.idle":"2024-12-31T14:23:40.507049Z","shell.execute_reply.started":"2024-12-31T14:23:40.414911Z","shell.execute_reply":"2024-12-31T14:23:40.506217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMSLEs.loc['lgbm_model_M'] = np.sqrt(mean_squared_log_error(y, lgbm_valid_preds_M['lgbm_model_M']))\nRMSLEs.loc['lgbm_model_M']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:23:40.507921Z","iopub.execute_input":"2024-12-31T14:23:40.508209Z","iopub.status.idle":"2024-12-31T14:23:40.578269Z","shell.execute_reply.started":"2024-12-31T14:23:40.508184Z","shell.execute_reply":"2024-12-31T14:23:40.577377Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"lgbm_test_preds_M['lgbm_test_preds_M_mean'] = lgbm_test_preds_M.mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:23:40.579075Z","iopub.execute_input":"2024-12-31T14:23:40.579399Z","iopub.status.idle":"2024-12-31T14:23:40.712638Z","shell.execute_reply.started":"2024-12-31T14:23:40.579359Z","shell.execute_reply":"2024-12-31T14:23:40.711806Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 8. XGBoost","metadata":{}},{"cell_type":"code","source":"xgb_valid_preds = pd.DataFrame()\nxgb_test_preds = pd.DataFrame()\n\nxgb_params = {'n_estimators': 10000,\n              'learning_rate': 0.05,\n              'min_child_weight': 7,\n              'max_depth': 11,\n              'colsample_bytree': 0.236,\n              'subsample': 0.803,\n              'reg_alpha': 0.053,\n              'reg_lambda': 0.035,\n              'gamma': 0.080}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:23:40.713535Z","iopub.execute_input":"2024-12-31T14:23:40.713953Z","iopub.status.idle":"2024-12-31T14:23:40.721954Z","shell.execute_reply.started":"2024-12-31T14:23:40.713913Z","shell.execute_reply":"2024-12-31T14:23:40.719923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_b)):\n    X_train, X_valid = X_b.iloc[train_idx], X_b.iloc[valid_idx]\n    y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n    xgb_model = xgb.XGBRegressor(**xgb_params,\n                            eval_set=[(X_valid, y_valid)],\n                            eval_metric='rmse',\n                            silent=False,\n                            random_state=0,\n                            n_jobs=-1,\n                            verbose=False)\n    xgb_model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], early_stopping_rounds=50, verbose=False)\n    \n    xgb_valid_pred = np.expm1(xgb_model.predict(X_valid))\n    xgb_valid_pred = pd.DataFrame({'index':y_valid.index, 'xgboost': xgb_valid_pred})\n    xgb_valid_preds = pd.concat([xgb_valid_preds, xgb_valid_pred], axis=0)\n\n    xgb_test_pred = np.expm1(xgb_model.predict(X_test_b))\n    xgb_test_pred = pd.DataFrame({f'xgb_test_pred_{i+1}': xgb_test_pred})\n    xgb_test_preds = pd.concat([xgb_test_preds, xgb_test_pred], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:23:40.723196Z","iopub.execute_input":"2024-12-31T14:23:40.723601Z","iopub.status.idle":"2024-12-31T14:32:23.77349Z","shell.execute_reply.started":"2024-12-31T14:23:40.72354Z","shell.execute_reply":"2024-12-31T14:32:23.772493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_valid_preds = xgb_valid_preds.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:32:23.77447Z","iopub.execute_input":"2024-12-31T14:32:23.774762Z","iopub.status.idle":"2024-12-31T14:32:23.866653Z","shell.execute_reply.started":"2024-12-31T14:32:23.774728Z","shell.execute_reply":"2024-12-31T14:32:23.865821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMSLEs.loc['xgboost'] = np.sqrt(mean_squared_log_error(y, xgb_valid_preds['xgboost']))\nRMSLEs.loc['xgboost']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:32:23.867482Z","iopub.execute_input":"2024-12-31T14:32:23.86786Z","iopub.status.idle":"2024-12-31T14:32:23.930681Z","shell.execute_reply.started":"2024-12-31T14:32:23.867826Z","shell.execute_reply":"2024-12-31T14:32:23.929469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"xgb_test_preds['xgb_test_preds_mean'] = xgb_test_preds.mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:32:23.931929Z","iopub.execute_input":"2024-12-31T14:32:23.932325Z","iopub.status.idle":"2024-12-31T14:32:24.03502Z","shell.execute_reply.started":"2024-12-31T14:32:23.932286Z","shell.execute_reply":"2024-12-31T14:32:24.034085Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 9. RandomForest","metadata":{}},{"cell_type":"code","source":"RF_valid_preds = pd.DataFrame()\nRF_test_preds = pd.DataFrame()\n\nRF_params = {\n    'random_state' : 0,\n    'n_estimators'  : 10,\n    'max_depth' : 10\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:32:24.036061Z","iopub.execute_input":"2024-12-31T14:32:24.036461Z","iopub.status.idle":"2024-12-31T14:32:24.042295Z","shell.execute_reply.started":"2024-12-31T14:32:24.03642Z","shell.execute_reply":"2024-12-31T14:32:24.041298Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_b)):\n    X_train, X_valid = X_b.iloc[train_idx], X_b.iloc[valid_idx]\n    y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n    RF_model = RandomForestRegressor(**RF_params)\n    RF_model.fit(X_train, y_train)\n    \n    RF_valid_pred = np.expm1(RF_model.predict(X_valid))\n    RF_valid_pred = pd.DataFrame({'index':y_valid.index, 'RandomForest': RF_valid_pred})\n    RF_valid_preds = pd.concat([RF_valid_preds, RF_valid_pred], axis=0)\n\n    RF_test_pred = np.expm1(RF_model.predict(X_test_b))\n    RF_test_pred = pd.DataFrame({f'RF_test_pred_{i+1}': RF_test_pred})\n    RF_test_preds = pd.concat([RF_test_preds, RF_test_pred], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:32:24.043353Z","iopub.execute_input":"2024-12-31T14:32:24.043771Z","iopub.status.idle":"2024-12-31T14:38:57.665885Z","shell.execute_reply.started":"2024-12-31T14:32:24.043722Z","shell.execute_reply":"2024-12-31T14:38:57.664623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RF_valid_preds = RF_valid_preds.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:38:57.667253Z","iopub.execute_input":"2024-12-31T14:38:57.66763Z","iopub.status.idle":"2024-12-31T14:38:57.763011Z","shell.execute_reply.started":"2024-12-31T14:38:57.667552Z","shell.execute_reply":"2024-12-31T14:38:57.762083Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMSLEs.loc['RandomForest'] = np.sqrt(mean_squared_log_error(y, RF_valid_preds['RandomForest']))\nRMSLEs.loc['RandomForest']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:38:57.764038Z","iopub.execute_input":"2024-12-31T14:38:57.764412Z","iopub.status.idle":"2024-12-31T14:38:57.827527Z","shell.execute_reply.started":"2024-12-31T14:38:57.764376Z","shell.execute_reply":"2024-12-31T14:38:57.826557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RF_test_preds['RandomForest_test_preds_mean'] = RF_test_preds.mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:38:57.82849Z","iopub.execute_input":"2024-12-31T14:38:57.828887Z","iopub.status.idle":"2024-12-31T14:38:57.937635Z","shell.execute_reply.started":"2024-12-31T14:38:57.828849Z","shell.execute_reply":"2024-12-31T14:38:57.936485Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 10. LinearRegression","metadata":{}},{"cell_type":"code","source":"linear_valid_preds = pd.DataFrame()\nlinear_test_preds = pd.DataFrame()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:38:57.938762Z","iopub.execute_input":"2024-12-31T14:38:57.939073Z","iopub.status.idle":"2024-12-31T14:38:57.944218Z","shell.execute_reply.started":"2024-12-31T14:38:57.939046Z","shell.execute_reply":"2024-12-31T14:38:57.943167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(X_d_std)):\n    X_train, X_valid = X_d_std.iloc[train_idx], X_d_std.iloc[valid_idx]\n    y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n    linear_model = LinearRegression()\n    linear_model.fit(X_train, y_train)\n    \n    linear_valid_pred = np.expm1(linear_model.predict(X_valid))\n    linear_valid_pred = pd.DataFrame({'index':y_valid.index, 'Multiple_regression': linear_valid_pred})\n    linear_valid_preds = pd.concat([linear_valid_preds, linear_valid_pred], axis=0)\n\n    linear_test_pred = np.expm1(linear_model.predict(X_test_d_std))\n    linear_test_pred = pd.DataFrame({f'linear_test_pred_{i+1}': linear_test_pred})\n    linear_test_preds = pd.concat([linear_test_preds, linear_test_pred], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:38:57.945149Z","iopub.execute_input":"2024-12-31T14:38:57.945445Z","iopub.status.idle":"2024-12-31T14:39:12.429042Z","shell.execute_reply.started":"2024-12-31T14:38:57.94542Z","shell.execute_reply":"2024-12-31T14:39:12.427501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"linear_valid_preds = linear_valid_preds.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:12.430146Z","iopub.execute_input":"2024-12-31T14:39:12.430631Z","iopub.status.idle":"2024-12-31T14:39:12.540689Z","shell.execute_reply.started":"2024-12-31T14:39:12.430572Z","shell.execute_reply":"2024-12-31T14:39:12.539656Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMSLEs.loc['Multiple_regression'] = np.sqrt(mean_squared_log_error(y, linear_valid_preds['Multiple_regression']))\nRMSLEs.loc['Multiple_regression']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:12.541627Z","iopub.execute_input":"2024-12-31T14:39:12.541987Z","iopub.status.idle":"2024-12-31T14:39:12.605893Z","shell.execute_reply.started":"2024-12-31T14:39:12.541951Z","shell.execute_reply":"2024-12-31T14:39:12.604963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"linear_test_preds['linear_test_preds_mean'] = linear_test_preds.mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:12.606764Z","iopub.execute_input":"2024-12-31T14:39:12.607125Z","iopub.status.idle":"2024-12-31T14:39:12.715916Z","shell.execute_reply.started":"2024-12-31T14:39:12.607098Z","shell.execute_reply":"2024-12-31T14:39:12.715031Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Stacking","metadata":{}},{"cell_type":"code","source":"# RMSLE of each model\nRMSLEs","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:12.716821Z","iopub.execute_input":"2024-12-31T14:39:12.717113Z","iopub.status.idle":"2024-12-31T14:39:12.725959Z","shell.execute_reply.started":"2024-12-31T14:39:12.717089Z","shell.execute_reply":"2024-12-31T14:39:12.724907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_preds['cat_model'] = cat_valid_preds\nvalid_preds['cat_model_C'] = cat_valid_preds_C\nvalid_preds['lgbm_model1'] = lgbm_valid_preds1\nvalid_preds['lgbm_model2'] = lgbm_valid_preds2\nvalid_preds['lgbm_model3'] = lgbm_valid_preds3\nvalid_preds['lgbm_model_C'] = lgbm_valid_preds_C\nvalid_preds['lgbm_model_M'] = lgbm_valid_preds_M\nvalid_preds['xgboost'] = xgb_valid_preds\nvalid_preds['RandomForest'] = RF_valid_preds\nvalid_preds['Multiple_regression'] = linear_valid_preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:12.726944Z","iopub.execute_input":"2024-12-31T14:39:12.727285Z","iopub.status.idle":"2024-12-31T14:39:12.81478Z","shell.execute_reply.started":"2024-12-31T14:39:12.727257Z","shell.execute_reply":"2024-12-31T14:39:12.813828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# The response variable for the test data predicted by each model.\ntest_preds['cat_model'] = cat_test_preds['cat_test_preds_mean']\ntest_preds['cat_model_C'] = cat_test_preds_C['cat_test_preds_C_max']\ntest_preds['lgbm_model1'] = lgbm_test_preds1['lgbm_test_preds1_mean']\ntest_preds['lgbm_model2'] = lgbm_test_preds2['lgbm_test_preds2_mean']\ntest_preds['lgbm_model3'] = lgbm_test_preds3['lgbm_test_preds3_mean']\ntest_preds['lgbm_model_C'] = lgbm_test_preds_C['lgbm_test_preds_C_max']\ntest_preds['lgbm_model_M'] = lgbm_test_preds_M['lgbm_test_preds_M_mean']\ntest_preds['xgboost'] = xgb_test_preds['xgb_test_preds_mean']\ntest_preds['RandomForest'] = RF_test_preds['RandomForest_test_preds_mean']\ntest_preds['Multiple_regression'] = linear_test_preds['linear_test_preds_mean']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:12.815835Z","iopub.execute_input":"2024-12-31T14:39:12.816212Z","iopub.status.idle":"2024-12-31T14:39:12.858425Z","shell.execute_reply.started":"2024-12-31T14:39:12.816174Z","shell.execute_reply":"2024-12-31T14:39:12.857086Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train2 = valid_preds.copy()\ntest2 = test_preds.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:12.85953Z","iopub.execute_input":"2024-12-31T14:39:12.859941Z","iopub.status.idle":"2024-12-31T14:39:12.980088Z","shell.execute_reply.started":"2024-12-31T14:39:12.859905Z","shell.execute_reply":"2024-12-31T14:39:12.979094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stacking_valid_preds = pd.DataFrame()\nstacking_test_preds = pd.DataFrame()\n\nstacking_params = {\n    'objective' : 'regression',\n    'boosting_type': 'gbdt',\n    'metric' : 'rmse',\n    'n_estimators': 10000,\n    'num_leaves' : 100,\n    'max_depth' : 20,\n    'lambda_l1': 0.712,\n    'lambda_l2': 0.003,\n    'seed': 88,\n    'verbose' : -1}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:12.981014Z","iopub.execute_input":"2024-12-31T14:39:12.981318Z","iopub.status.idle":"2024-12-31T14:39:12.987314Z","shell.execute_reply.started":"2024-12-31T14:39:12.981292Z","shell.execute_reply":"2024-12-31T14:39:12.986418Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"kf = KFold(n_splits=5, shuffle=True, random_state=88)\nfor i, (train_idx, valid_idx) in enumerate(kf.split(train2)):\n    X_train, X_valid = train2.iloc[train_idx], train2.iloc[valid_idx]\n    y_train, y_valid = y_log.iloc[train_idx], y_log.iloc[valid_idx]\n    \n    train_set = lgb.Dataset(X_train, y_train)\n    valid_sets = lgb.Dataset(X_valid, y_valid, reference=train_set)\n    stacking_model = lgb.train(stacking_params, train_set=train_set, valid_sets=valid_sets,\n                           callbacks=[lgb.early_stopping(stopping_rounds=10, verbose=False)])\n    \n    stacking_valid_pred = np.expm1(stacking_model.predict(X_valid))\n    stacking_valid_pred = pd.DataFrame({'index':y_valid.index, 'stacking_model': stacking_valid_pred})\n    stacking_valid_preds = pd.concat([stacking_valid_preds, stacking_valid_pred], axis=0)\n\n    stacking_test_pred = np.expm1(stacking_model.predict(test2))\n    stacking_test_pred = pd.DataFrame({f'stacking_test_pred_{i+1}': stacking_test_pred})\n    stacking_test_preds = pd.concat([stacking_test_preds, stacking_test_pred], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:12.988405Z","iopub.execute_input":"2024-12-31T14:39:12.988778Z","iopub.status.idle":"2024-12-31T14:39:29.005811Z","shell.execute_reply.started":"2024-12-31T14:39:12.98874Z","shell.execute_reply":"2024-12-31T14:39:29.004758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stacking_valid_preds = stacking_valid_preds.sort_values('index').set_index('index')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:29.006862Z","iopub.execute_input":"2024-12-31T14:39:29.007147Z","iopub.status.idle":"2024-12-31T14:39:29.099024Z","shell.execute_reply.started":"2024-12-31T14:39:29.007123Z","shell.execute_reply":"2024-12-31T14:39:29.098096Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"RMSLEs.loc['stacking_model'] = np.sqrt(mean_squared_log_error(y, stacking_valid_preds['stacking_model']))\nRMSLEs.loc['stacking_model']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:29.099952Z","iopub.execute_input":"2024-12-31T14:39:29.100207Z","iopub.status.idle":"2024-12-31T14:39:29.162869Z","shell.execute_reply.started":"2024-12-31T14:39:29.100184Z","shell.execute_reply":"2024-12-31T14:39:29.161858Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"stacking_test_preds['stacking_test_preds_mean'] = stacking_test_preds.mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:29.163862Z","iopub.execute_input":"2024-12-31T14:39:29.164154Z","iopub.status.idle":"2024-12-31T14:39:29.272061Z","shell.execute_reply.started":"2024-12-31T14:39:29.164128Z","shell.execute_reply":"2024-12-31T14:39:29.271067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df['Premium Amount'] = stacking_test_preds['stacking_test_preds_mean']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:29.272961Z","iopub.execute_input":"2024-12-31T14:39:29.273227Z","iopub.status.idle":"2024-12-31T14:39:29.279548Z","shell.execute_reply.started":"2024-12-31T14:39:29.273204Z","shell.execute_reply":"2024-12-31T14:39:29.278537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit_df = test_df[['id', 'Premium Amount']].set_index('id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:29.280888Z","iopub.execute_input":"2024-12-31T14:39:29.281177Z","iopub.status.idle":"2024-12-31T14:39:29.326493Z","shell.execute_reply.started":"2024-12-31T14:39:29.281152Z","shell.execute_reply":"2024-12-31T14:39:29.32504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submit_df.to_csv('submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T14:39:29.327791Z","iopub.execute_input":"2024-12-31T14:39:29.328247Z","iopub.status.idle":"2024-12-31T14:39:31.092632Z","shell.execute_reply.started":"2024-12-31T14:39:29.328205Z","shell.execute_reply":"2024-12-31T14:39:31.091544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}