{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_log_error\nfrom xgboost import XGBRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:23:20.279607Z","iopub.execute_input":"2024-12-05T13:23:20.280809Z","iopub.status.idle":"2024-12-05T13:23:20.289952Z","shell.execute_reply.started":"2024-12-05T13:23:20.280767Z","shell.execute_reply":"2024-12-05T13:23:20.288849Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Class declaration\nClass contains methods:\n* get_numerical_cols(): Obtains and returns the columns that contain numerical data\n* get_categorical_cols(): Obtains and returns the columns that contain categorical data\n* replace_nan_numericals(): Replaces the missing numerical values with the median of the existing data\n* replace_nan_categoricals(): Replaces the missing numerical values with \"Unknown\"\n* replace_nans(): Replaces both numerical and categorical nans using the methods above\n* treat_data(): Replaces missing values and converts the policy start date to numerical year and month","metadata":{}},{"cell_type":"code","source":"class Insurance_Database:\n    def __init__(self, DB):\n        self.DB = DB\n\n    def get_numerical_cols(self):\n        numeric_cols = ((self.DB.dtypes == 'int64') | (self.DB.dtypes == 'float64'))\n        numeric_cols = numeric_cols[numeric_cols].index\n        \n        return numeric_cols\n\n    def get_categorical_cols(self):\n        object_cols = (self.DB.dtypes == 'object')\n        object_cols = object_cols[object_cols].index\n        \n        return object_cols\n\n    def replace_nan_numericals(self):\n        data_type_nan_info = pd.concat([self.DB.isnull().sum(), self.DB.dtypes], axis = 1)\n        data_type_nan_info = (data_type_nan_info[0] > 0) & ((data_type_nan_info[1] == 'int64') | (data_type_nan_info[1] == 'float64'))\n\n        data_type_nan_info = data_type_nan_info[data_type_nan_info].index\n        for col in data_type_nan_info:\n            self.DB[col].fillna(self.DB[col].median(), inplace = True)\n            \n        return self.DB\n\n    def replace_nan_categoricals(self):\n        data_type_nan_info = pd.concat([self.DB.isnull().sum(), self.DB.dtypes], axis = 1)\n        data_type_nan_info = (data_type_nan_info[0] > 0) & (data_type_nan_info[1] == 'object')\n        data_type_nan_info = data_type_nan_info[data_type_nan_info].index\n\n        for col in data_type_nan_info:\n            self.DB[col].fillna(\"Unknown\", inplace = True)\n            \n        return self.DB\n\n    def replace_nans(self):\n        X = self.replace_nan_numericals()\n        X = self.replace_nan_categoricals()\n        \n        return X\n\n    def treat_data(self):\n        X = self.replace_nans()\n        X[\"Policy Start Date\"] = pd.to_datetime(X[\"Policy Start Date\"], errors = \"coerce\")\n        X[\"Year\"] = X[\"Policy Start Date\"].dt.year.astype('float64')\n        X[\"Month\"] = X[\"Policy Start Date\"].dt.month.astype('float64')\n        X = X.drop(columns = [\"Policy Start Date\"])\n\n        return X\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:23:20.306657Z","iopub.execute_input":"2024-12-05T13:23:20.307415Z","iopub.status.idle":"2024-12-05T13:23:20.319099Z","shell.execute_reply.started":"2024-12-05T13:23:20.307369Z","shell.execute_reply":"2024-12-05T13:23:20.31795Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Loading in the data","metadata":{}},{"cell_type":"code","source":"# Load data and drop id columns\nX_train_data = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\").drop(columns = \"id\")\nX_test_data = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\").drop(columns = \"id\")\nprint(\"Data loaded successfully\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:23:20.321207Z","iopub.execute_input":"2024-12-05T13:23:20.322023Z","iopub.status.idle":"2024-12-05T13:23:27.753244Z","shell.execute_reply.started":"2024-12-05T13:23:20.321976Z","shell.execute_reply":"2024-12-05T13:23:27.752023Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploring the data","metadata":{}},{"cell_type":"code","source":"train_DB = Insurance_Database(X_train_data)\nnumeric_cols = train_DB.get_numerical_cols()\nX_train_data[numeric_cols].hist(figsize = (15, 10), bins = 20);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:23:27.754453Z","iopub.execute_input":"2024-12-05T13:23:27.754774Z","iopub.status.idle":"2024-12-05T13:23:30.302026Z","shell.execute_reply.started":"2024-12-05T13:23:27.754744Z","shell.execute_reply":"2024-12-05T13:23:30.300854Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_object = X_train_data.copy()\nobject_cols = train_DB.get_categorical_cols()\n\nfor col in object_cols:\n    x_object[col] = x_object[col].astype('category').cat.codes\n\n# Calculate and visualise correlation matrix\ncorr_coef = np.array(x_object.corr())\nnp.fill_diagonal(corr_coef, 0)\nxticklabels = x_object.columns\nax = sns.heatmap(corr_coef, linewidth=0.5, xticklabels = xticklabels, yticklabels = xticklabels)\nplt.title(\"Correlations between columns\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:23:30.304145Z","iopub.execute_input":"2024-12-05T13:23:30.304456Z","iopub.status.idle":"2024-12-05T13:23:33.929496Z","shell.execute_reply.started":"2024-12-05T13:23:30.304426Z","shell.execute_reply":"2024-12-05T13:23:33.928451Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model training","metadata":{}},{"cell_type":"code","source":"# Model training procedure\n# Treat training data\nX_train_data = train_DB.treat_data()\nif X_train_data.isnull().sum().sum() == 0:\n    print('Data treated successfully')\n\n# Extract premium amount\ny_train_data = X_train_data.pop(\"Premium Amount\")\ny_train_data = np.log1p(y_train_data)\n\n# Obtain numerical and categorical columns\nnumeric_cols = Insurance_Database(X_train_data).get_numerical_cols()\nobject_cols = Insurance_Database(X_train_data).get_categorical_cols()\n\npreprocessor = ColumnTransformer([(\"num\", StandardScaler(), numeric_cols),\n                                  (\"obj\", OneHotEncoder(), object_cols)])\n\n# Preprocessing the data (scaling and encoding)\nX_train_data = preprocessor.fit_transform(X_train_data)\n\n# Split the data\nX_train, X_test, y_train, y_test = train_test_split(X_train_data, \n                                                    y_train_data, \n                                                    test_size=0.25, \n                                                    random_state=0)\n\nmy_model = XGBRegressor(device='cuda', \n                        tree_method='hist', \n                        n_estimators=1500, \n                        learning_rate=0.01,\n                        verbosity = 2)\n\nprint(\"Commencing training...\")\nmy_model.fit(X_train, y_train)\n\nprint(\"Training finished\")\ny_pred = my_model.predict(X_test)\n\n# Calculate RMSLE\nrmsle = (mean_squared_log_error(np.expm1(y_test), np.expm1(y_pred)))**0.5\nprint(\"Validation RMSLE: \" + str(rmsle))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:23:33.930819Z","iopub.execute_input":"2024-12-05T13:23:33.93114Z","iopub.status.idle":"2024-12-05T13:25:03.726609Z","shell.execute_reply.started":"2024-12-05T13:23:33.931108Z","shell.execute_reply":"2024-12-05T13:25:03.725578Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prediction of test data using model","metadata":{}},{"cell_type":"code","source":"# Model prediction procedure\ntest_DB = Insurance_Database(X_test_data)\nX_test_data = test_DB.treat_data()\nif X_test_data.isnull().sum().sum() == 0:\n    print('Data treated successfully')\n\nnumeric_cols = Insurance_Database(X_test_data).get_numerical_cols()\nobject_cols = Insurance_Database(X_test_data).get_categorical_cols()\n\npreprocessor = ColumnTransformer([(\"num\", StandardScaler(), numeric_cols),\n                                  (\"obj\", OneHotEncoder(), object_cols)])\nX_test_data = preprocessor.fit_transform(X_test_data)\n\n# Use model on test data\ny_test_pred = my_model.predict(X_test_data)\nprint(\"Prediction successfully generated\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:25:03.727883Z","iopub.execute_input":"2024-12-05T13:25:03.728189Z","iopub.status.idle":"2024-12-05T13:25:21.029815Z","shell.execute_reply.started":"2024-12-05T13:25:03.728158Z","shell.execute_reply":"2024-12-05T13:25:21.027313Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualising prediction","metadata":{}},{"cell_type":"code","source":"plt.hist(np.expm1(y_test_pred), bins = np.linspace(0, 5000, 20));\nplt.title(\"Histogram of predicted premium amounts\");\nplt.xlabel(\"Premium Amount\");\nplt.ylabel(\"Frequency\");","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:25:21.03097Z","iopub.execute_input":"2024-12-05T13:25:21.03129Z","iopub.status.idle":"2024-12-05T13:25:21.40939Z","shell.execute_reply.started":"2024-12-05T13:25:21.031259Z","shell.execute_reply":"2024-12-05T13:25:21.408264Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\nsubmission[\"Premium Amount\"] = np.expm1(y_test_pred)\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T13:25:21.410885Z","iopub.execute_input":"2024-12-05T13:25:21.411215Z","iopub.status.idle":"2024-12-05T13:25:22.768368Z","shell.execute_reply.started":"2024-12-05T13:25:21.411182Z","shell.execute_reply":"2024-12-05T13:25:22.767235Z"}},"outputs":[],"execution_count":null}]}