{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Imports**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:08:02.116168Z","iopub.execute_input":"2024-12-06T08:08:02.116736Z","iopub.status.idle":"2024-12-06T08:08:02.120951Z","shell.execute_reply.started":"2024-12-06T08:08:02.116702Z","shell.execute_reply":"2024-12-06T08:08:02.120012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Reading the Datasets**","metadata":{}},{"cell_type":"code","source":"\n\ntrain_set = \"/kaggle/input/playground-series-s4e12/train.csv\"\ntest_set = '/kaggle/input/playground-series-s4e12/test.csv'\n\ntrain = pd.read_csv(train_set)\ntest = pd.read_csv(test_set)\ndata = pd.concat((train,test),ignore_index=True)\n\ntrain.shape,test.shape,data.shape","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:08:04.677629Z","iopub.execute_input":"2024-12-06T08:08:04.677995Z","iopub.status.idle":"2024-12-06T08:08:10.663093Z","shell.execute_reply.started":"2024-12-06T08:08:04.677963Z","shell.execute_reply":"2024-12-06T08:08:10.662101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:04:15.832559Z","iopub.execute_input":"2024-12-06T06:04:15.832923Z","iopub.status.idle":"2024-12-06T06:04:15.869974Z","shell.execute_reply.started":"2024-12-06T06:04:15.832891Z","shell.execute_reply":"2024-12-06T06:04:15.869146Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.tail(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:04:17.314543Z","iopub.execute_input":"2024-12-06T06:04:17.314902Z","iopub.status.idle":"2024-12-06T06:04:17.341205Z","shell.execute_reply.started":"2024-12-06T06:04:17.314872Z","shell.execute_reply":"2024-12-06T06:04:17.340365Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**EDA- Exploratory Data Analysis**","metadata":{}},{"cell_type":"code","source":"data.info(verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:04:20.0293Z","iopub.execute_input":"2024-12-06T06:04:20.029727Z","iopub.status.idle":"2024-12-06T06:04:20.05553Z","shell.execute_reply.started":"2024-12-06T06:04:20.029696Z","shell.execute_reply":"2024-12-06T06:04:20.054562Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From the above results we can devide the columns into two types based on their Dtype as,\n1. Numerical_columns (whose Dtye is int, float)\n2. Labeled/Catogorical_columns (whose Dtype is object)","metadata":{}},{"cell_type":"code","source":"data.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T06:04:23.992489Z","iopub.execute_input":"2024-12-06T06:04:23.99287Z","iopub.status.idle":"2024-12-06T06:04:23.99904Z","shell.execute_reply.started":"2024-12-06T06:04:23.992837Z","shell.execute_reply":"2024-12-06T06:04:23.998148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#looking for the null values\n\nmissing_values = data.drop(columns=['Premium Amount']).isnull().sum()\nmissing_values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:08:10.664986Z","iopub.execute_input":"2024-12-06T08:08:10.665396Z","iopub.status.idle":"2024-12-06T08:08:11.887104Z","shell.execute_reply.started":"2024-12-06T08:08:10.665351Z","shell.execute_reply":"2024-12-06T08:08:11.886072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From above,\n* Age, Annual income, Number of Dependents, Health Score, Previous Claims, Credit Score, Incurence Duration ----These are Numerical columns\n  Meaning that the missing values can be replaced with either of( Mean, Median, Average of their respective columns).\n\n* Maritual Status, Occupation, Customer Feedback --- These are catogorical columns.","metadata":{}},{"cell_type":"code","source":"# Calculate the percentage of NaN values for each column\nnan_percentages = (data.drop(columns=['Premium Amount']).isnull().sum() / len(data)) * 100\n\n# Create the horizontal bar plot\nplt.figure(figsize=(8, 6))\nsns.barplot(x=nan_percentages.values, y=nan_percentages.index, color=\"lightcoral\")\nplt.title('Percentage of Missing Values per Column', fontsize=13)\nplt.xlabel('Percentage of Missing Values', fontsize=11)\nplt.ylabel('Columns', fontsize=13)\nplt.grid(axis='x', linestyle='--', alpha=0.7)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T05:44:41.441111Z","iopub.execute_input":"2024-12-06T05:44:41.441402Z","iopub.status.idle":"2024-12-06T05:44:43.012984Z","shell.execute_reply.started":"2024-12-06T05:44:41.44137Z","shell.execute_reply":"2024-12-06T05:44:43.012077Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# cols = ['Previous Claims','Number of Dependents','Insurance Duration',\"Age\",\"Annual Income\", \"Health Score\", \"Credit Score\", \"Vehicle Age\",'Customer Feedback']\n# for col in cols:\n#     unique = data[col].value_counts()\n#     print(unique)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T05:44:43.014063Z","iopub.execute_input":"2024-12-06T05:44:43.014359Z","iopub.status.idle":"2024-12-06T05:44:43.018361Z","shell.execute_reply.started":"2024-12-06T05:44:43.014305Z","shell.execute_reply":"2024-12-06T05:44:43.01741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#droping the date column as it dosent make sense\ndata = data.drop(columns=['Policy Start Date'])\n\n\nnumeric_cols = data.select_dtypes(include=['float64', 'int64']).columns.to_list()\ncategorical_cols = data.select_dtypes(include=['object', 'category']).columns.to_list()\n\nprint(f\"Numeric columns:\\n{numeric_cols}\\n\\n\")\nprint(f\"Categorical columns:\\n{categorical_cols}\\n\\n\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:09:40.851386Z","iopub.execute_input":"2024-12-06T08:09:40.852256Z","iopub.status.idle":"2024-12-06T08:09:41.95873Z","shell.execute_reply.started":"2024-12-06T08:09:40.85222Z","shell.execute_reply":"2024-12-06T08:09:41.957779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#checking how the distribution of the unique values of each categorical column was, except for plicy start date.\n\ncols = ['Gender','Marital Status','Education Level','Occupation','Location','Policy Type','Customer Feedback','Smoking Status','Exercise Frequency','Property Type']\nfig,axes = plt.subplots(nrows=5,ncols=2,figsize=(10,15))\naxes = axes.flatten()\nfor i,col in enumerate(cols):\n    (data[col].value_counts()).plot(kind='bar',ax=axes[i])\n    axes[i].set_xlabel(col)\n    axes[i].set_ylabel('count')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T05:44:44.120266Z","iopub.execute_input":"2024-12-06T05:44:44.120621Z","iopub.status.idle":"2024-12-06T05:44:46.533352Z","shell.execute_reply.started":"2024-12-06T05:44:44.120591Z","shell.execute_reply":"2024-12-06T05:44:46.532455Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From this visualization it is clear that the distribution of values in each categorical column are even (i.e evenly distributed).\n","metadata":{}},{"cell_type":"code","source":"corelation_matrix = data[numeric_cols].corr()\n\nplt.figure(figsize=(8, 6))\nsns.heatmap(corelation_matrix, annot=True, cmap='coolwarm', linewidths=0.5)\nplt.title('Correlation Matrix')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T05:46:18.470773Z","iopub.execute_input":"2024-12-06T05:46:18.471089Z","iopub.status.idle":"2024-12-06T05:46:19.667776Z","shell.execute_reply.started":"2024-12-06T05:46:18.471062Z","shell.execute_reply":"2024-12-06T05:46:19.666819Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"From this visualization, it is clear that no specific feature from the numeric columns has good relation( not much co related) with the target column. So replacing the missing values with the median wont introduce the bias to the data set.","metadata":{}},{"cell_type":"markdown","source":"**PREPROCESSING**","metadata":{}},{"cell_type":"markdown","source":"* PREVIOUS CLAIMS:   \nMost of the cases the previous claim takes 0, meaning that there was no previous claims. Missing values would says the same as far i think. As     they have no previous claims instead of using 0 for that column, the just ignored and the resulted as missing value.\n* Number of Dependents:     \nFilling the missing values of this column with 0 as well.","metadata":{}},{"cell_type":"code","source":"# replacing the missing values with 0\ndata['Previous Claims'] = data['Previous Claims'].fillna(0.0)\ndata['Number of Dependents'] = data['Number of Dependents'].fillna(0)\ndata['Insurance Duration'] = data['Insurance Duration'].fillna(0)\n\n# Median\nfor col in [\"Age\", \"Annual Income\", \"Health Score\", \"Credit Score\", \"Vehicle Age\"]:\n    data[col] = data[col].fillna(data[col].median())\ndata['Customer Feedback'] = data['Customer Feedback'].fillna('Average')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:09:49.837604Z","iopub.execute_input":"2024-12-06T08:09:49.838305Z","iopub.status.idle":"2024-12-06T08:09:50.159225Z","shell.execute_reply.started":"2024-12-06T08:09:49.838258Z","shell.execute_reply":"2024-12-06T08:09:50.158468Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1.One-Hot Encoding: Useful for nominal (unordered) categories but can increase the number of columns significantly.\n\n2.Label Encoding: Assigns a unique number to each category but can impose an ordinal relationship where there isn't one.\n\n3.Target Encoding: Uses the target variable's mean to encode categories, useful for capturing the relationship between category and target.\n\n4.Frequency Encoding: Encodes categories based on their frequency of occurrence, balancing common and rare categories.","metadata":{}},{"cell_type":"code","source":"#using one-hot encoding for converting catogorial columns to numerical columns.\nencoded_data = pd.get_dummies(data, columns=categorical_cols, drop_first=False)\nencoded_data.shape\nencoded_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:09:52.499603Z","iopub.execute_input":"2024-12-06T08:09:52.5002Z","iopub.status.idle":"2024-12-06T08:09:54.532709Z","shell.execute_reply.started":"2024-12-06T08:09:52.500136Z","shell.execute_reply":"2024-12-06T08:09:54.531519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing = encoded_data['Premium Amount'][:1200000].isnull().sum()\nmissing","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:08:36.317528Z","iopub.execute_input":"2024-12-06T08:08:36.318336Z","iopub.status.idle":"2024-12-06T08:08:36.326432Z","shell.execute_reply.started":"2024-12-06T08:08:36.318283Z","shell.execute_reply":"2024-12-06T08:08:36.325314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#splitting the train and test data\n\ndata_train = encoded_data[:1200000]\ndata_test = encoded_data[1200000:].drop(columns=['Premium Amount'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:10:00.089851Z","iopub.execute_input":"2024-12-06T08:10:00.090702Z","iopub.status.idle":"2024-12-06T08:10:00.131157Z","shell.execute_reply.started":"2024-12-06T08:10:00.090665Z","shell.execute_reply":"2024-12-06T08:10:00.130476Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:08:48.747299Z","iopub.execute_input":"2024-12-06T08:08:48.748065Z","iopub.status.idle":"2024-12-06T08:08:48.772134Z","shell.execute_reply.started":"2024-12-06T08:08:48.748029Z","shell.execute_reply":"2024-12-06T08:08:48.771002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Model Building**","metadata":{}},{"cell_type":"markdown","source":"defining the function for RootMeanSquarError.","metadata":{}},{"cell_type":"code","source":"import numpy as np\ndef rms(y_pred,y_test):\n    squ_err = (y_pred-y_test)**2\n    mean_squ_err = np.mean(squ_err)\n    rt_mean_sqr_err = np.sqrt(mean_squ_err)\n    loged = np.log1p(rt_mean_sqr_err)\n    return loged","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:10:04.065251Z","iopub.execute_input":"2024-12-06T08:10:04.066028Z","iopub.status.idle":"2024-12-06T08:10:04.071243Z","shell.execute_reply.started":"2024-12-06T08:10:04.065992Z","shell.execute_reply":"2024-12-06T08:10:04.06999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nfeatures = data_train.drop(columns=['id','Premium Amount'])\ntarget = data_train['Premium Amount']\n\nX_train, Xtest, y_train, y_test = train_test_split(features,target,train_size=0.7,test_size=0.3,random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:10:06.665016Z","iopub.execute_input":"2024-12-06T08:10:06.665394Z","iopub.status.idle":"2024-12-06T08:10:06.939107Z","shell.execute_reply.started":"2024-12-06T08:10:06.665358Z","shell.execute_reply":"2024-12-06T08:10:06.93838Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"1. Linear Regression","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\n\nl_r = LinearRegression()\nl_r.fit(X_train,y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:10:09.010229Z","iopub.execute_input":"2024-12-06T08:10:09.010989Z","iopub.status.idle":"2024-12-06T08:10:11.103726Z","shell.execute_reply.started":"2024-12-06T08:10:09.010953Z","shell.execute_reply":"2024-12-06T08:10:11.10263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = l_r.predict(Xtest)\nprint(rms(y_pred,y_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:10:11.112762Z","iopub.execute_input":"2024-12-06T08:10:11.113508Z","iopub.status.idle":"2024-12-06T08:10:11.17824Z","shell.execute_reply.started":"2024-12-06T08:10:11.113467Z","shell.execute_reply":"2024-12-06T08:10:11.175087Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"2. XgBoost","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\n\nxgb_regressor = xgb.XGBRegressor(\n    n_estimators=100,\n    max_depth=8,\n    learning_rate=0.1,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    random_state=42\n)\n\nxgb_regressor.fit(X_train,y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:10:14.171911Z","iopub.execute_input":"2024-12-06T08:10:14.172888Z","iopub.status.idle":"2024-12-06T08:10:22.284259Z","shell.execute_reply.started":"2024-12-06T08:10:14.172845Z","shell.execute_reply":"2024-12-06T08:10:22.283511Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# predicting \ny_pred = xgb_regressor.predict(Xtest)\n# rsm \nprint(rms(y_pred,y_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:10:22.285541Z","iopub.execute_input":"2024-12-06T08:10:22.285821Z","iopub.status.idle":"2024-12-06T08:10:23.453218Z","shell.execute_reply.started":"2024-12-06T08:10:22.285793Z","shell.execute_reply":"2024-12-06T08:10:23.450344Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"3.MLP","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\n\nnn_model = Sequential()\nnn_model.add(Dense(64,input_dim=X_train.shape[1],activation='relu'))\nnn_model.add(Dense(32,activation='relu'))\nnn_model.add(Dense(16,activation='relu'))\nnn_model.add(Dense(1,activation='linear'))\nnn_model.compile(loss='mean_squared_error',optimizer='adam',metrics=['mean_squared_error'])\n\nnn_model.summary()\nhistroy = nn_model.fit(X_train,y_train,epochs=10,batch_size=32,validation_split=0.2)\n# loss,mse = nn_model.evalute(Xtest,y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T07:32:35.433917Z","iopub.execute_input":"2024-12-06T07:32:35.434314Z","iopub.status.idle":"2024-12-06T07:38:20.428094Z","shell.execute_reply.started":"2024-12-06T07:32:35.434276Z","shell.execute_reply":"2024-12-06T07:38:20.42687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# so the actual mse is np.log(val_mean_squared_error) of final epoch\nnp.log1p(757213.7500)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T07:41:19.117243Z","iopub.execute_input":"2024-12-06T07:41:19.11822Z","iopub.status.idle":"2024-12-06T07:41:19.12389Z","shell.execute_reply.started":"2024-12-06T07:41:19.11818Z","shell.execute_reply":"2024-12-06T07:41:19.122919Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"4.Decision Tree\n","metadata":{}},{"cell_type":"code","source":"# from sklearn.ensemble import RandomForestRegressor\n# r_f = RandomForestRegressor(n_estimators=100, random_state=42)\n# r_f.fit(X_train,y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T07:49:46.22912Z","iopub.execute_input":"2024-12-06T07:49:46.229799Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Submission**","metadata":{}},{"cell_type":"code","source":"sub = '/kaggle/input/playground-series-s4e12/sample_submission.csv'\nsub_df = pd.read_csv(sub)\npred = xgb_regressor.predict(data_test.drop(columns=['id']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:18:05.910221Z","iopub.execute_input":"2024-12-06T08:18:05.911034Z","iopub.status.idle":"2024-12-06T08:18:08.840694Z","shell.execute_reply.started":"2024-12-06T08:18:05.910994Z","shell.execute_reply":"2024-12-06T08:18:08.839869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_df['Premium Amount'] = pred\nsub_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T08:18:34.009808Z","iopub.execute_input":"2024-12-06T08:18:34.010221Z","iopub.status.idle":"2024-12-06T08:18:34.978512Z","shell.execute_reply.started":"2024-12-06T08:18:34.010186Z","shell.execute_reply":"2024-12-06T08:18:34.977358Z"}},"outputs":[],"execution_count":null}]}