{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30822,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, KFold\nfrom sklearn.metrics import mean_squared_error\nfrom catboost import CatBoostRegressor, Pool\nfrom lightgbm import LGBMRegressor\nfrom xgboost import XGBRegressor\nfrom sklearn.ensemble import StackingRegressor\nfrom sklearn.preprocessing import LabelEncoder\nfrom xgboost import XGBRegressor\nfrom sklearn.metrics import mean_squared_log_error\nfrom sklearn.model_selection import GridSearchCV\nimport xgboost as xgb\nimport optuna\nimport re\nimport warnings\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\nimport numpy as np","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:18.772457Z","iopub.execute_input":"2024-12-31T11:05:18.772771Z","iopub.status.idle":"2024-12-31T11:05:18.7775Z","shell.execute_reply.started":"2024-12-31T11:05:18.772745Z","shell.execute_reply":"2024-12-31T11:05:18.776807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv', index_col='id')\ntest_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv', index_col='id')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:21.967304Z","iopub.execute_input":"2024-12-31T11:05:21.967614Z","iopub.status.idle":"2024-12-31T11:05:27.409274Z","shell.execute_reply.started":"2024-12-31T11:05:21.967588Z","shell.execute_reply":"2024-12-31T11:05:27.408336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T10:38:12.802648Z","iopub.execute_input":"2024-12-31T10:38:12.802986Z","iopub.status.idle":"2024-12-31T10:38:12.822048Z","shell.execute_reply.started":"2024-12-31T10:38:12.80296Z","shell.execute_reply":"2024-12-31T10:38:12.821417Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop Irrelevant Features\ntrain_data = train_data.drop(columns=[\"Education Level\", \"Occupation\", \"Vehicle Age\", \"Policy Start Date\", \"Customer Feedback\", \"Property Type\"])\ntest_data = test_data.drop(columns=[\"Education Level\", \"Occupation\", \"Vehicle Age\", \"Policy Start Date\", \"Customer Feedback\", \"Property Type\"])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:27.410439Z","iopub.execute_input":"2024-12-31T11:05:27.410731Z","iopub.status.idle":"2024-12-31T11:05:27.596502Z","shell.execute_reply.started":"2024-12-31T11:05:27.410703Z","shell.execute_reply":"2024-12-31T11:05:27.595612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T10:38:26.575422Z","iopub.execute_input":"2024-12-31T10:38:26.575708Z","iopub.status.idle":"2024-12-31T10:38:26.591704Z","shell.execute_reply.started":"2024-12-31T10:38:26.575686Z","shell.execute_reply":"2024-12-31T10:38:26.590504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:32.697362Z","iopub.execute_input":"2024-12-31T11:05:32.697667Z","iopub.status.idle":"2024-12-31T11:05:32.995967Z","shell.execute_reply.started":"2024-12-31T11:05:32.697633Z","shell.execute_reply":"2024-12-31T11:05:32.995129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"missing_data = train_data.isnull().sum() / len(train_data) * 100\nprint(missing_data)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:35.997324Z","iopub.execute_input":"2024-12-31T11:05:35.997636Z","iopub.status.idle":"2024-12-31T11:05:36.292164Z","shell.execute_reply.started":"2024-12-31T11:05:35.997609Z","shell.execute_reply":"2024-12-31T11:05:36.291231Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Since Previous Claims has >30 % missing values and Credit Score has >10% missing values. Hence dropping this column also from both test and Trainign file\ntrain_data = train_data.drop(columns=[\"Previous Claims\", \"Credit Score\"])\ntest_data = test_data.drop(columns=[\"Previous Claims\", \"Credit Score\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:39.027689Z","iopub.execute_input":"2024-12-31T11:05:39.027999Z","iopub.status.idle":"2024-12-31T11:05:39.17501Z","shell.execute_reply.started":"2024-12-31T11:05:39.027971Z","shell.execute_reply":"2024-12-31T11:05:39.174021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handling Missing Values for rest features\n#Age \ntrain_data['Age'].fillna(train_data['Age'].median(), inplace=True)\ntest_data['Age'].fillna(test_data['Age'].median(), inplace=True)\n\n#Annual Income\ntrain_data['Annual Income'].fillna(train_data['Annual Income'].median(), inplace=True)\ntest_data['Annual Income'].fillna(test_data['Annual Income'].median(), inplace=True)\n\n#Marital Status\ntrain_data['Marital Status'].fillna(train_data['Marital Status'].mode()[0], inplace=True)\ntest_data['Marital Status'].fillna(test_data['Marital Status'].mode()[0], inplace=True)\n\n#Number of Dependents\ntrain_data['Number of Dependents'].fillna(train_data['Number of Dependents'].median(), inplace=True)\ntest_data['Number of Dependents'].fillna(test_data['Number of Dependents'].median(), inplace=True)\n\n#Health Score\ntrain_data['Health Score'].fillna(train_data['Health Score'].mean(), inplace=True)\ntest_data['Health Score'].fillna(test_data['Health Score'].mean(), inplace=True)\n\n#Insurance Duratio\ntrain_data['Insurance Duration'].fillna(train_data['Insurance Duration'].median(), inplace=True)\ntest_data['Insurance Duration'].fillna(test_data['Insurance Duration'].median(), inplace=True)\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:41.822425Z","iopub.execute_input":"2024-12-31T11:05:41.822763Z","iopub.status.idle":"2024-12-31T11:05:42.274021Z","shell.execute_reply.started":"2024-12-31T11:05:41.822736Z","shell.execute_reply":"2024-12-31T11:05:42.273108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:46.217241Z","iopub.execute_input":"2024-12-31T11:05:46.217553Z","iopub.status.idle":"2024-12-31T11:05:46.510135Z","shell.execute_reply.started":"2024-12-31T11:05:46.217526Z","shell.execute_reply":"2024-12-31T11:05:46.50926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:52.298102Z","iopub.execute_input":"2024-12-31T11:05:52.298421Z","iopub.status.idle":"2024-12-31T11:05:52.496658Z","shell.execute_reply.started":"2024-12-31T11:05:52.298397Z","shell.execute_reply":"2024-12-31T11:05:52.495952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\n# One-Hot Encode categorical columns\ncategorical_columns = ['Gender', 'Marital Status', 'Location', 'Policy Type', 'Smoking Status', 'Exercise Frequency']\ntrain_data = pd.get_dummies(train_data, columns=categorical_columns, drop_first=True)\ntest_data = pd.get_dummies(test_data, columns=categorical_columns, drop_first=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:55.69216Z","iopub.execute_input":"2024-12-31T11:05:55.692474Z","iopub.status.idle":"2024-12-31T11:05:56.527507Z","shell.execute_reply.started":"2024-12-31T11:05:55.692451Z","shell.execute_reply":"2024-12-31T11:05:56.526633Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature scaling\nfrom sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nnumerical_columns = ['Age', 'Annual Income', 'Number of Dependents', 'Health Score', 'Insurance Duration']\ntrain_data[numerical_columns] = scaler.fit_transform(train_data[numerical_columns])\ntest_data[numerical_columns] = scaler.fit_transform(test_data[numerical_columns])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:05:58.567333Z","iopub.execute_input":"2024-12-31T11:05:58.567613Z","iopub.status.idle":"2024-12-31T11:05:58.72166Z","shell.execute_reply.started":"2024-12-31T11:05:58.567591Z","shell.execute_reply":"2024-12-31T11:05:58.720962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Splitting Data\nfrom sklearn.model_selection import train_test_split\nX = train_data.drop(columns=['Premium Amount'])  \ny = train_data['Premium Amount'] \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T10:39:04.654489Z","iopub.execute_input":"2024-12-31T10:39:04.65477Z","iopub.status.idle":"2024-12-31T10:39:04.915072Z","shell.execute_reply.started":"2024-12-31T10:39:04.654748Z","shell.execute_reply":"2024-12-31T10:39:04.914379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Model Selection\nfrom xgboost import XGBRegressor\nmodel = XGBRegressor(n_estimators=500, learning_rate=0.15, max_depth=6, random_state=42)\nmodel.fit(X_train, y_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T10:39:10.45665Z","iopub.execute_input":"2024-12-31T10:39:10.456926Z","iopub.status.idle":"2024-12-31T10:39:25.144356Z","shell.execute_reply.started":"2024-12-31T10:39:10.456905Z","shell.execute_reply":"2024-12-31T10:39:25.142876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score\ny_pred = model.predict(X_test)\nprint(\"MAE:\", mean_absolute_error(y_test, y_pred))\nprint(\"MSE:\", mean_squared_error(y_test, y_pred))\nprint(\"R² Score:\", r2_score(y_test, y_pred))\nrmsle_train= np.sqrt(mean_squared_log_error(y_test, y_pred))\nprint(\"rmsle_train:\",rmsle_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T10:39:28.826867Z","iopub.execute_input":"2024-12-31T10:39:28.827207Z","iopub.status.idle":"2024-12-31T10:39:29.998265Z","shell.execute_reply.started":"2024-12-31T10:39:28.827178Z","shell.execute_reply":"2024-12-31T10:39:29.997583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions for test\ny_test_pred = model.predict(test_data)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:06:08.827242Z","iopub.execute_input":"2024-12-31T11:06:08.827525Z","iopub.status.idle":"2024-12-31T11:06:12.758241Z","shell.execute_reply.started":"2024-12-31T11:06:08.827504Z","shell.execute_reply":"2024-12-31T11:06:12.757522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Prepare submission file\ntest_data['id'] = range(1200000, 1200000 + len(test_data))\ntest_data['Premium Amount'] = y_test_pred\nsubmission = pd.DataFrame({'id': test_data['id'], 'Premium Amount': y_test_pred})\nsubmission.to_csv(\"submission.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T11:11:46.087679Z","iopub.execute_input":"2024-12-31T11:11:46.087975Z","iopub.status.idle":"2024-12-31T11:11:47.056114Z","shell.execute_reply.started":"2024-12-31T11:11:46.087953Z","shell.execute_reply":"2024-12-31T11:11:47.055407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #Hyperparameter Tuning\n# xgb_model = xgb.XGBRegressor(eval_metric='rmsle', tree_method='hist')\n# param_grid = {\n#     'n_estimators': [100, 200, 300],\n#     'learning_rate': [0.01, 0.1, 0.2],\n#     'max_depth': [4, 6, 8],\n#     'subsample': [0.7, 0.8, 1.0],\n#     'colsample_bytree': [0.7, 0.8, 1.0]\n# }\n\n# # grid_search = GridSearchCV(XGBRegressor(random_state=42), param_grid, cv=3, scoring='neg_mean_absolute_error')\n# grid_search = GridSearchCV(estimator=xgb_model, param_grid=param_grid,error_score='raise', n_jobs=-1)\n# grid_search.fit(X_train, y_train)\n# print(\"Best Parameters:\", grid_search.best_params_)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T06:37:23.302426Z","iopub.execute_input":"2024-12-31T06:37:23.302737Z","iopub.status.idle":"2024-12-31T08:27:49.546097Z","shell.execute_reply.started":"2024-12-31T06:37:23.302713Z","shell.execute_reply":"2024-12-31T08:27:49.545136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Get the best model from GridSearchCV\n# best_model = grid_search.best_estimator_\n\n# # Make predictions\n# y_train_pred = best_model.predict(X_train)\n# y_test_pred = best_model.predict(X_test)\n\n# # Calculate metrics for training set\n# mae_train = mean_absolute_error(y_train, y_train_pred)\n# mse_train = mean_squared_error(y_train, y_train_pred)\n# rmse_train = np.sqrt(mse_train)\n# rmsle_train= np.sqrt(mean_squared_log_error(y_train, y_train_pred))\n\n# # Print evaluation metrics\n# print(\"Training Set Evaluation:\")\n# print(f\"MAE: {mae_train:.2f}\")\n# print(f\"MSE: {mse_train:.2f}\")\n# print(f\"RMSE: {rmse_train:.2f}\")\n# print(f\"rmsle: {rmsle_train:.2f}\")\n\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-31T08:29:31.208292Z","iopub.execute_input":"2024-12-31T08:29:31.208608Z","iopub.status.idle":"2024-12-31T08:29:34.072465Z","shell.execute_reply.started":"2024-12-31T08:29:31.208583Z","shell.execute_reply":"2024-12-31T08:29:34.071749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}