{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:51:47.414946Z","iopub.execute_input":"2024-12-09T12:51:47.415388Z","iopub.status.idle":"2024-12-09T12:51:48.747472Z","shell.execute_reply.started":"2024-12-09T12:51:47.415351Z","shell.execute_reply":"2024-12-09T12:51:48.74567Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd;\nimport numpy as np;\nimport matplotlib.pyplot as plt ;\nimport seaborn as sns;\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.ensemble import RandomForestRegressor\n\nimport warnings\nwarnings.filterwarnings('ignore',category=UserWarning, module='lightgbm')\nos.environ[\"PYTHONWARNINGS\"] = \"ignore\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T13:32:22.024613Z","iopub.status.idle":"2024-12-09T13:32:22.025228Z","shell.execute_reply.started":"2024-12-09T13:32:22.024928Z","shell.execute_reply":"2024-12-09T13:32:22.024961Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import Data Set:","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv');\ntest_data = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\n\ntrain_data.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:53:59.246987Z","iopub.execute_input":"2024-12-09T12:53:59.24761Z","iopub.status.idle":"2024-12-09T12:54:11.002924Z","shell.execute_reply.started":"2024-12-09T12:53:59.247545Z","shell.execute_reply":"2024-12-09T12:54:11.001794Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis & Data Pre-Processing","metadata":{}},{"cell_type":"code","source":"train_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:54:43.481061Z","iopub.execute_input":"2024-12-09T12:54:43.481439Z","iopub.status.idle":"2024-12-09T12:54:44.186877Z","shell.execute_reply.started":"2024-12-09T12:54:43.481409Z","shell.execute_reply":"2024-12-09T12:54:44.185375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:54:55.612001Z","iopub.execute_input":"2024-12-09T12:54:55.612407Z","iopub.status.idle":"2024-12-09T12:54:56.325853Z","shell.execute_reply.started":"2024-12-09T12:54:55.612366Z","shell.execute_reply":"2024-12-09T12:54:56.324296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:55:07.019779Z","iopub.execute_input":"2024-12-09T12:55:07.020266Z","iopub.status.idle":"2024-12-09T12:55:07.675152Z","shell.execute_reply.started":"2024-12-09T12:55:07.020227Z","shell.execute_reply":"2024-12-09T12:55:07.673886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:55:17.824344Z","iopub.execute_input":"2024-12-09T12:55:17.824822Z","iopub.status.idle":"2024-12-09T12:55:19.643111Z","shell.execute_reply.started":"2024-12-09T12:55:17.824784Z","shell.execute_reply":"2024-12-09T12:55:19.64169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Handle Missing Values:\n# Impute Numerical Values:\n\nnumerical_cols = train_data.select_dtypes(include=[np.number]).columns\nimputer = SimpleImputer(strategy = 'median')\ntrain_data[numerical_cols] = imputer.fit_transform(train_data[numerical_cols])\n\n# Impute Categrical Values:\n\ncategorical_cols = train_data.select_dtypes(exclude = [np.number]).columns\nfor col in categorical_cols:\n    train_data[col].fillna(train_data[col].mode()[0], inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:55:30.967321Z","iopub.execute_input":"2024-12-09T12:55:30.96779Z","iopub.status.idle":"2024-12-09T12:55:35.646181Z","shell.execute_reply.started":"2024-12-09T12:55:30.967754Z","shell.execute_reply":"2024-12-09T12:55:35.644982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:55:45.072174Z","iopub.execute_input":"2024-12-09T12:55:45.072549Z","iopub.status.idle":"2024-12-09T12:55:45.745642Z","shell.execute_reply.started":"2024-12-09T12:55:45.072519Z","shell.execute_reply":"2024-12-09T12:55:45.744297Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compute corelation for numerical columns:\n\nnumerical_cols = train_data.select_dtypes(include = ['float64', 'int64']).columns\n\nnumerical_corr = train_data[numerical_cols].corr(method = 'pearson')\nnumerical_corr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:55:55.649891Z","iopub.execute_input":"2024-12-09T12:55:55.650313Z","iopub.status.idle":"2024-12-09T12:55:56.24463Z","shell.execute_reply.started":"2024-12-09T12:55:55.650279Z","shell.execute_reply":"2024-12-09T12:55:56.243303Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize = (8,8))\nsns.heatmap(numerical_corr, annot = True, cmap = 'coolwarm', cbar = True, fmt = '.2f')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:56:17.207386Z","iopub.execute_input":"2024-12-09T12:56:17.207812Z","iopub.status.idle":"2024-12-09T12:56:17.854989Z","shell.execute_reply.started":"2024-12-09T12:56:17.207777Z","shell.execute_reply":"2024-12-09T12:56:17.853473Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Checking outliers:\n\nnumerical_cols = train_data.select_dtypes(include = ['float64', 'int64']).columns\n\nfor col in numerical_cols:\n    plt.figure(figsize = (8,4))\n    sns.boxplot(train_data[col])\n    plt.title(f'Box plot of {col}')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:56:42.106499Z","iopub.execute_input":"2024-12-09T12:56:42.106969Z","iopub.status.idle":"2024-12-09T12:56:45.21487Z","shell.execute_reply.started":"2024-12-09T12:56:42.106933Z","shell.execute_reply":"2024-12-09T12:56:45.213285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Columns like Annual Income,Previous Claims,and premium amount has outliers\n\n# IQR Method:\n\nQ1 = train_data[numerical_cols].quantile(0.25)\nQ3 = train_data[numerical_cols].quantile(0.75)\n\nIQR = Q3 - Q1\n\nlower_bound = Q1 - 1.5*IQR\nupper_bound = Q3 + 1.5*IQR\n\noutliers_iqr = (train_data[numerical_cols] < lower_bound) | (train_data[numerical_cols] > upper_bound)\noutliers_iqr.sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:57:05.769502Z","iopub.execute_input":"2024-12-09T12:57:05.769966Z","iopub.status.idle":"2024-12-09T12:57:06.529486Z","shell.execute_reply.started":"2024-12-09T12:57:05.769932Z","shell.execute_reply":"2024-12-09T12:57:06.528125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Capping Method to keep the outliers:\n\ndef cap_outliers(df,column):\n    Q1 = train_data[column].quantile(0.25)\n    Q3 = train_data[column].quantile(0.75)\n    IQR = Q3-Q1\n    lower_bound = Q1 - 1.5*IQR\n    upper_bound = Q3 + 1.5*IQR\n    df[column] = df[column].apply(lambda x:upper_bound if x>upper_bound else (lower_bound if x<lower_bound else x))\n    return df[column]\n\ncolumns_with_outliers = ['Annual Income','Previous Claims','Premium Amount']\n\nfor col in columns_with_outliers:\n    train_data[col] = cap_outliers(train_data,col)\n\nfor col in columns_with_outliers:\n    sns.boxplot(data = train_data[col])\n    plt.title(f'Box Plot of {col}')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:57:18.399075Z","iopub.execute_input":"2024-12-09T12:57:18.399661Z","iopub.status.idle":"2024-12-09T12:57:20.344155Z","shell.execute_reply.started":"2024-12-09T12:57:18.399607Z","shell.execute_reply":"2024-12-09T12:57:20.343108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['Education Level'] = train_data['Education Level'].map({\"Bachelor's\":0,\"Master's\":1,'High School':2,'PhD':3})\n\ncorr = train_data['Education Level'].corr(train_data['Premium Amount'])\nprint(corr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:58:26.854664Z","iopub.execute_input":"2024-12-09T12:58:26.855095Z","iopub.status.idle":"2024-12-09T12:58:26.973469Z","shell.execute_reply.started":"2024-12-09T12:58:26.855062Z","shell.execute_reply":"2024-12-09T12:58:26.972088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data['Exercise Frequency'] = train_data['Exercise Frequency'].map({'Weekly':0,'Monthly':1,'Daily':2,'Rarely':3})\n\ncorr = train_data['Exercise Frequency'].corr(train_data['Premium Amount'])\nprint(corr)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:58:37.286215Z","iopub.execute_input":"2024-12-09T12:58:37.286726Z","iopub.status.idle":"2024-12-09T12:58:37.401523Z","shell.execute_reply.started":"2024-12-09T12:58:37.286686Z","shell.execute_reply":"2024-12-09T12:58:37.400309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop columns like Location, Policy Start Date, Customer Feedback,Property Type , Education Level, Exercise Frequency from both train and test data sets.\n\ncolumns_to_drop = ['Location','Policy Start Date','Customer Feedback','Property Type', 'Education Level','Exercise Frequency']\n\ntrain_data.drop(columns = columns_to_drop , inplace = True);\ntest_data.drop(columns = columns_to_drop, inplace = True);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:58:48.79181Z","iopub.execute_input":"2024-12-09T12:58:48.792226Z","iopub.status.idle":"2024-12-09T12:58:48.991253Z","shell.execute_reply.started":"2024-12-09T12:58:48.792191Z","shell.execute_reply":"2024-12-09T12:58:48.990119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encoding for categorical columns:\n\n# One-Hot Encoding for Gender, Marital Status, Occupation, Smoking Status for both train and test data sets.\n\ntrain_data = pd.get_dummies(train_data, columns = ['Gender','Marital Status','Occupation','Smoking Status','Policy Type'], drop_first = True)\ntest_data = pd.get_dummies(test_data, columns = ['Gender','Marital Status','Occupation','Smoking Status', 'Policy Type'], drop_first = True)\n\n# Align test_data columns with train_data\n\ntest_data = test_data.reindex(columns = train_data.columns, fill_value = 0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:59:02.032342Z","iopub.execute_input":"2024-12-09T12:59:02.032808Z","iopub.status.idle":"2024-12-09T12:59:03.174697Z","shell.execute_reply.started":"2024-12-09T12:59:02.032772Z","shell.execute_reply":"2024-12-09T12:59:03.173358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"assert train_data.columns.equals(test_data.columns), \"Train and Test columns do not match!\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:59:13.38335Z","iopub.execute_input":"2024-12-09T12:59:13.384355Z","iopub.status.idle":"2024-12-09T12:59:13.389473Z","shell.execute_reply.started":"2024-12-09T12:59:13.384314Z","shell.execute_reply":"2024-12-09T12:59:13.388251Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Building the Model:","metadata":{}},{"cell_type":"code","source":"numerical_cols = train_data.select_dtypes(include = ['float64', 'int64']).columns\nnumerical_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:59:42.499553Z","iopub.execute_input":"2024-12-09T12:59:42.500562Z","iopub.status.idle":"2024-12-09T12:59:42.528044Z","shell.execute_reply.started":"2024-12-09T12:59:42.500521Z","shell.execute_reply":"2024-12-09T12:59:42.526823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define feature X and target variable y\n\nX = train_data.drop(columns = ['id','Premium Amount'])\ny = train_data['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T12:59:54.09863Z","iopub.execute_input":"2024-12-09T12:59:54.099072Z","iopub.status.idle":"2024-12-09T12:59:54.137056Z","shell.execute_reply.started":"2024-12-09T12:59:54.099036Z","shell.execute_reply":"2024-12-09T12:59:54.135864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split data into train and test sets\n\nX_train, X_test, y_train, y_test = train_test_split(X,y, test_size = 0.2, random_state = 42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T13:00:10.301023Z","iopub.execute_input":"2024-12-09T13:00:10.3015Z","iopub.status.idle":"2024-12-09T13:00:10.620049Z","shell.execute_reply.started":"2024-12-09T13:00:10.301462Z","shell.execute_reply":"2024-12-09T13:00:10.618785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Scale numerical features:\n\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T13:00:22.6257Z","iopub.execute_input":"2024-12-09T13:00:22.626154Z","iopub.status.idle":"2024-12-09T13:00:23.97822Z","shell.execute_reply.started":"2024-12-09T13:00:22.626118Z","shell.execute_reply":"2024-12-09T13:00:23.97704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install lightgbm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T13:01:08.06192Z","iopub.execute_input":"2024-12-09T13:01:08.062344Z","iopub.status.idle":"2024-12-09T13:01:19.884751Z","shell.execute_reply.started":"2024-12-09T13:01:08.062311Z","shell.execute_reply":"2024-12-09T13:01:19.883099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_squared_log_error\nimport numpy as np\n\n# Initialize the LightGBM Regressor\nlgb_model = LGBMRegressor(\n    learning_rate=0.1,\n    max_depth=7,\n    n_estimators=200,\n    subsample=0.8,\n    random_state=42,\n    verbose = -1\n)\n\n# Train the model\nlgb_model.fit(X_train_scaled, y_train)\n\n# Make predictions\ny_pred = lgb_model.predict(X_test_scaled)\n\n# Evaluate Performance\nmse = mean_squared_error(y_test, y_pred)\nrmse = np.sqrt(mse)\nr2 = r2_score(y_test, y_pred)\nrmsle = np.sqrt(mean_squared_log_error(y_test, np.maximum(0, y_pred)))  # Ensure non-negative predictions for log error\n\nprint(f\"Mean Squared Error (MSE): {mse}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse}\")\nprint(f\"R-squared (R²): {r2}\")\nprint(f\"Root Mean Squared Log Error (RMSLE): {rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T13:01:33.975819Z","iopub.execute_input":"2024-12-09T13:01:33.976261Z","iopub.status.idle":"2024-12-09T13:01:47.981152Z","shell.execute_reply.started":"2024-12-09T13:01:33.976225Z","shell.execute_reply":"2024-12-09T13:01:47.979886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Hyper parameter tuning:\n\nfrom sklearn.model_selection import RandomizedSearchCV, train_test_split\nfrom scipy.stats import uniform\nimport lightgbm as lgb\n\nwith warnings.catch_warnings():\n    warnings.simplefilter('ignore')\n    \n# Split data into training and validation\nX_train_sub, X_valid, y_train_sub, y_valid = train_test_split(X_train_scaled, y_train, test_size=0.2, random_state=42)\n\n# Define parameter distribution\nparam_dist = {\n    'n_estimators': [100, 200, 300],\n    'learning_rate': uniform(0.01, 0.1),\n    'max_depth': [3, 5, 7],\n    'num_leaves': [31,50,100],\n    'subsample': [0.6, 0.8, 1.0],\n    'colsample_bytree': [0.6, 0.8, 1.0]\n}\n\n# Model and RandomizedSearchCV\nmodel = lgb.LGBMRegressor(random_state=42)\nrandom_search = RandomizedSearchCV(estimator=model, param_distributions=param_dist, n_iter=10, cv=3, n_jobs=-1)\n\nfit_params = {\n    'eval_set': [(X_valid, y_valid)],\n    'eval_metric': 'rmsle'\n}\n\n# Fit with early stopping\nrandom_search.fit(\n    X_train_sub,\n    y_train_sub,\n    **fit_params\n)\n\nprint(\"Best Parameters:\", random_search.best_params_)\nprint(\"Best Score:\", random_search.best_score_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T13:39:35.50068Z","iopub.execute_input":"2024-12-09T13:39:35.501131Z","iopub.status.idle":"2024-12-09T13:46:12.643715Z","shell.execute_reply.started":"2024-12-09T13:39:35.501098Z","shell.execute_reply":"2024-12-09T13:46:12.642269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Retrain the model using the best parameters\nbest_model = random_search.best_estimator_\ny_pred = best_model.predict(X_test_scaled)\n\n# Evaluate the model\nfrom sklearn.metrics import mean_squared_error, r2_score\nmse = mean_squared_error(y_test, y_pred)\nrmse = np.sqrt(mse)\nrmsle = np.sqrt(mean_squared_log_error(y_test, np.maximum(0, y_pred)))  # Ensure non-negative predictions for log error\n\nr2 = r2_score(y_test, y_pred)\nprint(\"MSE:\", mse)\nprint(\"R-squared:\", r2)\nprint(f\"Root Mean Squared Error (RMSE): {rmse}\")\n\nprint(f\"Root Mean Squared Log Error (RMSLE): {rmsle}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T13:46:47.791922Z","iopub.execute_input":"2024-12-09T13:46:47.792392Z","iopub.status.idle":"2024-12-09T13:46:49.572524Z","shell.execute_reply.started":"2024-12-09T13:46:47.792354Z","shell.execute_reply":"2024-12-09T13:46:49.57113Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission:","metadata":{}},{"cell_type":"code","source":"X_test = test_data.drop(columns=['id','Premium Amount'])  # Drop non-feature columns\nX_test_scaled = scaler.transform(X_test)  #\n# Make predictions on the test set\ny_pred_test = best_model.predict(X_test_scaled)\n\n# Create the submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test_data['id'], \n    'Premium Amount': y_pred_test  # Predictions made by model\n})\n\n# Save the DataFrame to a CSV file\nsubmission.to_csv('final_submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T13:48:31.911623Z","iopub.execute_input":"2024-12-09T13:48:31.912606Z","iopub.status.idle":"2024-12-09T13:48:40.884472Z","shell.execute_reply.started":"2024-12-09T13:48:31.912529Z","shell.execute_reply":"2024-12-09T13:48:40.883235Z"}},"outputs":[],"execution_count":null}]}