{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30806,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Importing required libraries\nimport jax\nimport jax.numpy as jnp\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n#Import sklearn\nfrom sklearn.model_selection import train_test_split, cross_val_score, GridSearchCV\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import r2_score, mean_squared_error, mean_absolute_error\n\n#Import Models\nfrom sklearn.linear_model import LinearRegression, Ridge, Lasso\nfrom sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor\nfrom xgboost import XGBRegressor","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:45:54.70607Z","iopub.execute_input":"2024-12-08T14:45:54.706542Z","iopub.status.idle":"2024-12-08T14:45:59.146837Z","shell.execute_reply.started":"2024-12-08T14:45:54.706494Z","shell.execute_reply":"2024-12-08T14:45:59.145709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndata = data.dropna(subset=['Premium Amount'])\ndata.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:45:59.148917Z","iopub.execute_input":"2024-12-08T14:45:59.14943Z","iopub.status.idle":"2024-12-08T14:46:05.889989Z","shell.execute_reply.started":"2024-12-08T14:45:59.149394Z","shell.execute_reply":"2024-12-08T14:46:05.888897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:46:05.891186Z","iopub.execute_input":"2024-12-08T14:46:05.891566Z","iopub.status.idle":"2024-12-08T14:46:06.670939Z","shell.execute_reply.started":"2024-12-08T14:46:05.891524Z","shell.execute_reply":"2024-12-08T14:46:06.669804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop('id', axis=1, inplace=True)\ndata.drop('Policy Start Date', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:46:06.672241Z","iopub.execute_input":"2024-12-08T14:46:06.672585Z","iopub.status.idle":"2024-12-08T14:46:07.036109Z","shell.execute_reply.started":"2024-12-08T14:46:06.672554Z","shell.execute_reply":"2024-12-08T14:46:07.034743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Seperating the features\nNUMERICAL_COLUMNS = []\nCATEGORICAL_COLUMNS = []\n\nfor name in data.columns:\n    if data[name].dtype == 'object':\n        CATEGORICAL_COLUMNS.append(name)\n    else:\n        NUMERICAL_COLUMNS.append(name)\n\nNUMERICAL_COLUMNS.remove('Premium Amount')\nFEATURES = NUMERICAL_COLUMNS + CATEGORICAL_COLUMNS\nprint('Features:', data.columns)\nprint('Numerical Columns:', NUMERICAL_COLUMNS)\nprint('Categorical Columns:', CATEGORICAL_COLUMNS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:46:07.038557Z","iopub.execute_input":"2024-12-08T14:46:07.038909Z","iopub.status.idle":"2024-12-08T14:46:07.047537Z","shell.execute_reply.started":"2024-12-08T14:46:07.038877Z","shell.execute_reply":"2024-12-08T14:46:07.046391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Converting categorical values to numerical values\nle = LabelEncoder()\ndata_processed = data.copy()\nfor name in CATEGORICAL_COLUMNS:\n    data_processed[name] = data_processed[name].fillna('missing')\n    data_processed[name] = le.fit_transform(data_processed[name])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:46:07.048825Z","iopub.execute_input":"2024-12-08T14:46:07.049209Z","iopub.status.idle":"2024-12-08T14:46:10.227486Z","shell.execute_reply.started":"2024-12-08T14:46:07.049175Z","shell.execute_reply":"2024-12-08T14:46:10.22625Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Creating a preprocessing pipeline\npreprocessor = Pipeline([\n    ('imputer', SimpleImputer(strategy='median')),\n    ('scaler', StandardScaler())\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:46:10.228863Z","iopub.execute_input":"2024-12-08T14:46:10.229168Z","iopub.status.idle":"2024-12-08T14:46:10.234064Z","shell.execute_reply.started":"2024-12-08T14:46:10.229139Z","shell.execute_reply":"2024-12-08T14:46:10.232909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = data_processed.drop('Premium Amount', axis=1)\ny = data_processed['Premium Amount']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:46:10.235546Z","iopub.execute_input":"2024-12-08T14:46:10.235886Z","iopub.status.idle":"2024-12-08T14:46:10.341609Z","shell.execute_reply.started":"2024-12-08T14:46:10.235854Z","shell.execute_reply":"2024-12-08T14:46:10.340513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\nX_processed_train = preprocessor.fit_transform(X_train)\nX_processed_test = preprocessor.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T14:46:10.342832Z","iopub.execute_input":"2024-12-08T14:46:10.343233Z","iopub.status.idle":"2024-12-08T14:46:13.984532Z","shell.execute_reply.started":"2024-12-08T14:46:10.343191Z","shell.execute_reply":"2024-12-08T14:46:13.983397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = LinearRegression()\nmodel.fit(X_processed_train, y_train)\ny_pred = model.predict(X_processed_test)\n\nr2 = r2_score(y_test, y_pred)\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nprint('R2 Score:', r2)\nprint('Mean Squared Error:', mse)\nprint('Mean Absolute Error:', mae)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Ridge()\nmodel.fit(X_processed_train, y_train)\ny_pred = model.predict(X_processed_test)\n\nr2 = r2_score(y_test, y_pred)\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nprint('R2 Score:', r2)\nprint('Mean Squared Error:', mse)\nprint('Mean Absolute Error:', mae)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Lasso()\nmodel.fit(X_processed_train, y_train)\ny_pred = model.predict(X_processed_test)\n\nr2 = r2_score(y_test, y_pred)\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nprint('R2 Score:', r2)\nprint('Mean Squared Error:', mse)\nprint('Mean Absolute Error:', mae)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = RandomForestRegressor(random_state=42)\nmodel.fit(X_processed_train, y_train)\ny_pred = model.predict(X_processed_test)\n\nr2 = r2_score(y_test, y_pred)\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nprint('R2 Score:', r2)\nprint('Mean Squared Error:', mse)\nprint('Mean Absolute Error:', mae)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = GradientBoostingRegressor(random_state=42)\nmodel.fit(X_processed_train, y_train)\ny_pred = model.predict(X_processed_test)\n\nr2 = r2_score(y_test, y_pred)\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nprint('R2 Score:', r2)\nprint('Mean Squared Error:', mse)\nprint('Mean Absolute Error:', mae)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = XGBRegressor(random_state=42)\nmodel.fit(X_processed_train, y_train)\ny_pred = model.predict(X_processed_test)\n\nr2 = r2_score(y_test, y_pred)\nmse = mean_squared_error(y_test, y_pred)\nmae = mean_absolute_error(y_test, y_pred)\nprint('R2 Score:', r2)\nprint('Mean Squared Error:', mse)\nprint('Mean Absolute Error:', mae)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 8))\ncorrelation_matrix = data[NUMERICAL_COLUMNS + ['Premium Amount']].corr()\nsns.heatmap(correlation_matrix, cmap='coolwarm')\nplt.title('Correlation Matrix')\nplt.show()","metadata":{},"outputs":[],"execution_count":null}]}