{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.15","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"provenance":[],"gpuType":"V28"},"accelerator":"TPU","kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"id":"lAJBit4t0vsx","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:21:50.461788Z","iopub.execute_input":"2024-12-08T15:21:50.462169Z","iopub.status.idle":"2024-12-08T15:21:52.505664Z","shell.execute_reply.started":"2024-12-08T15:21:50.462135Z","shell.execute_reply":"2024-12-08T15:21:52.504882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train=pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest=pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nprint(\"train shape\",train.shape)\nprint(\"test shape\",test.shape)\n","metadata":{"id":"s5RQlk5Q2mEO","outputId":"9466cbf8-814e-4c82-b6b2-4867612dd0ec","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:21:52.507019Z","iopub.execute_input":"2024-12-08T15:21:52.507335Z","iopub.status.idle":"2024-12-08T15:22:00.724453Z","shell.execute_reply.started":"2024-12-08T15:21:52.507306Z","shell.execute_reply":"2024-12-08T15:22:00.723537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"id":"orVsCYk427y1","outputId":"7c5bb8c1-ebfa-4610-f2d7-3d0d43bfecd0","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:00.725633Z","iopub.execute_input":"2024-12-08T15:22:00.725917Z","iopub.status.idle":"2024-12-08T15:22:00.755665Z","shell.execute_reply.started":"2024-12-08T15:22:00.725888Z","shell.execute_reply":"2024-12-08T15:22:00.754599Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()/len(train)*100","metadata":{"id":"jLkphl_QE-bA","outputId":"7379c63a-a016-4d44-d795-294a2b935f85","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:00.757555Z","iopub.execute_input":"2024-12-08T15:22:00.757838Z","iopub.status.idle":"2024-12-08T15:22:01.32847Z","shell.execute_reply.started":"2024-12-08T15:22:00.75781Z","shell.execute_reply":"2024-12-08T15:22:01.32752Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Correlation analysis**","metadata":{"id":"ZnCanVWFn9NS"}},{"cell_type":"code","source":"numerical_cols=train.select_dtypes(include=['float','int'])\n# Calculate skewness for numerical columns\nskewness = numerical_cols.skew()\n\n# Display the results\nprint(skewness)","metadata":{"id":"Wu3wFxjIFB3x","outputId":"382e47e2-54a4-4fd7-e1bf-c2e2c1745452","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:01.329632Z","iopub.execute_input":"2024-12-08T15:22:01.32991Z","iopub.status.idle":"2024-12-08T15:22:01.489851Z","shell.execute_reply.started":"2024-12-08T15:22:01.329882Z","shell.execute_reply":"2024-12-08T15:22:01.488809Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# For numerical columns","metadata":{"id":"VAX4_QIsqco_"}},{"cell_type":"code","source":"target=train['Premium Amount']\n# Define the columns for plotting (only the x-axis columns)\n# Get the name of the target column\ntarget_column_name = target.name  # Get the name 'Premium Amount'\n\n# Drop the target column by name\nnumerical_cols = numerical_cols.drop(columns=[target_column_name])\n\n","metadata":{"id":"xWgiuHhZF3jm","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:01.491355Z","iopub.execute_input":"2024-12-08T15:22:01.491801Z","iopub.status.idle":"2024-12-08T15:22:01.526805Z","shell.execute_reply.started":"2024-12-08T15:22:01.491762Z","shell.execute_reply":"2024-12-08T15:22:01.525733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n\n# Target column (assuming it is named 'target')\ntarget_column = target  # Replace with the actual target column name\n\n# Create subplots based on the number of columns\nfig, axes = plt.subplots(len(numerical_cols.columns), 1, figsize=(10, 20)) # Use len(numerical_cols.columns)\n\n# Loop through each column and plot against the target\nfor i, x_col in enumerate(numerical_cols.columns): # Iterate through column names\n    sns.scatterplot(x=x_col, y=target_column, data=train, ax=axes[i])\n    axes[i].set_title(f'{x_col} vs {target_column.name}') # Use target_column.name for title\n\nplt.tight_layout()  # Adjust layout to prevent overlap\nplt.show()","metadata":{"id":"9UQQBSv0Ffg1","outputId":"4513b5c6-52bc-466a-8c2a-a257c68881d1","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:01.528054Z","iopub.execute_input":"2024-12-08T15:22:01.528314Z","iopub.status.idle":"2024-12-08T15:22:27.568601Z","shell.execute_reply.started":"2024-12-08T15:22:01.528289Z","shell.execute_reply":"2024-12-08T15:22:27.567849Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# For Categorical Columns","metadata":{"id":"-H-tda4iqiIg"}},{"cell_type":"code","source":"categorical_cols=train.select_dtypes(include=['object'])\n\n# Target column (assuming it is named 'target')\ntarget_column = target  # Replace with the actual target column name\n\n# Create subplots based on the number of columns\nfig, axes = plt.subplots(len(categorical_cols.columns), 1, figsize=(10, 20)) # Use len(numerical_cols.columns)\n\n# Loop through each column and plot against the target\nfor i, x_col in enumerate(numerical_cols.columns): # Iterate through column names\n    sns.histplot(x=x_col, y=target_column, data=train, ax=axes[i])\n    axes[i].set_title(f'{x_col} vs {target_column.name}') # Use target_column.name for title\n\nplt.tight_layout()  # Adjust layout to prevent overlap\nplt.show()","metadata":{"id":"z93Xq5nzHill","outputId":"93b1fe4c-d1ea-46ed-a3a6-d38c1d298d5a","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:27.569721Z","iopub.execute_input":"2024-12-08T15:22:27.570129Z","iopub.status.idle":"2024-12-08T15:22:38.464345Z","shell.execute_reply.started":"2024-12-08T15:22:27.5701Z","shell.execute_reply":"2024-12-08T15:22:38.463549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_cols.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:38.465424Z","iopub.execute_input":"2024-12-08T15:22:38.465769Z","iopub.status.idle":"2024-12-08T15:22:39.031902Z","shell.execute_reply.started":"2024-12-08T15:22:38.46574Z","shell.execute_reply":"2024-12-08T15:22:39.031213Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n### **Use Spearman Correlation**\n\nI chose **Spearman correlation** to analyze the relationship between each feature and the target, **Premium Amount**, because:\n\n1. **Skewed Data**:  \n   Many features, including **Premium Amount**, have skewed distributions. Spearman works well with skewed data by focusing on the **rank** (order) of values rather than their exact values.\n\n2. **Non-linear Relationships**:  \n   The data shows **non-linear** relationships between features and **Premium Amount** (as seen in scatter plots). Spearman correlation can capture these types of relationships, unlike Pearson, which assumes linearity.\n\n3. **Outlier Resilience**:  \n   Spearman is less affected by outliers, making it more reliable for data that contains extreme values.\n\n4. **Categorical Data**:  \n   For categorical features, **histplots** can help visualize how different categories are distributed and how they might influence the **Premium Amount**.\n\nBy using **Spearman correlation**, we can better understand how each feature relates to **Premium Amount**, even in cases of **non-linear** patterns.\n","metadata":{"id":"nuidyN_Tp1FN"}},{"cell_type":"code","source":"correlation = numerical_cols.corrwith(target, method='spearman').sort_values(ascending=False)\nprint(correlation)","metadata":{"id":"9TNQLIb4JNGA","outputId":"29234418-12ef-40ff-b861-d11562cdbded","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:39.035411Z","iopub.execute_input":"2024-12-08T15:22:39.03573Z","iopub.status.idle":"2024-12-08T15:22:40.650386Z","shell.execute_reply.started":"2024-12-08T15:22:39.035704Z","shell.execute_reply":"2024-12-08T15:22:40.649694Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### **Interpretation of the Correlations:**\n\n---\n\n#### **Low Positive Correlation:**\n- **Previous Claims (0.044560)** and **Health Score (0.016084)**  \n  These features have low positive correlations with the target. This indicates a very weak positive relationship, meaning they don't significantly impact the target but could still be worth considering, depending on the context.So we will keep them\n\n---\n\n#### **Very Weak or No Correlation:**\n- **Vehicle Age (0.000862)** and **ID (0.000304)**  \n  These features have near-zero correlation with the target. They do not provide meaningful predictive power and might be candidates for exclusion, especially if they don’t logically contribute to the prediction task.So we will drop them.\n\n---\n\n\n","metadata":{"id":"gBrxYWjTnGtc"}},{"cell_type":"code","source":"train.columns","metadata":{"id":"CHmgPZoFta0k","outputId":"ad0958c8-92b0-47fb-98b6-7811d931396b","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:40.651331Z","iopub.execute_input":"2024-12-08T15:22:40.651616Z","iopub.status.idle":"2024-12-08T15:22:40.656659Z","shell.execute_reply.started":"2024-12-08T15:22:40.651589Z","shell.execute_reply":"2024-12-08T15:22:40.656085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_processed=train.drop(['Vehicle Age','id'],axis=1)\ntest_processed=test.drop(['Vehicle Age','id'],axis=1)","metadata":{"id":"GfDl7wActiGA","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:40.657581Z","iopub.execute_input":"2024-12-08T15:22:40.657869Z","iopub.status.idle":"2024-12-08T15:22:40.923322Z","shell.execute_reply.started":"2024-12-08T15:22:40.657842Z","shell.execute_reply":"2024-12-08T15:22:40.922539Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n#### **Weak Negative Correlation:**\n- **Insurance Duration (-0.000045)** and **Number of Dependents (-0.001607)**  \n  These features show a very weak negative correlation, indicating a slight inverse relationship with the target. They have almost no predictive value based on this correlation.\n\n---\n\n#### **Moderate Negative Correlation:**\n- **Age (-0.002244)**, **Credit Score (-0.043883)**, and **Annual Income (-0.061550)**  \n  These features exhibit weak to moderate negative correlations. This suggests that as these features increase, the target variable tends to decrease slightly. Although the correlations are still relatively weak, they might contain some predictive information.","metadata":{"id":"f9l7b0CHrH28"}},{"cell_type":"code","source":"train_processed.columns","metadata":{"id":"E9xDqAH9uK-a","outputId":"47fe23d9-697a-4552-e352-b1ccddd013cb","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:40.92451Z","iopub.execute_input":"2024-12-08T15:22:40.9248Z","iopub.status.idle":"2024-12-08T15:22:40.9298Z","shell.execute_reply.started":"2024-12-08T15:22:40.924771Z","shell.execute_reply":"2024-12-08T15:22:40.929155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_cols.info()","metadata":{"id":"Lt_2J2oVwKMJ","outputId":"f8132211-29e5-4d20-c7af-ca016b11269d","trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:40.930715Z","iopub.execute_input":"2024-12-08T15:22:40.930959Z","iopub.status.idle":"2024-12-08T15:22:41.501027Z","shell.execute_reply.started":"2024-12-08T15:22:40.930933Z","shell.execute_reply":"2024-12-08T15:22:41.500203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Feature** **Engineering**\n","metadata":{"id":"eUNnGjLiSJWQ"}},{"cell_type":"code","source":"def parse_date_time_column(df, column_name, new_column_name=\"Policy Start Date1\"):\n   \n    # Remove the word 'time' and strip extra spaces from the original column\n    df[column_name] = df[column_name].str.replace('Policy Start Date', '', regex=False).str.strip()\n    # Parse the cleaned-up date-time strings into datetime objects\n    df[new_column_name] = pd.to_datetime(df[column_name], format='%Y-%m-%d %H:%M:%S', errors='coerce')\n\n    # Drop the original time column\n    df.drop(columns=[column_name], inplace=True)\n\n    return df\n\n# Call the function\ntrain_processed = parse_date_time_column(train, 'Policy Start Date')\ntest_processed = parse_date_time_column(test,'Policy Start Date')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:41.502204Z","iopub.execute_input":"2024-12-08T15:22:41.502529Z","iopub.status.idle":"2024-12-08T15:22:46.331011Z","shell.execute_reply.started":"2024-12-08T15:22:41.502466Z","shell.execute_reply":"2024-12-08T15:22:46.330203Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Impute missing values","metadata":{"id":"7T1rosIiSURe"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error\n\n# Assuming your dataset is in 'train_processed' and target column is 'Premium Amount'\nX = train_processed.drop('Premium Amount', axis=1)  # Features\ny = train_processed['Premium Amount']  # Target\n\n# Remove rows where target 'Premium Amount' is NaN\nX = X[~y.isna()]\ny = y.dropna()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:46.33205Z","iopub.execute_input":"2024-12-08T15:22:46.332304Z","iopub.status.idle":"2024-12-08T15:22:47.714142Z","shell.execute_reply.started":"2024-12-08T15:22:46.332278Z","shell.execute_reply":"2024-12-08T15:22:47.713305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_processed.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:47.715455Z","iopub.execute_input":"2024-12-08T15:22:47.716066Z","iopub.status.idle":"2024-12-08T15:22:48.250768Z","shell.execute_reply.started":"2024-12-08T15:22:47.716026Z","shell.execute_reply":"2024-12-08T15:22:48.24998Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test=test_processed","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:22:48.251926Z","iopub.execute_input":"2024-12-08T15:22:48.25222Z","iopub.status.idle":"2024-12-08T15:22:48.255956Z","shell.execute_reply.started":"2024-12-08T15:22:48.252191Z","shell.execute_reply":"2024-12-08T15:22:48.255173Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training data","metadata":{}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n\n# Identify categorical and numerical columns\ncategorical_columns = ['Gender', 'Marital Status', 'Education Level', 'Location', 'Occupation',\n                       'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']\nnumerical_cols = X.select_dtypes(include=['float', 'int']).columns\n\n# Step 1: Impute numerical columns\nimputer_median = SimpleImputer(strategy='median')\nX_numerical_imputed = pd.DataFrame(imputer_median.fit_transform(X[numerical_cols]), columns=numerical_cols)\n\n# Step 2: Impute categorical columns with the most frequent value (mode), excluding problematic column\nimputer_mode = SimpleImputer(strategy='most_frequent')\n\n# First, check if 'Policy Start Date1' has any non-missing values\nif 'Policy Start Date1' in categorical_columns:\n    # Drop 'Policy Start Date1' from the categorical columns for imputation\n    categorical_columns_without_policy_date = [col for col in categorical_columns if col != 'Policy Start Date1']\nelse:\n    categorical_columns_without_policy_date = categorical_columns\n\n# Impute the remaining categorical columns\nX_categorical_imputed = pd.DataFrame(imputer_mode.fit_transform(X[categorical_columns_without_policy_date]), \n                                     columns=categorical_columns_without_policy_date)\n\n# Step 3: Reattach the problematic 'Policy Start Date1' column (if it exists) with the imputed categorical data\nif 'Policy Start Date1' in categorical_columns:\n    X_categorical_imputed['Policy Start Date1'] = X['Policy Start Date1']\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:27:42.462869Z","iopub.execute_input":"2024-12-08T15:27:42.46327Z","iopub.status.idle":"2024-12-08T15:27:45.56765Z","shell.execute_reply.started":"2024-12-08T15:27:42.463237Z","shell.execute_reply":"2024-12-08T15:27:45.566838Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Tresting data","metadata":{}},{"cell_type":"code","source":"\n\n# Step 1: Impute missing numerical values in the test data using the fitted imputer\nX_numerical_imputed_test = pd.DataFrame(imputer_median.transform(test[numerical_cols]), columns=numerical_cols)\n\n# Convert transformed numerical data back to DataFrame\nX_numerical_imputed_test = pd.DataFrame(X_numerical_imputed_test, columns=numerical_cols)\n\n# Step 3: Impute categorical columns in the test data using the fitted imputer\nX_categorical_imputed_test = pd.DataFrame(imputer_mode.transform(test[categorical_columns]), columns=categorical_columns)\n\n# Verify the transformed test data\nprint(\"Transformed Test Data (Numerical):\")\nprint(X_numerical_imputed_test.head())\n\nprint(\"\\nTransformed Test Data (Categorical):\")\nprint(X_categorical_imputed_test.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:27:45.569116Z","iopub.execute_input":"2024-12-08T15:27:45.569376Z","iopub.status.idle":"2024-12-08T15:27:46.136322Z","shell.execute_reply.started":"2024-12-08T15:27:45.569351Z","shell.execute_reply":"2024-12-08T15:27:46.135514Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PowerTransformer for normal distribution","metadata":{}},{"cell_type":"markdown","source":" # Training data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer\n\n# Apply Power Transform on the numerical features\npower_transformer = PowerTransformer(method='yeo-johnson', standardize=True)  # 'box-cox' for positive-only data\n\n# Fit and transform the numerical data\nX_numerical_transformed = power_transformer.fit_transform(X_numerical_imputed)\n\n# Convert back to DataFrame\nX_numerical_imputed = pd.DataFrame(X_numerical_transformed, columns=numerical_cols)\n\n# Verify transformation\nprint(X_numerical_imputed.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:27:46.137276Z","iopub.execute_input":"2024-12-08T15:27:46.137574Z","iopub.status.idle":"2024-12-08T15:27:51.185232Z","shell.execute_reply.started":"2024-12-08T15:27:46.137527Z","shell.execute_reply":"2024-12-08T15:27:51.18447Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Testing data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import PowerTransformer\n\n# Apply Power Transform on the numerical features\npower_transformer = PowerTransformer(method='yeo-johnson', standardize=True)  # 'box-cox' for positive-only data\n\n# Fit and transform the numerical data\nX_numerical_transformed = power_transformer.fit_transform(X_numerical_imputed_test)\n\n# Convert back to DataFrame\nX_numerical_imputed_test = pd.DataFrame(X_numerical_transformed, columns=numerical_cols)\n\n# Verify transformation\nprint(X_numerical_imputed_test.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:27:51.187187Z","iopub.execute_input":"2024-12-08T15:27:51.187518Z","iopub.status.idle":"2024-12-08T15:27:53.941476Z","shell.execute_reply.started":"2024-12-08T15:27:51.18746Z","shell.execute_reply":"2024-12-08T15:27:53.940629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# RFE for feature selection","metadata":{"id":"VGggKNmQRkUc"}},{"cell_type":"code","source":"y = y.values.ravel()  # Converts the DataFrame into a 1D numpy array\ny","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:27:53.942688Z","iopub.execute_input":"2024-12-08T15:27:53.943016Z","iopub.status.idle":"2024-12-08T15:27:53.948677Z","shell.execute_reply.started":"2024-12-08T15:27:53.94298Z","shell.execute_reply":"2024-12-08T15:27:53.947712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install xgboost","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:27:53.949651Z","iopub.execute_input":"2024-12-08T15:27:53.949894Z","iopub.status.idle":"2024-12-08T15:28:05.97248Z","shell.execute_reply.started":"2024-12-08T15:27:53.949869Z","shell.execute_reply":"2024-12-08T15:28:05.971307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_selection import RFE\nfrom xgboost import XGBRegressor\nfrom sklearn.preprocessing import LabelEncoder\n\n# Apply LabelEncoder to each column in X_categorical_imputed\nlabel_encoder = LabelEncoder()\nfor col in X_categorical_imputed.columns:\n    X_categorical_imputed[col] = label_encoder.fit_transform(X_categorical_imputed[col])\n\n# Combine numerical and categorical imputed data back together\nX_imputed = pd.concat([X_numerical_imputed, X_categorical_imputed], axis=1)\n\n# Split data\nX_train, X_test, y_train, y_test = train_test_split(X_imputed, y, test_size=0.2, random_state=42)\n\n# Define the model (XGBoost)\nmodel = XGBRegressor()\n\n# Apply RFE for feature selection\nselector = RFE(estimator=model)\nselector = selector.fit(X_train, y_train)\n\n# Get selected features\nselected_features = X_train.columns[selector.support_]\nprint(\"Selected Features:\", selected_features)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:28:05.974048Z","iopub.execute_input":"2024-12-08T15:28:05.974342Z","iopub.status.idle":"2024-12-08T15:28:26.490039Z","shell.execute_reply.started":"2024-12-08T15:28:05.974312Z","shell.execute_reply":"2024-12-08T15:28:26.489209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the model using only the selected features\nX_train_selected = X_train[selected_features]\nX_test_selected = X_test[selected_features]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:28:26.491399Z","iopub.execute_input":"2024-12-08T15:28:26.492009Z","iopub.status.idle":"2024-12-08T15:28:26.5245Z","shell.execute_reply.started":"2024-12-08T15:28:26.491971Z","shell.execute_reply":"2024-12-08T15:28:26.523389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install optuna catboost lightgbm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:28:26.52585Z","iopub.execute_input":"2024-12-08T15:28:26.526121Z","iopub.status.idle":"2024-12-08T15:28:53.889779Z","shell.execute_reply.started":"2024-12-08T15:28:26.526095Z","shell.execute_reply":"2024-12-08T15:28:53.888878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import optuna\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.metrics import mean_squared_error\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\n\n# --- Define the Optuna Objective Function ---\ndef objective(trial):\n    # XGBoost hyperparameters\n    xgb_params = {\n        'n_estimators': trial.suggest_int('xgb_n_estimators', 100, 500),\n        'learning_rate': trial.suggest_float('xgb_learning_rate', 0.01, 0.3),\n        'max_depth': trial.suggest_int('xgb_max_depth', 3, 10),\n        'subsample': trial.suggest_float('xgb_subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_float('xgb_colsample_bytree', 0.5, 1.0),\n        'random_state': 42\n    }\n    \n    # LightGBM hyperparameters\n    lgbm_params = {\n        'n_estimators': trial.suggest_int('lgbm_n_estimators', 100, 500),\n        'learning_rate': trial.suggest_float('lgbm_learning_rate', 0.01, 0.3),\n        'max_depth': trial.suggest_int('lgbm_max_depth', -1, 15),\n        'num_leaves': trial.suggest_int('lgbm_num_leaves', 20, 100),\n        'random_state': 42\n    }\n\n    # CatBoost hyperparameters\n    cat_params = {\n        'iterations': trial.suggest_int('cat_iterations', 100, 500),\n        'learning_rate': trial.suggest_float('cat_learning_rate', 0.01, 0.3),\n        'depth': trial.suggest_int('cat_depth', 4, 10),\n        'random_seed': 42,\n        'verbose': 0\n    }\n\n    # Initialize models\n    xgb_model = XGBRegressor(**xgb_params)\n    lgbm_model = LGBMRegressor(**lgbm_params)\n    cat_model = CatBoostRegressor(**cat_params)\n\n    # Create Voting Ensemble\n    voting_ensemble = VotingRegressor(\n        estimators=[\n            ('xgb', xgb_model),\n            ('lgbm', lgbm_model),\n            ('cat', cat_model)\n        ]\n    )\n\n    # Train Voting Ensemble\n    voting_ensemble.fit(X_train, y_train)\n\n    # Predict and calculate RMSE\n    y_pred = voting_ensemble.predict(X_test)\n    rmse = mean_squared_error(y_test, y_pred, squared=False)\n    \n    return rmse\n\n# --- Run the Optuna Study ---\nstudy = optuna.create_study(direction='minimize')\nstudy.optimize(objective, n_trials=50)\n\n# Print best hyperparameters\nprint(\"Best Hyperparameters:\")\nprint(study.best_params)\n\n# --- Train Final Models with Best Hyperparameters ---\nbest_params = study.best_params\n\n# Extract best hyperparameters for each model\nxgb_best_params = {\n    'n_estimators': best_params['xgb_n_estimators'],\n    'learning_rate': best_params['xgb_learning_rate'],\n    'max_depth': best_params['xgb_max_depth'],\n    'subsample': best_params['xgb_subsample'],\n    'colsample_bytree': best_params['xgb_colsample_bytree'],\n    'random_state': 42\n}\n\nlgbm_best_params = {\n    'n_estimators': best_params['lgbm_n_estimators'],\n    'learning_rate': best_params['lgbm_learning_rate'],\n    'max_depth': best_params['lgbm_max_depth'],\n    'num_leaves': best_params['lgbm_num_leaves'],\n    'random_state': 42\n}\n\ncat_best_params = {\n    'iterations': best_params['cat_iterations'],\n    'learning_rate': best_params['cat_learning_rate'],\n    'depth': best_params['cat_depth'],\n    'random_seed': 42,\n    'verbose': 0\n}\n\n# Initialize models with best hyperparameters\nxgb_model = XGBRegressor(**xgb_best_params)\nlgbm_model = LGBMRegressor(**lgbm_best_params)\ncat_model = CatBoostRegressor(**cat_best_params)\n\n# Create Voting Ensemble\nvoting_ensemble = VotingRegressor(\n    estimators=[\n        ('xgb', xgb_model),\n        ('lgbm', lgbm_model),\n        ('cat', cat_model)\n    ]\n)\n\n# Fit ensemble on the entire training data\nvoting_ensemble.fit(X_train_selected, y_train)\n\n# Predict and evaluate\ny_pred = voting_ensemble.predict(X_test_selected)\nrmse = mean_squared_error(y_test, y_pred, squared=False)\nprint(f\"Final Voting Ensemble RMSE: {rmse:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:28:53.892654Z","iopub.execute_input":"2024-12-08T15:28:53.893205Z","iopub.status.idle":"2024-12-08T15:43:23.921141Z","shell.execute_reply.started":"2024-12-08T15:28:53.893171Z","shell.execute_reply":"2024-12-08T15:43:23.920073Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# For test data","metadata":{}},{"cell_type":"code","source":"# Apply LabelEncoder to each column in X_categorical_imputed\nlabel_encoder = LabelEncoder()\nfor col in X_categorical_imputed_test.columns:\n    X_categorical_imputed_test[col] = label_encoder.fit_transform(X_categorical_imputed_test[col])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:43:23.922592Z","iopub.execute_input":"2024-12-08T15:43:23.922988Z","iopub.status.idle":"2024-12-08T15:43:25.388674Z","shell.execute_reply.started":"2024-12-08T15:43:23.922955Z","shell.execute_reply":"2024-12-08T15:43:25.38782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Combine the imputed numerical and categorical data for the test dataset\nX_combined_test = pd.concat([X_numerical_imputed_test, X_categorical_imputed_test], axis=1)\n\n# Verify the combined test data\nprint(\"Combined Test Data:\")\nprint(X_combined_test.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:43:25.389769Z","iopub.execute_input":"2024-12-08T15:43:25.390054Z","iopub.status.idle":"2024-12-08T15:43:25.482367Z","shell.execute_reply.started":"2024-12-08T15:43:25.390026Z","shell.execute_reply":"2024-12-08T15:43:25.481536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor\nfrom sklearn.metrics import mean_squared_error\n\n# Assuming best_trial is obtained from an Optuna study\nbest_model_name = best_trial.params['model']\n\n# Initialize the appropriate model based on the best trial\nif best_model_name == \"XGBoost\":\n    model = XGBRegressor(**{k: v for k, v in best_trial.params.items() if k != 'model'})\nelif best_model_name == \"LightGBM\":\n    model = LGBMRegressor(**{k: v for k, v in best_trial.params.items() if k != 'model'})\nelse:  # CatBoost\n    # Extract the parameters correctly for CatBoost\n    catboost_params = {k: v for k, v in best_trial.params.items() if k != 'model'}\n    model = CatBoostRegressor(**catboost_params, verbose=0)  # Suppress verbose output\n\n# Fit the model to the entire training set\nmodel.fit(X_train_selected, y_train)\n\n# Predict on the test set\ny_pred = model.predict(X_test_selected)\n\n# Evaluate the model\nrmse = np.sqrt(mean_squared_error(y_test, y_pred))\nprint(f\"RMSE on Test Set: {rmse:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:47:39.433871Z","iopub.execute_input":"2024-12-08T15:47:39.434381Z","iopub.status.idle":"2024-12-08T15:47:39.484598Z","shell.execute_reply.started":"2024-12-08T15:47:39.434333Z","shell.execute_reply":"2024-12-08T15:47:39.48361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Fit ensemble on the entire training data with error handling\ntry:\n    voting_ensemble.fit(X_train_selected, y_train)\n    print(\"Voting ensemble fitted successfully.\")\nexcept Exception as e:\n    print(f\"Error during fitting: {e}\")\n\n# Predict and evaluate\ntry:\n    y_pred = voting_ensemble.predict(X_combined_test)\n    rmse = mean_squared_error(y_test, y_pred, squared=False)\n    print(f\"Final Voting Ensemble RMSE: {rmse:.4f}\")\nexcept Exception as e:\n    print(f\"Error during prediction: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:53:05.041581Z","iopub.execute_input":"2024-12-08T15:53:05.042016Z","iopub.status.idle":"2024-12-08T15:53:20.341628Z","shell.execute_reply.started":"2024-12-08T15:53:05.041979Z","shell.execute_reply":"2024-12-08T15:53:20.340543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test data\ny_pred = model.predict(X_combined_test)  # Assuming X_test_selected is the test set with selected features\n\n# Print the length of predictions and the first few predictions\nprint(len(y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:48:21.363647Z","iopub.execute_input":"2024-12-08T15:48:21.363992Z","iopub.status.idle":"2024-12-08T15:48:21.624081Z","shell.execute_reply.started":"2024-12-08T15:48:21.363963Z","shell.execute_reply":"2024-12-08T15:48:21.623168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.DataFrame()\nt = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\ndf['id'] = t['id']\ndf['Premium Amount'] = y_pred\ndf.to_csv('submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:53:40.859014Z","iopub.execute_input":"2024-12-08T15:53:40.860164Z","iopub.status.idle":"2024-12-08T15:53:44.915694Z","shell.execute_reply.started":"2024-12-08T15:53:40.860124Z","shell.execute_reply":"2024-12-08T15:53:44.914572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-08T15:53:49.65927Z","iopub.execute_input":"2024-12-08T15:53:49.66033Z","iopub.status.idle":"2024-12-08T15:53:49.670714Z","shell.execute_reply.started":"2024-12-08T15:53:49.660287Z","shell.execute_reply":"2024-12-08T15:53:49.669837Z"}},"outputs":[],"execution_count":null}]}