{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.feature_selection import SequentialFeatureSelector\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom statsmodels.stats.outliers_influence import variance_inflation_factor\nfrom sklearn.model_selection import cross_val_score\nimport statsmodels.api as sm\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:11.853689Z","iopub.execute_input":"2024-12-22T19:56:11.8541Z","iopub.status.idle":"2024-12-22T19:56:17.633403Z","shell.execute_reply.started":"2024-12-22T19:56:11.854051Z","shell.execute_reply":"2024-12-22T19:56:17.631999Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load the Data","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')\nsample_sub = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:17.636403Z","iopub.execute_input":"2024-12-22T19:56:17.637142Z","iopub.status.idle":"2024-12-22T19:56:29.364091Z","shell.execute_reply.started":"2024-12-22T19:56:17.63708Z","shell.execute_reply":"2024-12-22T19:56:29.363049Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns = df_train.columns.str.lower().str.replace(' ', '_')\n\ndf_test.columns = df_test.columns.str.lower().str.replace(' ', '_')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:29.365555Z","iopub.execute_input":"2024-12-22T19:56:29.365874Z","iopub.status.idle":"2024-12-22T19:56:29.373222Z","shell.execute_reply.started":"2024-12-22T19:56:29.365843Z","shell.execute_reply":"2024-12-22T19:56:29.371878Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Missing Values","metadata":{}},{"cell_type":"code","source":"print(df_train.isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:29.374566Z","iopub.execute_input":"2024-12-22T19:56:29.374885Z","iopub.status.idle":"2024-12-22T19:56:30.034735Z","shell.execute_reply.started":"2024-12-22T19:56:29.374856Z","shell.execute_reply":"2024-12-22T19:56:30.033431Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Impute missing values based on median per gender, education_level and exercise_frequency","metadata":{}},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:30.036277Z","iopub.execute_input":"2024-12-22T19:56:30.036743Z","iopub.status.idle":"2024-12-22T19:56:30.046033Z","shell.execute_reply.started":"2024-12-22T19:56:30.036694Z","shell.execute_reply":"2024-12-22T19:56:30.044813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def impute_by_categories(df_train, df_test, category_columns=['gender', 'education_level', 'exercise_frequency']):\n    \"\"\"\n    Impute missing values for numeric columns based on the median value within each\n    combination of specified categorical variables.\n    \n    Parameters:\n    -----------\n    df_train : pandas.DataFrame\n        Training dataset\n    df_test : pandas.DataFrame\n        Test dataset\n    category_columns : list\n        List of categorical columns to group by for imputation\n    \n    Returns:\n    --------\n    df_train, df_test : tuple of pandas.DataFrame\n        Processed datasets with imputed values\n    \"\"\"\n        \n    # Get numeric columns to impute, excluding premium_amount\n    numeric_columns = df_train.select_dtypes(include=[np.number]).columns.drop('premium_amount')\n    \n    # Create a function to impute values for a given dataframe\n    def impute_groups(df):\n        # Create a copy to store imputed values\n        df_imputed = df.copy()\n        \n        # Group by the specified categorical columns\n        for group_values, group_df in df.groupby(category_columns):\n            # Create a mask for the current group\n            mask = pd.Series(True, index=df.index)\n            for col, val in zip(category_columns, group_values):\n                mask &= (df[col] == val)\n            \n            if mask.sum() > 0:  # Check if the group is not empty\n                # Initialize and fit imputer on the current group\n                imputer = SimpleImputer(strategy='median')\n                df_imputed.loc[mask, numeric_columns] = imputer.fit_transform(\n                    group_df[numeric_columns]\n                )\n        \n        return df_imputed\n    \n    # Apply imputation to both train and test sets\n    df_train = impute_groups(df_train)\n    df_test = impute_groups(df_test)\n    \n    return df_train, df_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:30.047831Z","iopub.execute_input":"2024-12-22T19:56:30.048944Z","iopub.status.idle":"2024-12-22T19:56:30.062866Z","shell.execute_reply.started":"2024-12-22T19:56:30.048886Z","shell.execute_reply":"2024-12-22T19:56:30.061681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train, df_test = impute_by_categories(df_train, df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:30.066368Z","iopub.execute_input":"2024-12-22T19:56:30.066799Z","iopub.status.idle":"2024-12-22T19:56:51.79017Z","shell.execute_reply.started":"2024-12-22T19:56:30.066764Z","shell.execute_reply":"2024-12-22T19:56:51.788745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_train.isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:51.791675Z","iopub.execute_input":"2024-12-22T19:56:51.79213Z","iopub.status.idle":"2024-12-22T19:56:52.442926Z","shell.execute_reply.started":"2024-12-22T19:56:51.792082Z","shell.execute_reply":"2024-12-22T19:56:52.441326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df_test.isna().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:52.444479Z","iopub.execute_input":"2024-12-22T19:56:52.445012Z","iopub.status.idle":"2024-12-22T19:56:52.877756Z","shell.execute_reply.started":"2024-12-22T19:56:52.444952Z","shell.execute_reply":"2024-12-22T19:56:52.876409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dummy for Categorical","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\n\ncolm = ['gender','marital_status','smoking_status','education_level','location','customer_feedback','exercise_frequency']\nfor i in colm:\n    df_train[i] = le.fit_transform(df_train[i])\n\ndf_train = pd.get_dummies(df_train,columns=['occupation','policy_type','property_type'])\n\n\nle = LabelEncoder()\ncolm = ['gender','marital_status','smoking_status','education_level','location','customer_feedback','exercise_frequency']\n\nfor i in colm:\n    le.fit(df_test[i])\n    df_test[i] = le.transform(df_test[i])\n\ndf_test = pd.get_dummies(df_test, columns=['occupation', 'policy_type', 'property_type'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Normalize Predictors","metadata":{}},{"cell_type":"code","source":"print(df_train.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:52.879496Z","iopub.execute_input":"2024-12-22T19:56:52.879971Z","iopub.status.idle":"2024-12-22T19:56:53.428648Z","shell.execute_reply.started":"2024-12-22T19:56:52.879921Z","shell.execute_reply":"2024-12-22T19:56:53.427376Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = df_train['premium_amount']\n\nscaler = StandardScaler()\nnumeric_columns = df_train.select_dtypes(include=[np.number]).columns.drop('premium_amount')\n\ndf_train[numeric_columns] = scaler.fit_transform(df_train[numeric_columns])\ndf_test[numeric_columns] = scaler.transform(df_test[numeric_columns])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:53.430153Z","iopub.execute_input":"2024-12-22T19:56:53.43053Z","iopub.status.idle":"2024-12-22T19:56:53.877671Z","shell.execute_reply.started":"2024-12-22T19:56:53.430493Z","shell.execute_reply":"2024-12-22T19:56:53.876243Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Correlations With Premium Amount","metadata":{}},{"cell_type":"code","source":"correlations = df_train[numeric_columns].join(y).corr()['premium_amount'].sort_values()\nprint(correlations)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:53.879874Z","iopub.execute_input":"2024-12-22T19:56:53.88033Z","iopub.status.idle":"2024-12-22T19:56:54.453943Z","shell.execute_reply.started":"2024-12-22T19:56:53.880279Z","shell.execute_reply":"2024-12-22T19:56:54.452705Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Summary of the training data","metadata":{}},{"cell_type":"markdown","source":"# Prepare Data","metadata":{}},{"cell_type":"code","source":"y = df_train['premium_amount']\n\n# Prepare X_full with column names\nX_full = df_train[numeric_columns]\nX_full = sm.add_constant(X_full)\n\ny = pd.Series(y, name='premium_amount')\n\n# Drop NaNs and infinite values consistently\ncombined = pd.concat([X_full, y], axis=1).replace([np.inf, -np.inf], np.nan).dropna()\nX_full = combined.drop(columns=['premium_amount'])\ny = combined['premium_amount']\n\n# Check shapes\nprint(\"Shape of X_full:\", X_full.shape)\nprint(\"Shape of y:\", y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:54.455618Z","iopub.execute_input":"2024-12-22T19:56:54.456083Z","iopub.status.idle":"2024-12-22T19:56:55.153186Z","shell.execute_reply.started":"2024-12-22T19:56:54.456035Z","shell.execute_reply":"2024-12-22T19:56:55.151961Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hexbin(X_full['credit_score'], y, gridsize=50, cmap=plt.cm.jet) # Adjust gridsize\nplt.xlabel(\"Credit Score\")\nplt.ylabel(\"Insurance Premium\")\nplt.title(\"Hexbin Plot\")\nplt.colorbar(label=\"Density\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:55.15463Z","iopub.execute_input":"2024-12-22T19:56:55.15498Z","iopub.status.idle":"2024-12-22T19:56:55.737934Z","shell.execute_reply.started":"2024-12-22T19:56:55.154946Z","shell.execute_reply":"2024-12-22T19:56:55.736678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Bucket Credit Score","metadata":{}},{"cell_type":"code","source":"# 1. Bin credit_score into multiple bins to find a suitable cut-off\nmin_val = np.min(X_full['credit_score'])\nmax_val = np.max(X_full['credit_score'])\n\n# Start with more bins to get a granular view\nn_bins = 10  # Increased for initial analysis\n\nbin_width = (max_val - min_val) / n_bins\nbins = [min_val + i * bin_width for i in range(n_bins + 1)]\n\n# Add -inf and inf for edge cases\nbins[0] = -np.inf\nbins[-1] = np.inf\n\n# Use numerical labels for easier analysis\nlabels = [f'Bin_{i+1}' for i in range(n_bins)]\n\n# Bin the credit_score\nX_full['credit_score_binned'] = pd.cut(X_full['credit_score'], bins=bins, labels=labels, right=False, include_lowest=True)\n\n# 2. Analyze Median Premium for Each Bin\ndf_analysis = X_full.copy()\ndf_analysis['premium_amount'] = y\n\n# Calculate median premium for each bin\nmedian_premiums = df_analysis.groupby('credit_score_binned')['premium_amount'].median()\n\n# Print median premiums for each bin\nprint(\"Median Premiums for each bin:\")\nprint(median_premiums)\n\n# 3. Identify a Suitable Cut-Off\n# Find a bin boundary where the median premium changes significantly\n# This is a manual step where you examine the median_premiums\n# You can also use the mean.\n# Example: Let's say the median premiums show a large jump between Bin_4 and Bin_5\n\n# Assuming the significant change is found to be between Bin_4 and Bin_5:\ncut_off_bin = 'Bin_6'  # Replace with the observed bin label\n\n# Get the credit score value corresponding to the identified cut-off\ncut_off_value = bins[labels.index(cut_off_bin)]\n\nprint(f\"\\nIdentified cut-off value: {cut_off_value}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:55.739313Z","iopub.execute_input":"2024-12-22T19:56:55.739792Z","iopub.status.idle":"2024-12-22T19:56:55.900411Z","shell.execute_reply.started":"2024-12-22T19:56:55.739753Z","shell.execute_reply":"2024-12-22T19:56:55.899171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Create Final Bins Based on the Cut-Off\nfinal_bins = [-np.inf, cut_off_value, np.inf]\nfinal_labels = ['Low', 'High']\n\n# Apply the final binning to both training and testing sets\nX_full['credit_score_binned'] = pd.cut(X_full['credit_score'], bins=final_bins, labels=final_labels, right=False, include_lowest=True)\ndf_test['credit_score_binned'] = pd.cut(df_test['credit_score'], bins=final_bins, labels=final_labels, right=False, include_lowest=True)\n\n# 5. One-Hot Encode the Binned Variable\nX_full = pd.get_dummies(X_full, columns=['credit_score_binned'], prefix='credit_score_bin', drop_first=True)\ndf_test = pd.get_dummies(df_test, columns=['credit_score_binned'], prefix='credit_score_bin', drop_first=True)\n\n# Explicitly convert to integer (0/1)\nfor col in X_full.columns:\n    if X_full[col].dtype == bool:\n        X_full[col] = X_full[col].astype(int)\n\nfor col in df_test.columns:\n    if df_test[col].dtype == bool:\n        df_test[col] = df_test[col].astype(int)\n\n# Drop the original 'credit_score' column\nX_full = X_full.drop('credit_score', axis=1)\ndf_test = df_test.drop('credit_score', axis=1)\n\nprint(\"\\nFinal Columns in X_full:\", X_full.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:55.901752Z","iopub.execute_input":"2024-12-22T19:56:55.902091Z","iopub.status.idle":"2024-12-22T19:56:56.616973Z","shell.execute_reply.started":"2024-12-22T19:56:55.902059Z","shell.execute_reply":"2024-12-22T19:56:56.61574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, bin_name in enumerate(labels):\n    col_name = f'credit_score_bin_{bin_name}'\n    if col_name in df_analysis.columns: #check if column exists, important if drop_first=True\n        df_analysis.loc[df_analysis[col_name] == 1, 'credit_score_binned'] = bin_name\n\ngrouped = df_analysis.groupby('credit_score_binned', observed=True)['premium_amount']\n\nprint(\"Descriptive Statistics per Credit Score Bin:\\n\")\nprint(grouped.describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:56.618735Z","iopub.execute_input":"2024-12-22T19:56:56.619114Z","iopub.status.idle":"2024-12-22T19:56:56.774477Z","shell.execute_reply.started":"2024-12-22T19:56:56.619078Z","shell.execute_reply":"2024-12-22T19:56:56.773023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_plot = X_full.copy()\ndf_plot['premium_amount'] = y\n\n# Add a 'credit_score_bin' column for plotting\ndf_plot['credit_score_bin'] = 'Low'  # Default to Low\ndf_plot.loc[df_plot['credit_score_bin_High'] == 1, 'credit_score_bin'] = 'High'\n\n# Create the violin plot\nplt.figure(figsize=(6, 6))  # Adjust figure size as needed\nsns.violinplot(x='credit_score_bin', y='premium_amount', data=df_plot)\nplt.xlabel(\"Credit Score Bin\")\nplt.ylabel(\"Premium Amount\")\nplt.title(\"Violin Plot of Premium Amount vs. Credit Score Bins\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:56:56.77608Z","iopub.execute_input":"2024-12-22T19:56:56.776509Z","iopub.status.idle":"2024-12-22T19:57:00.263022Z","shell.execute_reply.started":"2024-12-22T19:56:56.776443Z","shell.execute_reply":"2024-12-22T19:57:00.261787Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Forward Selection Linear Regression Model","metadata":{}},{"cell_type":"code","source":"# Ensure X_full and y have the correct shapes\n# X_full = X_full.values\ny = y.values\n\n# Ensure y is 1D\ny = y.ravel() if y.ndim > 1 else y\n\n# Perform forward selection using SequentialFeatureSelector\nsfs = SequentialFeatureSelector(LinearRegression(), direction='forward', scoring='r2', cv=5, n_features_to_select='auto', tol=None)\nsfs.fit(X_full, y)\n\n# Get the indices of the selected features\nselected_features_indices = sfs.get_support(indices=True)\n\n# Get the names of the selected features directly from X_full.columns using the indices\nselected_feature_names = X_full.columns[selected_features_indices].tolist()\n\n# Create a new DataFrame with only the selected features\nX_selected = X_full[selected_feature_names]\n\nprint(\"Selected feature names:\", selected_feature_names)\n\n# Ensure X_selected is 2D\nif X_selected.ndim == 1:\n    X_selected = X_selected.reshape(-1, 1)\n\n# Check the dimensions of X_selected and y\nprint(f\"X_selected shape: {X_selected.shape}\")\nprint(f\"y shape: {y.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:57:00.264586Z","iopub.execute_input":"2024-12-22T19:57:00.26495Z","iopub.status.idle":"2024-12-22T19:57:55.496229Z","shell.execute_reply.started":"2024-12-22T19:57:00.264915Z","shell.execute_reply":"2024-12-22T19:57:55.491783Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Summary","metadata":{}},{"cell_type":"code","source":"final_model = sm.OLS(y, X_selected).fit()\nprint(final_model.summary())\n\n# Make predictions\npredictions = final_model.predict(X_selected)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:57:55.498329Z","iopub.execute_input":"2024-12-22T19:57:55.50309Z","iopub.status.idle":"2024-12-22T19:57:56.562185Z","shell.execute_reply.started":"2024-12-22T19:57:55.503005Z","shell.execute_reply":"2024-12-22T19:57:56.559964Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model Diagnostics and Cross-Validation","metadata":{}},{"cell_type":"code","source":"# Check for multicollinearity using Variance Inflation Factor (VIF)\nvif_data = pd.DataFrame()\nvif_data[\"feature\"] = selected_feature_names\nvif_data[\"VIF\"] = [variance_inflation_factor(X_full[selected_feature_names].values, i) for i in range(len(selected_feature_names))]\nprint(vif_data)\n\n# Identify predictors with high VIF\nhigh_vif_features = vif_data[vif_data[\"VIF\"] > 5]\nif not high_vif_features.empty:\n    print(\"Warning: High multicollinearity detected for the following predictors:\")\n    print(high_vif_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:57:56.563781Z","iopub.execute_input":"2024-12-22T19:57:56.564217Z","iopub.status.idle":"2024-12-22T19:58:00.326082Z","shell.execute_reply.started":"2024-12-22T19:57:56.56415Z","shell.execute_reply":"2024-12-22T19:58:00.324001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate RMSE (Root Mean Squared Error)\nrmse = np.sqrt(mean_squared_error(y, predictions))\nprint(\"Root Mean Squared Error (RMSE):\", rmse)\n\n# Calculate R-squared\nr_squared = r2_score(y, predictions)\nprint(\"R-squared:\", r_squared)\n\nX_selected_with_constant = sm.add_constant(X_selected)\nfinal_model_with_constant = sm.OLS(y, X_selected_with_constant).fit()\n\n# Diagnostic plots\nfig, axs = plt.subplots(2, 2, figsize=(15, 10))\nsm.graphics.plot_regress_exog(final_model_with_constant, 'previous_claims', fig=fig)\nplt.tight_layout()\nplt.show()\n\n# Residual Analysis\nresiduals = final_model_with_constant.resid\nprint(\"Residuals Summary:\")\nprint(residuals.describe())\n\n# Check for heteroscedasticity using the Breusch-Pagan test\nbp_test = sm.stats.diagnostic.het_breuschpagan(residuals, final_model_with_constant.model.exog)\nprint(\"Breusch-Pagan Test for Heteroscedasticity:\", bp_test)\n\n# Check for normality of residuals \nad_test = sm.stats.diagnostic.normal_ad(residuals)\nprint(\"Anderson-Darling Test for Normality:\", ad_test)\n\n# Cross-validation to validate the model\ncross_val_scores = cross_val_score(LinearRegression(), X_selected_with_constant, y, cv=10, scoring='r2')\nprint(\"Cross-Validation Results:\", cross_val_scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:58:00.337197Z","iopub.execute_input":"2024-12-22T19:58:00.338703Z","iopub.status.idle":"2024-12-22T19:59:33.394899Z","shell.execute_reply.started":"2024-12-22T19:58:00.338619Z","shell.execute_reply":"2024-12-22T19:59:33.393524Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submit Predicitons","metadata":{}},{"cell_type":"code","source":"X_test_selected = sm.add_constant(df_test[selected_feature_names])\npredictions_test = final_model_with_constant.predict(X_test_selected)\n\nsample_sub['Premium Amount'] = predictions_test\n\nsample_sub.to_csv(\"submission.csv\", index=False)\nprint(\"Submission file 'submission.csv' created successfully.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-22T19:59:33.396724Z","iopub.execute_input":"2024-12-22T19:59:33.397358Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Check for missing values in the predictions}\")","metadata":{}},{"cell_type":"code","source":"missing_values = predictions_test.isnull().sum()\n\nprint(f\"Number of missing values in the predictions: {missing_values}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}