{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from collections import Counter\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport optuna\nimport pandas as pd\nimport numpy as np\nimport pandas as pd\nimport numpy as np\nfrom time import time\nfrom sklearn.metrics import roc_auc_score\n\nfrom sklearn.model_selection import train_test_split,GridSearchCV,StratifiedKFold,RandomizedSearchCV,RepeatedStratifiedKFold\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import cross_val_score\nfrom xgboost import XGBClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\n\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_auc_score\nfrom sklearn.model_selection import GridSearchCV, KFold\nfrom sklearn.ensemble import ExtraTreesClassifier\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom sklearn.metrics import cohen_kappa_score,make_scorer\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.ensemble import StackingClassifier\nfrom lightgbm import LGBMClassifier\nfrom catboost import CatBoostClassifier\nfrom pandas.api.types import is_numeric_dtype\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore', message='use_inf_as_na option is deprecated')\n# warnings.filterwarnings('ignore', category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:27.21263Z","iopub.execute_input":"2024-12-07T12:53:27.214391Z","iopub.status.idle":"2024-12-07T12:53:32.959245Z","shell.execute_reply.started":"2024-12-07T12:53:27.214332Z","shell.execute_reply":"2024-12-07T12:53:32.958344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsample_sub = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\n\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:32.96091Z","iopub.execute_input":"2024-12-07T12:53:32.961437Z","iopub.status.idle":"2024-12-07T12:53:41.602569Z","shell.execute_reply.started":"2024-12-07T12:53:32.961409Z","shell.execute_reply":"2024-12-07T12:53:41.601657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:41.603773Z","iopub.execute_input":"2024-12-07T12:53:41.604147Z","iopub.status.idle":"2024-12-07T12:53:41.622647Z","shell.execute_reply.started":"2024-12-07T12:53:41.604109Z","shell.execute_reply":"2024-12-07T12:53:41.621796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:41.623693Z","iopub.execute_input":"2024-12-07T12:53:41.623919Z","iopub.status.idle":"2024-12-07T12:53:42.177207Z","shell.execute_reply.started":"2024-12-07T12:53:41.623897Z","shell.execute_reply":"2024-12-07T12:53:42.176152Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:42.180992Z","iopub.execute_input":"2024-12-07T12:53:42.181921Z","iopub.status.idle":"2024-12-07T12:53:42.546337Z","shell.execute_reply.started":"2024-12-07T12:53:42.181866Z","shell.execute_reply":"2024-12-07T12:53:42.545441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = train.drop(['id'], axis=1)\ntest = test.drop(['id'], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:42.547469Z","iopub.execute_input":"2024-12-07T12:53:42.547754Z","iopub.status.idle":"2024-12-07T12:53:42.794448Z","shell.execute_reply.started":"2024-12-07T12:53:42.547722Z","shell.execute_reply":"2024-12-07T12:53:42.793749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe(include='all')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:42.795327Z","iopub.execute_input":"2024-12-07T12:53:42.795614Z","iopub.status.idle":"2024-12-07T12:53:44.906773Z","shell.execute_reply.started":"2024-12-07T12:53:42.795588Z","shell.execute_reply":"2024-12-07T12:53:44.905844Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.describe(include='all')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:44.907662Z","iopub.execute_input":"2024-12-07T12:53:44.907902Z","iopub.status.idle":"2024-12-07T12:53:46.248806Z","shell.execute_reply.started":"2024-12-07T12:53:44.907879Z","shell.execute_reply":"2024-12-07T12:53:46.248012Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Visualization","metadata":{}},{"cell_type":"code","source":"age_counts = train['Age'].value_counts()\n\nplt.figure(figsize=(12, 8))\nage_counts.sort_index().plot(kind='bar', color='skyblue', width=0.8)  # Sorting by index (Age) for logical order\nplt.title('Frequency Distribution of Age')\nplt.xlabel('Age')\nplt.ylabel('Frequency')\nplt.xticks(rotation=45)\nplt.grid(axis='y', alpha=0.3)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:46.250161Z","iopub.execute_input":"2024-12-07T12:53:46.250603Z","iopub.status.idle":"2024-12-07T12:53:46.736366Z","shell.execute_reply.started":"2024-12-07T12:53:46.250561Z","shell.execute_reply":"2024-12-07T12:53:46.7356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(train['Age'], bins=15, kde=True)\nplt.title('Age Distribution')\nplt.xlabel('Age')\nplt.ylabel('Frequency')\nplt.grid(alpha=0.3)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:46.73795Z","iopub.execute_input":"2024-12-07T12:53:46.738335Z","iopub.status.idle":"2024-12-07T12:53:51.309509Z","shell.execute_reply.started":"2024-12-07T12:53:46.738278Z","shell.execute_reply":"2024-12-07T12:53:51.308669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\nsns.histplot(train['Premium Amount'], kde=True, bins=50)\nplt.title('Premium Amount Distribution')\nplt.xlabel(\"Premium Amount\")\nplt.ylabel(\"Frequency\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:51.310713Z","iopub.execute_input":"2024-12-07T12:53:51.311071Z","iopub.status.idle":"2024-12-07T12:53:56.0849Z","shell.execute_reply.started":"2024-12-07T12:53:51.311033Z","shell.execute_reply":"2024-12-07T12:53:56.084209Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.lineplot(x='Policy Type', y='Premium Amount', data=train)\nplt.title('Policy Type vs Premium Amount')\nplt.xlabel('Policy Type')\nplt.ylabel('Premium Amount')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:53:56.086178Z","iopub.execute_input":"2024-12-07T12:53:56.086559Z","iopub.status.idle":"2024-12-07T12:54:05.285881Z","shell.execute_reply.started":"2024-12-07T12:53:56.086522Z","shell.execute_reply":"2024-12-07T12:54:05.285129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming your DataFrame is 'df'\navg_premium_by_gender = train.groupby('Gender')['Premium Amount'].mean().reset_index()\nsns.barplot(x='Gender', y='Premium Amount', data=avg_premium_by_gender)\nplt.title('Average Premium Amount by Gender')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:05.28682Z","iopub.execute_input":"2024-12-07T12:54:05.287171Z","iopub.status.idle":"2024-12-07T12:54:05.47303Z","shell.execute_reply.started":"2024-12-07T12:54:05.287132Z","shell.execute_reply":"2024-12-07T12:54:05.472094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(x='Marital Status',y='Premium Amount',data=train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:05.477957Z","iopub.execute_input":"2024-12-07T12:54:05.478237Z","iopub.status.idle":"2024-12-07T12:54:06.293489Z","shell.execute_reply.started":"2024-12-07T12:54:05.478209Z","shell.execute_reply":"2024-12-07T12:54:06.292612Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.boxplot(x='Number of Dependents',y='Premium Amount',data=train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:06.294571Z","iopub.execute_input":"2024-12-07T12:54:06.294839Z","iopub.status.idle":"2024-12-07T12:54:06.781513Z","shell.execute_reply.started":"2024-12-07T12:54:06.294812Z","shell.execute_reply":"2024-12-07T12:54:06.780631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols= train.select_dtypes(include='number').columns\nnum_cols = [val for val in num_cols]\ncat_cols = train.select_dtypes(exclude='number').columns\ndatetime_cols = ['Policy Start Date']\ncat_cols = [col for col in cat_cols if col not in datetime_cols]\n\nprint(\"Numerical columns:\",)\nprint(num_cols)\nprint(\"\")\nprint(\"Categorical columns excluding datetime columns:\")\nprint(cat_cols)\nprint(\"\")\nprint(\"Datetime column:\")\ndatetime_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:06.782726Z","iopub.execute_input":"2024-12-07T12:54:06.783081Z","iopub.status.idle":"2024-12-07T12:54:06.96656Z","shell.execute_reply.started":"2024-12-07T12:54:06.783044Z","shell.execute_reply":"2024-12-07T12:54:06.965662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train[num_cols].hist(bins=30, figsize=(12, 15))\nplt.suptitle('Histograms of Numerical Columns')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:06.967575Z","iopub.execute_input":"2024-12-07T12:54:06.96785Z","iopub.status.idle":"2024-12-07T12:54:09.01889Z","shell.execute_reply.started":"2024-12-07T12:54:06.967825Z","shell.execute_reply":"2024-12-07T12:54:09.017976Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for col in cat_cols:\n    plt.figure(figsize=(4, 3))\n    sns.countplot(x=col, data=train)\n    plt.title(f'Count Plot of {col}')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:09.019804Z","iopub.execute_input":"2024-12-07T12:54:09.020057Z","iopub.status.idle":"2024-12-07T12:54:15.657895Z","shell.execute_reply.started":"2024-12-07T12:54:09.020032Z","shell.execute_reply":"2024-12-07T12:54:15.657026Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"corr_matrix = train[num_cols].corr()\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, cmap = 'Greens')\nplt.title('Correlation Heatmap of Numerical Columns')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:15.658903Z","iopub.execute_input":"2024-12-07T12:54:15.659168Z","iopub.status.idle":"2024-12-07T12:54:16.287184Z","shell.execute_reply.started":"2024-12-07T12:54:15.659143Z","shell.execute_reply":"2024-12-07T12:54:16.286358Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Feature engineering","metadata":{}},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:16.288406Z","iopub.execute_input":"2024-12-07T12:54:16.289064Z","iopub.status.idle":"2024-12-07T12:54:16.825957Z","shell.execute_reply.started":"2024-12-07T12:54:16.289023Z","shell.execute_reply":"2024-12-07T12:54:16.825084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:16.826856Z","iopub.execute_input":"2024-12-07T12:54:16.827093Z","iopub.status.idle":"2024-12-07T12:54:17.183546Z","shell.execute_reply.started":"2024-12-07T12:54:16.82707Z","shell.execute_reply":"2024-12-07T12:54:17.182669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols= train.select_dtypes(include='number').columns\nnum_cols = [val for val in num_cols]\ncat_cols = train.select_dtypes(exclude='number').columns\ndatetime_cols = ['Policy Start Date']\ncat_cols = [col for col in cat_cols if col not in datetime_cols]\n\nprint(\"Numerical columns:\",)\nprint(num_cols)\nprint(\"\")\nprint(\"Categorical columns excluding datetime columns:\")\nprint(cat_cols)\nprint(\"\")\nprint(\"Datetime column:\")\ndatetime_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:17.184701Z","iopub.execute_input":"2024-12-07T12:54:17.185058Z","iopub.status.idle":"2024-12-07T12:54:17.366477Z","shell.execute_reply.started":"2024-12-07T12:54:17.185019Z","shell.execute_reply":"2024-12-07T12:54:17.365657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"numerical_columns = train.select_dtypes(include=['int32', 'int64', 'float64']).columns.tolist()\nnumerical_columns.remove('Premium Amount') \n\nfor col in numerical_columns:\n    median_value = train[col].median()\n    train[col] = train[col].fillna(median_value)\n    test[col] = test[col].fillna(median_value)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:17.367569Z","iopub.execute_input":"2024-12-07T12:54:17.367847Z","iopub.status.idle":"2024-12-07T12:54:17.659246Z","shell.execute_reply.started":"2024-12-07T12:54:17.36782Z","shell.execute_reply":"2024-12-07T12:54:17.658519Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"categorical_columns = train.select_dtypes(include=['object']).columns.tolist()\n\nfor col in categorical_columns:\n    train[col] = train[col].astype(str).fillna(\"Unknown\")\n    test[col] = test[col].astype(str).fillna(\"Unknown\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:17.66014Z","iopub.execute_input":"2024-12-07T12:54:17.660387Z","iopub.status.idle":"2024-12-07T12:54:19.39114Z","shell.execute_reply.started":"2024-12-07T12:54:17.660363Z","shell.execute_reply":"2024-12-07T12:54:19.390337Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:19.392194Z","iopub.execute_input":"2024-12-07T12:54:19.392473Z","iopub.status.idle":"2024-12-07T12:54:19.927903Z","shell.execute_reply.started":"2024-12-07T12:54:19.392447Z","shell.execute_reply":"2024-12-07T12:54:19.927079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:19.928751Z","iopub.execute_input":"2024-12-07T12:54:19.928997Z","iopub.status.idle":"2024-12-07T12:54:20.290619Z","shell.execute_reply.started":"2024-12-07T12:54:19.928972Z","shell.execute_reply":"2024-12-07T12:54:20.28976Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Models","metadata":{}},{"cell_type":"markdown","source":"XGB Regression","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\n# Create a copy of your original dataframe\ntrain_copy1 = train.copy()\ntest_copy1 = test.copy()\n\nle = LabelEncoder()\n\n# Loop through each column in the dataframe\nfor column in train_copy1.columns:\n    if train_copy1[column].dtype == 'object':  # For categorical columns\n        train_copy1[column] = le.fit_transform(train_copy1[column])\n\n# Loop through each column in the dataframe\nfor column in test_copy1.columns:\n    if test_copy1[column].dtype == 'object':  # For categorical columns\n        test_copy1[column] = le.fit_transform(test_copy1[column])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:20.291758Z","iopub.execute_input":"2024-12-07T12:54:20.292017Z","iopub.status.idle":"2024-12-07T12:54:25.837965Z","shell.execute_reply.started":"2024-12-07T12:54:20.291992Z","shell.execute_reply":"2024-12-07T12:54:25.837238Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_copy1.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:25.838873Z","iopub.execute_input":"2024-12-07T12:54:25.839133Z","iopub.status.idle":"2024-12-07T12:54:25.855979Z","shell.execute_reply.started":"2024-12-07T12:54:25.839107Z","shell.execute_reply":"2024-12-07T12:54:25.855179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from xgboost.sklearn import XGBRegressor\nimport optuna\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, r2_score\nimport warnings\n\n# Suppress warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning)\n\n# Prepare your data\nX = train_copy1.drop(columns=['Premium Amount'])\ny = train_copy1['Premium Amount']\n\n# Split the data\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Define RMSLE function\ndef rmsle(y_true, y_pred):\n    return np.sqrt(np.mean((np.log1p(y_pred) - np.log1p(y_true))**2))\n\n# Objective function for Optuna optimization\ndef objective(trial):\n    # Defining the hyperparameter search space\n    params = {\n        'n_estimators': trial.suggest_categorical('n_estimators', [100, 200, 300]),\n        'max_depth': trial.suggest_int('max_depth', 3, 9),\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.01, 0.1),\n        'subsample': trial.suggest_uniform('subsample', 0.8, 1.0),\n        'colsample_bytree': trial.suggest_uniform('colsample_bytree', 0.8, 1.0),\n        'gamma': trial.suggest_uniform('gamma', 0, 0.2),\n        'objective': 'reg:squarederror',  # Regression objective\n        'missing': np.nan,  # Handle missing values\n        'tree_method': 'gpu_hist',  # Use GPU for training\n        'gpu_id': 0  # Specify GPU ID\n    }\n\n    # Initialize XGBoost model with suggested parameters\n    model = XGBRegressor(**params)\n    \n    # Fit the model\n    model.fit(X_train, y_train)\n    \n    # Predict the values\n    y_pred = model.predict(X_val)\n    \n    # Evaluate the model using Mean Squared Error (MSE)\n    mse = mean_squared_error(y_val, y_pred)\n    return mse  # Minimize MSE\n\n# Create a study object for Optuna and optimize the objective function\nstudy = optuna.create_study(direction='minimize')  # Minimize MSE\nstudy.optimize(objective, n_trials=50)  # Perform 50 trials\n\n# Get the best hyperparameters and best model\nbest_params = study.best_params\nprint(f\"Best Hyperparameters: {best_params}\")\n\n# Using the best hyperparameters to train the final model\nbest_model = XGBRegressor(**best_params, tree_method=\"gpu_hist\", gpu_id=0)\nbest_model.fit(X_train, y_train)\n\n# Predict with the tuned model\ny_pred = best_model.predict(X_val)\n\n# Evaluate the model\nmse = mean_squared_error(y_val, y_pred)\nrmse = np.sqrt(mse)\nr2 = r2_score(y_val, y_pred)\n\n# Output the evaluation metrics\nprint(f\"Mean Squared Error: {mse}\")\nprint(f\"Root Mean Squared Error: {rmse}\")\nprint(f\"R² Score: {r2}\")\nprint(f\"Root Mean Squared Log Error: {rmsle(y_val, y_pred)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:54:25.856861Z","iopub.execute_input":"2024-12-07T12:54:25.857112Z","iopub.status.idle":"2024-12-07T12:57:24.465418Z","shell.execute_reply.started":"2024-12-07T12:54:25.857086Z","shell.execute_reply":"2024-12-07T12:57:24.464493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = best_model.predict(test_copy1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:57:24.466516Z","iopub.execute_input":"2024-12-07T12:57:24.466871Z","iopub.status.idle":"2024-12-07T12:57:24.81789Z","shell.execute_reply.started":"2024-12-07T12:57:24.46684Z","shell.execute_reply":"2024-12-07T12:57:24.817136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\nsub['Premium Amount'] = test_predictions\nprint(sub.info())\nsub.to_csv('submission.csv', index=False)\nsub.info()   ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T12:57:24.818919Z","iopub.execute_input":"2024-12-07T12:57:24.819258Z","iopub.status.idle":"2024-12-07T12:57:25.939864Z","shell.execute_reply.started":"2024-12-07T12:57:24.819222Z","shell.execute_reply":"2024-12-07T12:57:25.938926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"XGBoost","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}