{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# IMPORT LIBRARIES\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import cross_val_score, KFold\nfrom sklearn.metrics import make_scorer, mean_squared_log_error  # Import mean_squared_log_error\nimport xgboost as xgb\nimport numpy as np\nfrom sklearn.ensemble import RandomForestRegressor\nfrom xgboost import plot_importance","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T10:56:10.272219Z","iopub.execute_input":"2024-12-01T10:56:10.272553Z","iopub.status.idle":"2024-12-01T10:56:13.050632Z","shell.execute_reply.started":"2024-12-01T10:56:10.272523Z","shell.execute_reply":"2024-12-01T10:56:13.049902Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# UPLOAD DATA","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/playground-series-s4e12/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/playground-series-s4e12/test.csv\")\nsample_submission = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:33:03.023477Z","iopub.execute_input":"2024-12-01T11:33:03.024169Z","iopub.status.idle":"2024-12-01T11:33:08.999255Z","shell.execute_reply.started":"2024-12-01T11:33:03.024139Z","shell.execute_reply":"2024-12-01T11:33:08.998518Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DATA ANALYSIS","metadata":{}},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:33:09.000665Z","iopub.execute_input":"2024-12-01T11:33:09.000934Z","iopub.status.idle":"2024-12-01T11:33:09.642019Z","shell.execute_reply.started":"2024-12-01T11:33:09.00091Z","shell.execute_reply":"2024-12-01T11:33:09.641123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:33:09.643199Z","iopub.execute_input":"2024-12-01T11:33:09.64348Z","iopub.status.idle":"2024-12-01T11:33:10.183543Z","shell.execute_reply.started":"2024-12-01T11:33:09.643454Z","shell.execute_reply":"2024-12-01T11:33:10.182565Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:33:10.185058Z","iopub.execute_input":"2024-12-01T11:33:10.185374Z","iopub.status.idle":"2024-12-01T11:33:10.718284Z","shell.execute_reply.started":"2024-12-01T11:33:10.185347Z","shell.execute_reply":"2024-12-01T11:33:10.717397Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DATA PREPROCESSING","metadata":{}},{"cell_type":"code","source":"train = df_train.copy()\ntest = df_test.copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:33:10.719718Z","iopub.execute_input":"2024-12-01T11:33:10.720064Z","iopub.status.idle":"2024-12-01T11:33:10.948662Z","shell.execute_reply.started":"2024-12-01T11:33:10.720025Z","shell.execute_reply":"2024-12-01T11:33:10.947953Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**FILLING NULL VALUES**","metadata":{}},{"cell_type":"code","source":"# Drop 'Occupation' column from both train and test datasets\ntrain.drop('Occupation', axis=1, inplace=True)\ntest.drop('Occupation', axis=1, inplace=True)\n\n# 2. Filling Missing Values\n# -------------------------\n\n# Define a helper function to fill missing values in both train and test datasets\ndef fill_missing_values(train_df, test_df):\n    # 2.1 Fill 'Age' with median\n    age_median = train_df['Age'].median()\n    train_df['Age'].fillna(age_median, inplace=True)\n    test_df['Age'].fillna(age_median, inplace=True)\n\n    # 2.2 Fill 'Annual Income' based on skewness\n    income_skew = train_df['Annual Income'].skew()\n    if income_skew > 1 or income_skew < -1:\n        income_fill = train_df['Annual Income'].median()\n    else:\n        income_fill = train_df['Annual Income'].mean()\n    train_df['Annual Income'].fillna(income_fill, inplace=True)\n    test_df['Annual Income'].fillna(income_fill, inplace=True)\n\n    # 2.3 Fill 'Marital Status' with mode\n    marital_mode = train_df['Marital Status'].mode()[0]\n    train_df['Marital Status'].fillna(marital_mode, inplace=True)\n    test_df['Marital Status'].fillna(marital_mode, inplace=True)\n\n    # 2.4 Fill 'Number of Dependents' with median\n    dependents_median = train_df['Number of Dependents'].median()\n    train_df['Number of Dependents'].fillna(dependents_median, inplace=True)\n    test_df['Number of Dependents'].fillna(dependents_median, inplace=True)\n\n    # 2.5 Fill 'Health Score' with median\n    health_median = train_df['Health Score'].median()\n    train_df['Health Score'].fillna(health_median, inplace=True)\n    test_df['Health Score'].fillna(health_median, inplace=True)\n\n    # 2.6 Fill 'Previous Claims' with 0\n    train_df['Previous Claims'].fillna(0, inplace=True)\n    test_df['Previous Claims'].fillna(0, inplace=True)\n\n    # 2.7 Fill 'Vehicle Age' with median\n    vehicle_median = train_df['Vehicle Age'].median()\n    train_df['Vehicle Age'].fillna(vehicle_median, inplace=True)\n    test_df['Vehicle Age'].fillna(vehicle_median, inplace=True)\n\n    # 2.8 Fill 'Insurance Duration' with mean or median based on skewness\n    duration_skew = train_df['Insurance Duration'].skew()\n    if duration_skew > 1 or duration_skew < -1:\n        duration_fill = train_df['Insurance Duration'].median()\n    else:\n        duration_fill = train_df['Insurance Duration'].mean()\n    train_df['Insurance Duration'].fillna(duration_fill, inplace=True)\n    test_df['Insurance Duration'].fillna(duration_fill, inplace=True)\n\n    # 2.9 Fill 'Policy Start Date' with median date\n    # Convert to datetime\n    train_df['Policy Start Date'] = pd.to_datetime(train_df['Policy Start Date'], errors='coerce')\n    test_df['Policy Start Date'] = pd.to_datetime(test_df['Policy Start Date'], errors='coerce')\n\n    # Convert datetime to ordinal for numerical imputation\n    policy_start_median = train_df['Policy Start Date'].median()\n    # If median is NaT, choose a default date or another strategy\n    if pd.isnull(policy_start_median):\n        policy_start_median = pd.Timestamp('2020-01-01')  # Example default date\n    train_df['Policy Start Date'].fillna(policy_start_median, inplace=True)\n    test_df['Policy Start Date'].fillna(policy_start_median, inplace=True)\n\n    # Optionally, convert back to datetime if needed\n    # train_df['Policy Start Date'] = pd.to_datetime(train_df['Policy Start Date'])\n    # test_df['Policy Start Date'] = pd.to_datetime(test_df['Policy Start Date'])\n\n    # 2.10 Fill 'Customer Feedback' with median or mode based on data type\n    if pd.api.types.is_numeric_dtype(train_df['Customer Feedback']):\n        feedback_median = train_df['Customer Feedback'].median()\n        train_df['Customer Feedback'].fillna(feedback_median, inplace=True)\n        test_df['Customer Feedback'].fillna(feedback_median, inplace=True)\n    else:\n        feedback_mode = train_df['Customer Feedback'].mode()[0]\n        train_df['Customer Feedback'].fillna(feedback_mode, inplace=True)\n        test_df['Customer Feedback'].fillna(feedback_mode, inplace=True)\n\n    return train_df, test_df\n\n# Apply the function to fill missing values\ntrain, test = fill_missing_values(train, test)\n\n# 3. Handling 'Credit Score'\n# ---------------------------\n\n# As per instructions, we will keep 'Credit Score' aside for now.\n# Optionally, you can create separate variables or just leave them as is with missing values.\n\n# Example:\n# credit_train = train['Credit Score']\n# credit_test = test['Credit Score']\n# You can later use these to train a model to predict missing values.\n\n# 4. Final Checks\n# ---------------\n\n# Verify that there are no missing values left (except 'Credit Score')\nprint(\"Missing values in training set after imputation:\")\nprint(train.isnull().sum())\n\nprint(\"\\nMissing values in testing set after imputation:\")\nprint(test.isnull().sum())\n\n# Optionally, drop 'Credit Score' for the time being\n# train.drop('Credit Score', axis=1, inplace=True)\n# test.drop('Credit Score', axis=1, inplace=True)\n\n# Now, your datasets 'train' and 'test' have missing values handled as per your specifications.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:33:10.949678Z","iopub.execute_input":"2024-12-01T11:33:10.949928Z","iopub.status.idle":"2024-12-01T11:33:13.073796Z","shell.execute_reply.started":"2024-12-01T11:33:10.949903Z","shell.execute_reply":"2024-12-01T11:33:13.072966Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**CONVERTING CATEGORICAL TO NUMERICAL**","metadata":{}},{"cell_type":"code","source":"# 1. Converting Categorical Variables to Numerical\n\nfrom sklearn.preprocessing import LabelEncoder\n\n# List of categorical columns\ncategorical_cols = [\n    'Gender', 'Marital Status', 'Education Level', 'Location',\n    'Policy Type', 'Customer Feedback', 'Smoking Status',\n    'Exercise Frequency', 'Property Type'\n]\n\n# Initialize a dictionary to keep track of label encoders\nlabel_encoders = {}\n\n# Loop through each categorical column and label encode\nfor col in categorical_cols:\n    le = LabelEncoder()\n    # Combine data to fit on both train and test to handle unseen categories\n    combined_data = pd.concat([train[col], test[col]], axis=0)\n    le.fit(combined_data)\n    train[col] = le.transform(train[col])\n    test[col] = le.transform(test[col])\n    label_encoders[col] = le  # Save the encoder for future use if needed\n\n# Handle 'Policy Start Date' separately\n\n# Ensure 'Policy Start Date' is in datetime format\ntrain['Policy Start Date'] = pd.to_datetime(train['Policy Start Date'], errors='coerce')\ntest['Policy Start Date'] = pd.to_datetime(test['Policy Start Date'], errors='coerce')\n\n# Extract date features\nfor df in [train, test]:\n    df['Policy_Start_Year'] = df['Policy Start Date'].dt.year\n    df['Policy_Start_Month'] = df['Policy Start Date'].dt.month\n    df['Policy_Start_Day'] = df['Policy Start Date'].dt.day\n\n# Drop the original 'Policy Start Date' column\ntrain.drop('Policy Start Date', axis=1, inplace=True)\ntest.drop('Policy Start Date', axis=1, inplace=True)\n\n# Optional: If 'id' is not useful, you can drop it\n# train.drop('id', axis=1, inplace=True)\n# test.drop('id', axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:42:03.957489Z","iopub.execute_input":"2024-12-01T11:42:03.957919Z","iopub.status.idle":"2024-12-01T11:42:07.336486Z","shell.execute_reply.started":"2024-12-01T11:42:03.957886Z","shell.execute_reply":"2024-12-01T11:42:07.335483Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## HANDLING THE CREDIT SCORE\nWe will train a seperate model to fill the 'Credit Score' values - since we observed that it has a high F-score.","metadata":{}},{"cell_type":"code","source":"# 2. Training an XGBoost Model to Fill 'Credit Score' in Both Train and Test\n\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\n\n# Exclude 'Credit Score' and 'Premium Amount' from features\nfeatures = train.columns.drop(['Credit Score', 'Premium Amount', 'id'])  # Exclude 'id' if not useful\n\n# Separate data where 'Credit Score' is not null\ntrain_credit = train[train['Credit Score'].notnull()].copy()\n# Rows where 'Credit Score' is null\ntrain_credit_missing = train[train['Credit Score'].isnull()].copy()\n\n# Prepare training data\nX_train_credit = train_credit[features]\ny_train_credit = train_credit['Credit Score']\n\n# Prepare data to predict missing 'Credit Score' in train set\nX_train_missing_credit = train_credit_missing[features]\n\n# Similarly for test set\ntest_credit_missing = test[test['Credit Score'].isnull()].copy()\nX_test_missing_credit = test_credit_missing[features]\n\n# Combine data to predict (both train and test missing 'Credit Score')\nX_missing_credit = pd.concat([X_train_missing_credit, X_test_missing_credit])\n\n# Train-validation split\nX_train_c, X_valid_c, y_train_c, y_valid_c = train_test_split(\n    X_train_credit, y_train_credit, test_size=0.2, random_state=42\n)\n\n# Create DMatrix for XGBoost\ndtrain_c = xgb.DMatrix(X_train_c, label=y_train_c)\ndvalid_c = xgb.DMatrix(X_valid_c, label=y_valid_c)\ndmissing_c = xgb.DMatrix(X_missing_credit)\n\n# Define parameters\nparams_c = {\n    'objective': 'reg:squarederror',\n    'eval_metric': 'rmse',\n    'seed': 42\n}\n\n# Train the model\nwatchlist = [(dtrain_c, 'train'), (dvalid_c, 'valid')]\nmodel_credit = xgb.train(\n    params_c,\n    dtrain_c,\n    num_boost_round=1000,\n    early_stopping_rounds=50,\n    evals=watchlist,\n    verbose_eval=50\n)\n\n# Predict missing 'Credit Score'\ncredit_score_pred = model_credit.predict(dmissing_c)\n\n# Assign predictions back to the missing data\n\n# Number of missing entries in train set\nnum_train_missing = len(X_train_missing_credit)\n\n# Split predictions back to train and test missing data\ntrain_credit_missing['Credit Score'] = credit_score_pred[:num_train_missing]\ntest_credit_missing['Credit Score'] = credit_score_pred[num_train_missing:]\n\n# Now update the original train and test datasets\ntrain.loc[train['Credit Score'].isnull(), 'Credit Score'] = train_credit_missing['Credit Score']\ntest.loc[test['Credit Score'].isnull(), 'Credit Score'] = test_credit_missing['Credit Score']\n\n# Verify that there are no missing values left in 'Credit Score'\nprint(\"Missing values in 'Credit Score' after imputation:\")\nprint(\"Train set:\", train['Credit Score'].isnull().sum())\nprint(\"Test set:\", test['Credit Score'].isnull().sum())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:42:41.767911Z","iopub.execute_input":"2024-12-01T11:42:41.768623Z","iopub.status.idle":"2024-12-01T11:42:49.991281Z","shell.execute_reply.started":"2024-12-01T11:42:41.768591Z","shell.execute_reply":"2024-12-01T11:42:49.99039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Save the Updated Datasets","metadata":{}},{"cell_type":"code","source":"train.to_csv(\"train_filled.csv\", index=False)\ntest.to_csv(\"test_filled.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:43:46.38571Z","iopub.execute_input":"2024-12-01T11:43:46.386052Z","iopub.status.idle":"2024-12-01T11:44:02.519451Z","shell.execute_reply.started":"2024-12-01T11:43:46.386024Z","shell.execute_reply":"2024-12-01T11:44:02.51871Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Review of Our New Data\n","metadata":{}},{"cell_type":"code","source":"# Reload the updated datasets\ntrain = pd.read_csv(\"train_filled.csv\")\ntest = pd.read_csv(\"test_filled.csv\")\n\n# Check for null values in each column\nprint(\"Null values in each column (Train):\")\nprint(train.isnull().sum())\n\nprint(\"\\nNull values in each column (Test):\")\nprint(test.isnull().sum())\n\n# Display data types of each column\nprint(\"\\nData types of each column (Train):\")\nprint(train.dtypes)\n\nprint(\"\\nData types of each column (Test):\")\nprint(test.dtypes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:45:03.889667Z","iopub.execute_input":"2024-12-01T11:45:03.890544Z","iopub.status.idle":"2024-12-01T11:45:07.033511Z","shell.execute_reply.started":"2024-12-01T11:45:03.890509Z","shell.execute_reply":"2024-12-01T11:45:07.032671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model Training & Evaluation","metadata":{}},{"cell_type":"markdown","source":"### Define the RMSLE Scorer","metadata":{}},{"cell_type":"code","source":"# Import necessary libraries\n# Some of these may be unnecesary\nfrom sklearn.metrics import mean_squared_log_error, make_scorer\nfrom sklearn.model_selection import KFold\nimport matplotlib.pyplot as plt\nfrom xgboost import plot_importance\n\n# Define RMSLE scorer\ndef rmsle(y_true, y_pred):\n    y_pred = np.clip(y_pred, 0, None)  # Ensure predictions are non-negative\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\nrmsle_scorer = make_scorer(rmsle, greater_is_better=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:49:54.577289Z","iopub.execute_input":"2024-12-01T11:49:54.578166Z","iopub.status.idle":"2024-12-01T11:49:54.582983Z","shell.execute_reply.started":"2024-12-01T11:49:54.578123Z","shell.execute_reply":"2024-12-01T11:49:54.58216Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Initialize the XGBoost Model","metadata":{}},{"cell_type":"code","source":"# Initialize XGBoost model\nmodel = xgb.XGBRegressor(\n    n_estimators=2000,\n    learning_rate=0.1,\n    max_depth=5,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    random_state=42,\n    objective='reg:squarederror',\n    tree_method='gpu_hist'  # Use 'gpu_hist' if GPU is available; else 'hist'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:49:55.712573Z","iopub.execute_input":"2024-12-01T11:49:55.712925Z","iopub.status.idle":"2024-12-01T11:49:55.717467Z","shell.execute_reply.started":"2024-12-01T11:49:55.712892Z","shell.execute_reply":"2024-12-01T11:49:55.716477Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Define Features and Target Variable","metadata":{}},{"cell_type":"code","source":"# Define X and y\nX = train.drop(columns=['Premium Amount', 'id'])  # Exclude target and 'id' columns\ny = train['Premium Amount']                       # Target variable","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:49:57.11244Z","iopub.execute_input":"2024-12-01T11:49:57.112781Z","iopub.status.idle":"2024-12-01T11:49:57.167653Z","shell.execute_reply.started":"2024-12-01T11:49:57.112752Z","shell.execute_reply":"2024-12-01T11:49:57.16689Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Perform Cross-Validation","metadata":{}},{"cell_type":"code","source":"# Define whether to perform cross-validation\nDO_CV = True  # Set to True to perform CV; False to skip\n\nif DO_CV:\n    kfold = KFold(n_splits=5, shuffle=True, random_state=42)\n    fold_scores = []\n    \n    # Perform custom cross-validation\n    for fold, (train_idx, val_idx) in enumerate(kfold.split(X, y), 1):\n        # Split data\n        X_train_fold, X_val_fold = X.iloc[train_idx], X.iloc[val_idx]\n        y_train_fold, y_val_fold = y.iloc[train_idx], y.iloc[val_idx]\n        \n        # Train the model\n        model.fit(X_train_fold, y_train_fold)\n        \n        # Make predictions\n        y_pred_fold = model.predict(X_val_fold)\n        y_pred_fold = np.clip(y_pred_fold, 0, None)  # Ensure no negative predictions\n        \n        # Calculate RMSLE for the fold\n        fold_rmsle = rmsle(y_val_fold, y_pred_fold)\n        fold_scores.append(fold_rmsle)\n        \n        # Print score for the current fold\n        print(f\"Fold {fold}: RMSLE = {fold_rmsle:.4f}\")\n    \n    # Summary of results\n    print(f\"\\nMean RMSLE: {np.mean(fold_scores):.4f}\")\n    print(f\"Standard Deviation of RMSLE: {np.std(fold_scores):.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:49:58.113908Z","iopub.execute_input":"2024-12-01T11:49:58.114486Z","iopub.status.idle":"2024-12-01T11:51:23.721044Z","shell.execute_reply.started":"2024-12-01T11:49:58.114455Z","shell.execute_reply":"2024-12-01T11:51:23.720127Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Train the Model on the Entire Training Data","metadata":{}},{"cell_type":"code","source":"model.fit(X, y) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:52:35.73331Z","iopub.execute_input":"2024-12-01T11:52:35.734048Z","iopub.status.idle":"2024-12-01T11:52:55.728496Z","shell.execute_reply.started":"2024-12-01T11:52:35.734017Z","shell.execute_reply":"2024-12-01T11:52:55.727535Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Plot Feature Importance","metadata":{}},{"cell_type":"code","source":"DO_PLOT_IMPORTANCE = True  # Set to True to plot; False to skip\n\nif DO_PLOT_IMPORTANCE:\n    # Plot top 10 feature importances\n    plt.figure(figsize=(12, 8))\n    plot_importance(model, max_num_features=10)\n    plt.title(\"Top 10 Feature Importances\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:52:55.729812Z","iopub.execute_input":"2024-12-01T11:52:55.730116Z","iopub.status.idle":"2024-12-01T11:52:56.031113Z","shell.execute_reply.started":"2024-12-01T11:52:55.730087Z","shell.execute_reply":"2024-12-01T11:52:56.030183Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Make Predictions on the Test Set","metadata":{}},{"cell_type":"code","source":"# Ensure 'Premium Amount' and 'id' are not included in features\ntest_features = test.drop(columns=['Premium Amount', 'id'], errors='ignore')\n\n# Make predictions using the trained model\ntest_predictions = model.predict(test_features)\ntest_predictions = np.clip(test_predictions, 0, None)  # Ensure no negative predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:52:56.032183Z","iopub.execute_input":"2024-12-01T11:52:56.032477Z","iopub.status.idle":"2024-12-01T11:52:56.648844Z","shell.execute_reply.started":"2024-12-01T11:52:56.032448Z","shell.execute_reply":"2024-12-01T11:52:56.647651Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Prepare the Submission File","metadata":{}},{"cell_type":"code","source":"# Prepare the submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test['id'],  # Use 'id' from the test dataset\n    'Premium Amount': test_predictions\n})\n\n# Save the submission file\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created successfully and saved to 'submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:52:56.65101Z","iopub.execute_input":"2024-12-01T11:52:56.651473Z","iopub.status.idle":"2024-12-01T11:52:57.672665Z","shell.execute_reply.started":"2024-12-01T11:52:56.651429Z","shell.execute_reply":"2024-12-01T11:52:57.671761Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Import Optuna and Libraries","metadata":{}},{"cell_type":"code","source":"import optuna\nfrom optuna.samplers import TPESampler\nimport xgboost as xgb\nfrom sklearn.model_selection import KFold\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:59:13.896591Z","iopub.execute_input":"2024-12-01T11:59:13.896927Z","iopub.status.idle":"2024-12-01T11:59:14.009171Z","shell.execute_reply.started":"2024-12-01T11:59:13.8969Z","shell.execute_reply":"2024-12-01T11:59:14.008267Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Objective Function For Optuna","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n    # Suggest hyperparameters\n    params = {\n        'n_estimators': trial.suggest_int('n_estimators', 500, 3000),\n        'learning_rate': trial.suggest_loguniform('learning_rate', 0.005, 0.2),\n        'max_depth': trial.suggest_int('max_depth', 3, 15),\n        'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n        'gamma': trial.suggest_float('gamma', 0, 10),\n        'reg_alpha': trial.suggest_float('reg_alpha', 0, 10),\n        'reg_lambda': trial.suggest_float('reg_lambda', 0, 10),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n        'objective': 'reg:squarederror',\n        'tree_method': 'gpu_hist',  # Use 'gpu_hist' if using GPU; else 'hist'\n        'random_state': 42,\n        'verbosity': 0\n    }\n    \n    # Initialize RMSLE scores list\n    rmsle_scores = []\n    \n    # Use K-Fold cross-validation\n    kfold = KFold(n_splits=5, shuffle=True, random_state=42)\n    \n    for train_idx, val_idx in kfold.split(X, y):\n        X_train_fold, X_val_fold = X.iloc[train_idx], X.iloc[val_idx]\n        y_train_fold, y_val_fold = y.iloc[train_idx], y.iloc[val_idx]\n        \n        # Create the model with current hyperparameters\n        model = xgb.XGBRegressor(**params)\n        \n        # Train the model\n        model.fit(\n            X_train_fold, y_train_fold,\n            eval_set=[(X_val_fold, y_val_fold)],\n            early_stopping_rounds=50,\n            verbose=False\n        )\n        \n        # Make predictions\n        y_pred = model.predict(X_val_fold)\n        y_pred = np.clip(y_pred, 0, None)  # Ensure no negative predictions\n        \n        # Calculate RMSLE for the current fold\n        score = rmsle(y_val_fold, y_pred)\n        rmsle_scores.append(score)\n    \n    # Return the mean RMSLE across all folds\n    return np.mean(rmsle_scores)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:59:16.745051Z","iopub.execute_input":"2024-12-01T11:59:16.74575Z","iopub.status.idle":"2024-12-01T11:59:16.753468Z","shell.execute_reply.started":"2024-12-01T11:59:16.745718Z","shell.execute_reply":"2024-12-01T11:59:16.752548Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Optuna Study to Find the Best Hyperparameters - Optional","metadata":{}},{"cell_type":"code","source":"# Create a study object and optimize the objective function\nsampler = TPESampler(seed=42)\nstudy = optuna.create_study(direction='minimize', sampler=sampler)\nstudy.optimize(objective, n_trials=50)\n\n# Print the best hyperparameters\nprint(\"Best hyperparameters:\")\nfor key, value in study.best_params.items():\n    print(f\"  {key}: {value:.4f}\" if isinstance(value, float) else f\"  {key}: {value}\")\n\nprint(f\"\\nBest RMSLE: {study.best_value:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T11:59:38.041376Z","iopub.execute_input":"2024-12-01T11:59:38.041713Z","iopub.status.idle":"2024-12-01T13:19:51.279012Z","shell.execute_reply.started":"2024-12-01T11:59:38.041684Z","shell.execute_reply":"2024-12-01T13:19:51.278116Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Best hyperparameters:\n  n_estimators: 1339\n  learning_rate: 0.0211\n  max_depth: 12\n  subsample: 0.7071\n  colsample_bytree: 0.9598\n  gamma: 1.4681\n  reg_alpha: 2.2229\n  reg_lambda: 8.4962\n  min_child_weight: 8\n\nBest RMSLE: 1.1435","metadata":{}},{"cell_type":"markdown","source":"### Train the Final Model with the Best Hyperparameters","metadata":{}},{"cell_type":"code","source":"# Update the model with the best hyperparameters\nbest_params = study.best_params\nbest_params['objective'] = 'reg:squarederror'\nbest_params['tree_method'] = 'gpu_hist'  # Use 'gpu_hist' if using GPU; else 'hist'\nbest_params['random_state'] = 42\nbest_params['verbosity'] = 0\n\n# Remove 'n_estimators' from params to set it separately (if needed)\nn_estimators = best_params.pop('n_estimators')\n\n# Train the final model on the entire training data\nfinal_model = xgb.XGBRegressor(\n    n_estimators=n_estimators,\n    **best_params\n)\n\nfinal_model.fit(\n    X, y,\n    eval_set=[(X, y)],\n    early_stopping_rounds=50,\n    verbose=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T13:19:51.280533Z","iopub.execute_input":"2024-12-01T13:19:51.280812Z","iopub.status.idle":"2024-12-01T13:20:53.528637Z","shell.execute_reply.started":"2024-12-01T13:19:51.280785Z","shell.execute_reply":"2024-12-01T13:20:53.527914Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Make Predictions - Submission File","metadata":{}},{"cell_type":"code","source":"# Ensure 'Premium Amount' and 'id' are not included in features\ntest_features = test.drop(columns=['Premium Amount', 'id'], errors='ignore')\n\n# Make predictions using the trained model\ntest_predictions = final_model.predict(test_features)\ntest_predictions = np.clip(test_predictions, 0, None)  # Ensure no negative predictions\n\n# Prepare the submission DataFrame\nsubmission = pd.DataFrame({\n    'id': test['id'],  # Use 'id' from the test dataset\n    'Premium Amount': test_predictions\n})\n\n# Save the submission file\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file created successfully and saved to 'submission.csv'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T13:20:53.529992Z","iopub.execute_input":"2024-12-01T13:20:53.530386Z","iopub.status.idle":"2024-12-01T13:20:55.352987Z","shell.execute_reply.started":"2024-12-01T13:20:53.530348Z","shell.execute_reply":"2024-12-01T13:20:55.352105Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Special Thanks to @merfarukelik**","metadata":{}}]}