{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 📥 Importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=UserWarning, module=\"seaborn\")\nwarnings.filterwarnings(\"ignore\", category=FutureWarning, module=\"seaborn\")\n\nimport tensorflow as tf\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.metrics import mean_squared_log_error, mean_squared_error, mean_absolute_error, r2_score\n\nimport optuna\nimport lightgbm as lgb\n\nimport torch\nfrom sklearn.pipeline import Pipeline\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:16.920616Z","iopub.execute_input":"2024-12-06T03:15:16.92131Z","iopub.status.idle":"2024-12-06T03:15:25.794218Z","shell.execute_reply.started":"2024-12-06T03:15:16.921277Z","shell.execute_reply":"2024-12-06T03:15:25.793253Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📥 Importing Dataset","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ntest = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:25.795958Z","iopub.execute_input":"2024-12-06T03:15:25.79678Z","iopub.status.idle":"2024-12-06T03:15:34.541088Z","shell.execute_reply.started":"2024-12-06T03:15:25.796737Z","shell.execute_reply":"2024-12-06T03:15:34.540319Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 🔎 Explotarory Data Analysis","metadata":{}},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:34.542178Z","iopub.execute_input":"2024-12-06T03:15:34.542469Z","iopub.status.idle":"2024-12-06T03:15:34.575126Z","shell.execute_reply.started":"2024-12-06T03:15:34.542442Z","shell.execute_reply":"2024-12-06T03:15:34.574373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:34.576931Z","iopub.execute_input":"2024-12-06T03:15:34.577196Z","iopub.status.idle":"2024-12-06T03:15:34.593929Z","shell.execute_reply.started":"2024-12-06T03:15:34.57717Z","shell.execute_reply":"2024-12-06T03:15:34.593085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:34.595045Z","iopub.execute_input":"2024-12-06T03:15:34.595492Z","iopub.status.idle":"2024-12-06T03:15:34.604863Z","shell.execute_reply.started":"2024-12-06T03:15:34.595451Z","shell.execute_reply":"2024-12-06T03:15:34.604074Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:34.606248Z","iopub.execute_input":"2024-12-06T03:15:34.606623Z","iopub.status.idle":"2024-12-06T03:15:35.164206Z","shell.execute_reply.started":"2024-12-06T03:15:34.606586Z","shell.execute_reply":"2024-12-06T03:15:35.163436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:35.165453Z","iopub.execute_input":"2024-12-06T03:15:35.165753Z","iopub.status.idle":"2024-12-06T03:15:35.696287Z","shell.execute_reply.started":"2024-12-06T03:15:35.165723Z","shell.execute_reply":"2024-12-06T03:15:35.695373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:35.697279Z","iopub.execute_input":"2024-12-06T03:15:35.697648Z","iopub.status.idle":"2024-12-06T03:15:36.246611Z","shell.execute_reply.started":"2024-12-06T03:15:35.697605Z","shell.execute_reply":"2024-12-06T03:15:36.245702Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save 'id' column for submission\ntest_ids = test['id']\n\n# Define the target column\ntarget_column = 'Premium Amount'\n\n# Select categorical and numerical columns (initial)\ncategorical_columns = train.select_dtypes(include=['object']).columns\nnumerical_columns = train.select_dtypes(exclude=['object']).columns\n\n# Print out column information\nprint(\"Target Column:\", target_column)\nprint(\"\\nCategorical Columns:\", categorical_columns.tolist())\nprint(\"\\nNumerical Columns:\", numerical_columns.tolist())\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:36.247895Z","iopub.execute_input":"2024-12-06T03:15:36.248253Z","iopub.status.idle":"2024-12-06T03:15:36.418295Z","shell.execute_reply.started":"2024-12-06T03:15:36.24821Z","shell.execute_reply":"2024-12-06T03:15:36.417391Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 📝 Dataset Overview and Findings\n\n## Key Statistics\n- **Entries**: 1,200,000  \n- **Columns**: 21  \n- **Missing Data**:\n  - **High Missingness** (~30%): `Occupation` (358,075), `Previous Claims` (364,029).\n  - **Moderate Missingness**: `Credit Score`, `Number of Dependents`, `Health Score`.\n  - **Low Missingness**: `Age`, `Annual Income`, `Marital Status`, `Customer Feedback`.\n\n## Data Types\n- **Numeric**: 9  \n- **Categorical**: 11  \n- **Integer**: 1, which is id.\n\n\n## Columns Overview\n\n### 1. **Floating-point (`float64`) Columns**\n- **`Age`**:  \n  - Range: 18–64  \n  - Mean: 41.14 years  \n  - Standard Deviation: 13.54 years  \n\n- **`Annual Income`**:  \n  - Range: 10,000–149,997  \n  - Mean: 32,745.22  \n  - Standard Deviation: 32,179.51  \n\n- **`Number of Dependents`**:  \n  - Range: 0–4  \n  - Mean: 2.01  \n  - Standard Deviation: 1.42  \n\n- **`Health Score`**:  \n  - Range: 2.01–58.97  \n  - Mean: 25.61  \n  - Standard Deviation: 12.20  \n\n- **`Previous Claims`**:  \n  - Range: 0–9  \n  - Mean: 1.00  \n  - Standard Deviation: 0.98  \n\n- **`Vehicle Age`**:  \n  - Range: 0–19  \n  - Mean: 9.57  \n  - Standard Deviation: 5.77  \n\n- **`Credit Score`**:  \n  - Range: 300–849  \n  - Mean: 592.92  \n  - Standard Deviation: 149.98  \n\n- **`Insurance Duration`**:  \n  - Range: 1–9  \n  - Mean: 5.02  \n  - Standard Deviation: 2.59  \n\n- **`Premium Amount`**:  \n  - Range: 200–4,999  \n  - Mean: 1,102.55  \n  - Standard Deviation: 864.99  \n\n---\n\n### 2. **Object (`object`) Columns**\n- **`Gender`**: Gender of the individual.  \n- **`Marital Status`**: Marital status of the individual.  \n- **`Education Level`**: Education level of the individual.  \n- **`Occupation`**: Occupation of the individual.  \n- **`Location`**: Geographic location.  \n- **`Policy Type`**: Type of insurance policy.  \n- **`Policy Start Date`**: Start date of the policy.  \n- **`Customer Feedback`**: Customer's feedback comments.  \n- **`Smoking Status`**: Indicates if the individual smokes.  \n- **`Exercise Frequency`**: Frequency of exercise by the individual.  \n- **`Property Type`**: Type of property owned by the individual.\n\n### Here the traget column is Premium Amount.\n","metadata":{}},{"cell_type":"markdown","source":"# 🔎 Visualization","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Identifying columns with missing values\ncols_with_na = train.columns[train.isnull().any()]\n\n# Create a mask of missing values (True for NaN, False otherwise)\nmissing_values = train[cols_with_na].isnull().values\n\n# Plotting using Matplotlib\nplt.figure(figsize=(10, 8))\nplt.imshow(missing_values, aspect='auto', cmap='viridis', interpolation='nearest')\nplt.colorbar(label='Missing Values')\nplt.title(\"Missing Values Heatmap\")\nplt.xticks(ticks=np.arange(len(cols_with_na)), labels=cols_with_na, rotation=90)\nplt.yticks([])  # No row labels\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:36.421062Z","iopub.execute_input":"2024-12-06T03:15:36.421355Z","iopub.status.idle":"2024-12-06T03:15:37.525096Z","shell.execute_reply.started":"2024-12-06T03:15:36.421314Z","shell.execute_reply":"2024-12-06T03:15:37.524301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = train\n#Filter out columns with more than 10 unique values for 'object' dtype and exclude datetime columns\nfor column in train_data.columns:\n    if train_data[column].dtype == 'object' and train_data[column].nunique() > 10:\n        continue  # Skip categorical columns with more than 10 unique values\n    if pd.api.types.is_datetime64_any_dtype(train_data[column]):\n        continue  # Skip datetime columns\n    \n    plt.figure(figsize=(10, 6))\n    \n    if train_data[column].dtype == 'object':\n        # Categorical data: Use a count plot\n        sns.countplot(data=train_data, y=column, palette='viridis')\n        plt.title(f'Count Plot of {column}', fontsize=16)\n        plt.xlabel('Count')\n        plt.ylabel(column)\n    else:\n        # Numerical data: Use a histogram\n        sns.histplot(train_data[column], kde=True, bins=30, color='blue')\n        plt.title(f'Distribution of {column}', fontsize=16)\n        plt.xlabel(column)\n        plt.ylabel('Frequency')\n    \n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:15:37.526221Z","iopub.execute_input":"2024-12-06T03:15:37.526619Z","iopub.status.idle":"2024-12-06T03:16:30.924565Z","shell.execute_reply.started":"2024-12-06T03:15:37.526575Z","shell.execute_reply":"2024-12-06T03:16:30.923666Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"def date(df):\n\n    df['Policy Start Date'] = pd.to_datetime(df['Policy Start Date'])\n    df['Year'] = df['Policy Start Date'].dt.year\n    df['Day'] = df['Policy Start Date'].dt.day\n    df['Month'] = df['Policy Start Date'].dt.month\n    df['Month_name'] = df['Policy Start Date'].dt.month_name()\n    df['Day_of_week'] = df['Policy Start Date'].dt.day_name()\n    df['Week'] = df['Policy Start Date'].dt.isocalendar().week\n    df['Year_sin'] = np.sin(2 * np.pi * df['Year'])\n    df['Year_cos'] = np.cos(2 * np.pi * df['Year'])\n    min_year = df['Year'].min()\n    max_year = df['Year'].max()\n    df['Year_sin'] = np.sin(2 * np.pi * (df['Year'] - min_year) / (max_year - min_year))\n    df['Year_cos'] = np.cos(2 * np.pi * (df['Year'] - min_year) / (max_year - min_year))\n    df['Month_sin'] = np.sin(2 * np.pi * df['Month'] / 12) \n    df['Month_cos'] = np.cos(2 * np.pi * df['Month'] / 12)\n    df['Day_sin'] = np.sin(2 * np.pi * df['Day'] / 31)  \n    df['Day_cos'] = np.cos(2 * np.pi * df['Day'] / 31)\n    df['Group']=(df['Year']-2020)*48+df['Month']*4+df['Day']//7\n    \n    df.drop('Policy Start Date', axis=1, inplace=True)\n\n    return df\n\n# Apply the date function to both datasets\ntrain_df = date(train)\ntest_df = date(test)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:16:30.925642Z","iopub.execute_input":"2024-12-06T03:16:30.925914Z","iopub.status.idle":"2024-12-06T03:16:33.598121Z","shell.execute_reply.started":"2024-12-06T03:16:30.925886Z","shell.execute_reply":"2024-12-06T03:16:33.597402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n# Define features and target\nnumerical_features = [\n    'Age', 'Annual Income', 'Number of Dependents', 'Health Score', \n    'Previous Claims', 'Vehicle Age', 'Credit Score', 'Insurance Duration', \n    'Year_sin', 'Year_cos', 'Month_sin', 'Month_cos', 'Day_sin', 'Day_cos'\n]\ncategorical_features = [\n    'Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n    'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', \n    'Property Type', 'Month_name', 'Day_of_week'\n]\ntarget_column = 'Premium Amount'\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:16:33.599222Z","iopub.execute_input":"2024-12-06T03:16:33.599614Z","iopub.status.idle":"2024-12-06T03:16:33.604498Z","shell.execute_reply.started":"2024-12-06T03:16:33.599566Z","shell.execute_reply":"2024-12-06T03:16:33.603618Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split train data into features and target\nX = train_df.drop(columns=[target_column, 'id', 'Group', 'Year', 'Month', 'Day', 'Week'])\ny = train_df[target_column]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:16:33.605744Z","iopub.execute_input":"2024-12-06T03:16:33.60614Z","iopub.status.idle":"2024-12-06T03:16:33.801897Z","shell.execute_reply.started":"2024-12-06T03:16:33.606102Z","shell.execute_reply":"2024-12-06T03:16:33.800807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocessing pipeline for numerical features\nnum_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    #('scaler', StandardScaler())                       # Scale numerical features\n])\n\n# Preprocessing pipeline for categorical features\ncat_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant', fill_value='Unknown')),  # Handle missing values\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))                      # Encode categorical features\n])\n\n# Combine pipelines into a ColumnTransformer\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', num_pipeline, numerical_features),\n        ('cat', cat_pipeline, categorical_features)\n    ]\n)\n\n# Preprocess train and test data\nX_processed = preprocessor.fit_transform(X)\ntest_processed = preprocessor.transform(test_df.drop(columns=['id', 'Group', 'Year', 'Month', 'Day', 'Week']))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:16:33.802998Z","iopub.execute_input":"2024-12-06T03:16:33.803287Z","iopub.status.idle":"2024-12-06T03:16:45.283081Z","shell.execute_reply.started":"2024-12-06T03:16:33.803258Z","shell.execute_reply":"2024-12-06T03:16:45.282322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data\nX_train, X_val, y_train, y_val = train_test_split(X_processed, y, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:16:45.283979Z","iopub.execute_input":"2024-12-06T03:16:45.284223Z","iopub.status.idle":"2024-12-06T03:16:45.584903Z","shell.execute_reply.started":"2024-12-06T03:16:45.284199Z","shell.execute_reply":"2024-12-06T03:16:45.584057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the RMSLE function\ndef root_mean_squared_log_error(y_true, y_pred):\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:16:45.585828Z","iopub.execute_input":"2024-12-06T03:16:45.586063Z","iopub.status.idle":"2024-12-06T03:16:45.590251Z","shell.execute_reply.started":"2024-12-06T03:16:45.586041Z","shell.execute_reply":"2024-12-06T03:16:45.589396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define Optuna optimization function\ndef objective(trial):\n    # Define parameter search space\n    param = {\n        \"objective\": \"regression\",\n        \"metric\": \"rmse\",\n        \"boosting_type\": trial.suggest_categorical(\"boosting_type\", [\"gbdt\", \"dart\"]),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 200, 512),\n        \"learning_rate\": trial.suggest_loguniform(\"learning_rate\", 1e-4, 1e-1),\n        \"feature_fraction\": trial.suggest_uniform(\"feature_fraction\", 0.6, 1.0),\n        \"bagging_fraction\": trial.suggest_uniform(\"bagging_fraction\", 0.6, 1.0),\n        \"bagging_freq\": trial.suggest_int(\"bagging_freq\", 5, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 20, 100),\n        \"max_depth\": trial.suggest_int(\"max_depth\", -1, 16),  # -1 means no limit\n        \"lambda_l1\": trial.suggest_loguniform(\"lambda_l1\", 1e-4, 10.0),\n        \"lambda_l2\": trial.suggest_loguniform(\"lambda_l2\", 1e-4, 10.0),\n        \"device_type\": \"cpu\",  \n        \"seed\" : 42\n\n    }\n\n    # Create a LightGBM dataset\n    dtrain = lgb.Dataset(X_train, label=y_train)\n    dval = lgb.Dataset(X_val, label=y_val, reference=dtrain)\n\n    # Train LightGBM model\n    model = lgb.train(\n        param,\n        dtrain,\n        valid_sets=[dval],\n    )\n\n    # Predict on validation set\n    y_val_pred = model.predict(X_val)\n    \n    # Compute RMSLE using sklearn's root_mean_squared_log_error\n    rmsle = root_mean_squared_log_error(y_val, np.maximum(y_val_pred, 0))\n    return rmsle\n\n# Run Optuna study\nstudy = optuna.create_study(direction=\"minimize\")\nstudy.optimize(objective, n_trials=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:16:45.591005Z","iopub.execute_input":"2024-12-06T03:16:45.591243Z","iopub.status.idle":"2024-12-06T03:17:01.723984Z","shell.execute_reply.started":"2024-12-06T03:16:45.591218Z","shell.execute_reply":"2024-12-06T03:17:01.723107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize or update the best_params dictionary\nbest_params = {\n    'boosting_type': 'dart',\n    'num_leaves': 384,\n    'learning_rate': 0.024680120465142227,\n    'feature_fraction': 0.9883068358315126,\n    'bagging_fraction': 0.7201712704805496,\n    'bagging_freq': 7,\n    'min_data_in_leaf': 50,\n    'max_depth': 15,\n    'lambda_l1': 0.0011290211269753322,\n    'lambda_l2': 3.056310541294088,\n    'device_type': \"gpu\",  # Enable GPU acceleration\n    'seed': 42\n}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:35.644877Z","iopub.execute_input":"2024-12-06T03:20:35.645226Z","iopub.status.idle":"2024-12-06T03:20:35.649988Z","shell.execute_reply.started":"2024-12-06T03:20:35.645193Z","shell.execute_reply":"2024-12-06T03:20:35.64913Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nimport lightgbm as lgb\n\n# Define K-Fold\nkf = KFold(n_splits=5, shuffle=True, random_state=42)  # 5 folds\n\n# Define LightGBM dataset\nlgb_data = lgb.Dataset(X_processed, label=y)\n\n# Add 'num_boost_round' for LightGBM cross-validation\nbest_params[\"num_boost_round\"] = 1000  # Maximum number of boosting iterations\nbest_params[\"early_stopping_rounds\"] = 50  # Stop early if no improvement\n\n# Perform K-Fold Cross-Validation\ncv_results = lgb.cv(\n    params=best_params,\n    train_set=lgb_data,\n    folds=kf,\n    metrics=[\"rmse\"],  # Metric to evaluate\n    stratified=False,  # Set to True if classification with imbalance\n    seed=42,\n    nfold=5,  # Ensure correct number of folds\n)\n\n# Retrieve the best score and number of boosting rounds\nbest_score = min(cv_results['rmse-mean'])  # Minimum RMSE score across folds\noptimal_rounds = len(cv_results['rmse-mean'])  # Total number of rounds used\n\nprint(\"Best RMSE:\", best_score)\nprint(\"Optimal Boosting Rounds:\", optimal_rounds)\n\n# Train final model using optimal number of boosting rounds\nfinal_model = lgb.train(\n    {**best_params, \"num_boost_round\": optimal_rounds},  # Include optimal rounds\n    lgb_data\n)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:37.342797Z","iopub.execute_input":"2024-12-06T03:20:37.343628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_model = lgb.train(\n    best_params,\n    lgb.Dataset(X_processed, label=y),\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:28.019151Z","iopub.status.idle":"2024-12-06T03:20:28.019501Z","shell.execute_reply.started":"2024-12-06T03:20:28.019322Z","shell.execute_reply":"2024-12-06T03:20:28.019351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom sklearn.metrics import mean_squared_log_error, mean_squared_error, mean_absolute_error, r2_score\n\n# Predictions\ny_pred = final_model.predict(X_processed)\n\n# Ensure y_pred and y have the same shape for metrics calculation\ny_pred = np.clip(y_pred, a_min=0, a_max=None)  # To avoid log errors due to negative predictions\n\n# Calculate performance metrics\nrmsle = root_mean_squared_log_error(y, y_pred)\nrmse = np.sqrt(mean_squared_error(y, y_pred))\nmae = mean_absolute_error(y, y_pred)\nr2 = r2_score(y, y_pred)\nmape = np.mean(np.abs((y - y_pred) / y)) * 100\n\n# Display performance metrics\nprint(f\"\\nPerformance Metrics:\\n{'-'*30}\")\nprint(f\"RMSLE: {rmsle:.4f}\")\nprint(f\"RMSE: {rmse:.4f}\")\nprint(f\"MAE: {mae:.4f}\")\nprint(f\"R²: {r2:.4f}\")\nprint(f\"MAPE: {mape:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:28.021031Z","iopub.status.idle":"2024-12-06T03:20:28.021312Z","shell.execute_reply.started":"2024-12-06T03:20:28.021178Z","shell.execute_reply":"2024-12-06T03:20:28.021192Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test set\ntest_predictions = final_model.predict(test_processed, num_iteration=final_model.best_iteration)\n\n# Prepare submission file\nsubmission = pd.DataFrame({'id': test_df['id'], 'Premium Amount': test_predictions})\nsubmission.to_csv(\"submission.csv\", index=False)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-06T03:20:28.022556Z","iopub.status.idle":"2024-12-06T03:20:28.022988Z","shell.execute_reply.started":"2024-12-06T03:20:28.022767Z","shell.execute_reply":"2024-12-06T03:20:28.022789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}