{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reload the necessary libraries and data due to the environment reset\nimport pandas as pd\n\n# Reloading dataset paths\ntrain_path = '/kaggle/input/playground-series-s4e12/train.csv'\ntest_path = '/kaggle/input/playground-series-s4e12/test.csv'\n\n# Load the datasets\ntrain_data = pd.read_csv(train_path)\ntest_data = pd.read_csv(test_path)\n\n# Display basic information about the training dataset\ntrain_data_info = train_data.info()\ntrain_data_head = train_data.head()\n\n# Display basic information about the test dataset\ntest_data_info = test_data.info()\ntest_data_head = test_data.head()\n\ntrain_data_info, train_data_head, test_data_info, test_data_head\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:42:22.421848Z","iopub.execute_input":"2025-01-02T11:42:22.422132Z","iopub.status.idle":"2025-01-02T11:42:31.964743Z","shell.execute_reply.started":"2025-01-02T11:42:22.42211Z","shell.execute_reply":"2025-01-02T11:42:31.963877Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get the total rows and columns for both datasets\ntrain_shape = train_data.shape\ntest_shape = test_data.shape\n\ntrain_shape, test_shape\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:42:31.965661Z","iopub.execute_input":"2025-01-02T11:42:31.966001Z","iopub.status.idle":"2025-01-02T11:42:31.971008Z","shell.execute_reply.started":"2025-01-02T11:42:31.965969Z","shell.execute_reply":"2025-01-02T11:42:31.970157Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Subset size for prototyping\nsubset_size = 100000\n\n# Load a subset of the training data\ntrain_data_subset = pd.read_csv(train_path, nrows=subset_size)\n\n# Drop irrelevant columns\ntrain_data_subset = train_data_subset.drop(columns=['Policy Start Date', 'id'])\n\n# Display the shape of the subset and a quick overview\ntrain_data_subset_shape = train_data_subset.shape\ntrain_data_subset_head = train_data_subset.head()\n\ntrain_data_subset_shape, train_data_subset_head","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:42:31.972038Z","iopub.execute_input":"2025-01-02T11:42:31.972343Z","iopub.status.idle":"2025-01-02T11:42:32.277346Z","shell.execute_reply.started":"2025-01-02T11:42:31.972313Z","shell.execute_reply":"2025-01-02T11:42:32.276439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the subset into features (X) and target (y)\ntarget_column = 'Premium Amount'\nX = train_data_subset.drop(columns=[target_column])\ny = train_data_subset[target_column]\n\n# Display shapes of features and target\nX_shape = X.shape\ny_shape = y.shape\n\nX_shape, y_shape\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:42:32.279224Z","iopub.execute_input":"2025-01-02T11:42:32.279467Z","iopub.status.idle":"2025-01-02T11:42:32.291059Z","shell.execute_reply.started":"2025-01-02T11:42:32.279447Z","shell.execute_reply":"2025-01-02T11:42:32.290282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode categorical variables using one-hot encoding\nX_encoded = pd.get_dummies(X, drop_first=True)\n\n# Display the shape of the encoded features\nX_encoded_shape = X_encoded.shape\n\nX_encoded_shape\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:42:32.292722Z","iopub.execute_input":"2025-01-02T11:42:32.29302Z","iopub.status.idle":"2025-01-02T11:42:32.385534Z","shell.execute_reply.started":"2025-01-02T11:42:32.292992Z","shell.execute_reply":"2025-01-02T11:42:32.384707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import necessary libraries\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV, cross_val_score\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\nimport xgboost as xgb\nfrom scipy.stats import uniform\n\n# Define RSMLE calculation\ndef rsmle(y_true, y_pred):\n    return np.sqrt(np.mean(np.square(np.log1p(y_pred) - np.log1p(y_true))))\n\n# Subset size for prototyping\nsubset_size = 100000\n\n# Load a subset of the training data\ntrain_data_subset = pd.read_csv(train_path, nrows=subset_size)\n\n# Drop irrelevant columns\ntrain_data_subset = train_data_subset.drop(columns=['Policy Start Date', 'id'])\n\n# Feature Engineering: Interaction and Transformation\ntrain_data_subset['Income_Health'] = train_data_subset['Annual Income'] * train_data_subset['Health Score']\ntrain_data_subset['Annual Income'] = np.log1p(train_data_subset['Annual Income'])\n\n# Split the subset into features and target\ntarget_column = 'Premium Amount'\nX = train_data_subset.drop(columns=[target_column])\ny = train_data_subset[target_column]\n\n# Encode categorical variables using one-hot encoding\nX_encoded = pd.get_dummies(X, drop_first=True)\n\n# Split into training and validation sets\nX_train, X_val, y_train, y_val = train_test_split(X_encoded, y, test_size=0.2, random_state=42)\n\n# Log-transform the target variable for training\ny_train_log = np.log1p(y_train)\n\n# Define parameter grid for RandomizedSearchCV\nparam_distributions = {\n    'learning_rate': uniform(0.01, 0.3),\n    'max_depth': [3, 4, 5, 6],\n    'n_estimators': [100, 200, 300, 500],\n    'subsample': uniform(0.5, 0.5),\n    'colsample_bytree': uniform(0.5, 0.5),\n    'min_child_weight': [1, 3, 5, 7],\n    'reg_alpha': uniform(0, 1),\n    'reg_lambda': uniform(1, 10)\n}\n\n# Initialize the XGBoost regressor\nxgb_model = xgb.XGBRegressor(random_state=42)\n\n# Perform RandomizedSearchCV for hyperparameter tuning\nrandom_search = RandomizedSearchCV(\n    estimator=xgb_model,\n    param_distributions=param_distributions,\n    n_iter=50,  # Number of random samples to evaluate\n    scoring='neg_root_mean_squared_error',\n    cv=3,\n    random_state=42,\n    verbose=1\n)\n\n# Fit the RandomizedSearchCV\nrandom_search.fit(X_train, y_train_log)\n\n# Get the best parameters\nbest_params = random_search.best_params_\nprint(f\"Best Parameters: {best_params}\")\n\n# Train the model with the best parameters\nbest_xgb_model = xgb.XGBRegressor(**best_params, random_state=42)\nbest_xgb_model.fit(\n    X_train, y_train_log,\n    early_stopping_rounds=20,\n    eval_set=[(X_val, np.log1p(y_val))],\n    verbose=True\n)\n\n# Make predictions and inverse log-transform\ny_pred_log = best_xgb_model.predict(X_val)\ny_pred = np.expm1(y_pred_log)\n\n# Cross-Validation on Training Data\ncross_val_rmse = cross_val_score(\n    best_xgb_model,\n    X_train,\n    y_train_log,\n    scoring='neg_root_mean_squared_error',\n    cv=5\n)\nprint(f\"Cross-Validation RMSE: {-cross_val_rmse.mean():.2f}\")\n\n# Calculate evaluation metrics\nmae = mean_absolute_error(y_val, y_pred)\nmse = mean_squared_error(y_val, y_pred)\nrmse = mse ** 0.5\nrsmle_score = rsmle(y_val, y_pred)\n\n# Print evaluation results\nprint(f\"Mean Absolute Error (MAE): {mae:.2f}\")\nprint(f\"Mean Squared Error (MSE): {mse:.2f}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.2f}\")\nprint(f\"Root Mean Squared Logarithmic Error (RSMLE): {rsmle_score:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:42:32.386429Z","iopub.execute_input":"2025-01-02T11:42:32.386763Z","iopub.status.idle":"2025-01-02T11:45:39.24918Z","shell.execute_reply.started":"2025-01-02T11:42:32.386732Z","shell.execute_reply":"2025-01-02T11:45:39.248471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import RandomizedSearchCV\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\nfrom lightgbm import early_stopping, log_evaluation\n\n# Log-transform the target variable for LightGBM\ny_train_log = np.log1p(y_train)\n\n# Define LightGBM model\nlgb_model = lgb.LGBMRegressor(random_state=42)\n\n# Define parameter grid for LightGBM\nparam_distributions = {\n    'num_leaves': [31, 50, 70],\n    'learning_rate': [0.01, 0.05, 0.1],\n    'n_estimators': [100, 200, 300],\n    'subsample': [0.8, 1.0],\n    'colsample_bytree': [0.8, 1.0],\n    'reg_alpha': [0, 0.1, 0.5],\n    'reg_lambda': [0, 1, 5, 10]\n}\n\n# Perform RandomizedSearchCV\nrandom_search = RandomizedSearchCV(\n    estimator=lgb_model,\n    param_distributions=param_distributions,\n    n_iter=50,\n    scoring='neg_root_mean_squared_error',\n    cv=3,\n    random_state=42,\n    verbose=1\n)\n\nrandom_search.fit(X_train, y_train_log)\n\n# Get the best parameters\nbest_params = random_search.best_params_\nprint(f\"Best Parameters for LightGBM: {best_params}\")\n\n# Train the LightGBM model with the best parameters and early stopping\nbest_lgb_model = lgb.LGBMRegressor(**best_params, random_state=42)\nbest_lgb_model.fit(\n    X_train,\n    y_train_log,\n    eval_set=[(X_val, np.log1p(y_val))],\n    callbacks=[\n        early_stopping(stopping_rounds=20),\n        log_evaluation(period=10)\n    ]\n)\n\n# Make predictions and inverse log-transform\ny_pred_log = best_lgb_model.predict(X_val)\ny_pred = np.expm1(y_pred_log)\n\n# Evaluate the model\nmae = mean_absolute_error(y_val, y_pred)\nmse = mean_squared_error(y_val, y_pred)\nrmse = mse ** 0.5\nrsmle_score = rsmle(y_val, y_pred)\n\nprint(f\"Mean Absolute Error (MAE): {mae:.2f}\")\nprint(f\"Mean Squared Error (MSE): {mse:.2f}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.2f}\")\nprint(f\"Root Mean Squared Logarithmic Error (RSMLE): {rsmle_score:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:45:39.249707Z","iopub.execute_input":"2025-01-02T11:45:39.249917Z","iopub.status.idle":"2025-01-02T11:47:56.32204Z","shell.execute_reply.started":"2025-01-02T11:45:39.249899Z","shell.execute_reply":"2025-01-02T11:47:56.321161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from catboost import CatBoostRegressor\n\n# Train the CatBoost model\ncat_model = CatBoostRegressor(\n    iterations=500,\n    learning_rate=0.1,\n    depth=6,\n    loss_function='RMSE',\n    random_seed=42,\n    verbose=100\n)\n\ncat_model.fit(X_train, y_train_log, eval_set=(X_val, np.log1p(y_val)), early_stopping_rounds=20)\n\n# Make predictions and inverse log-transform\ny_pred_log = cat_model.predict(X_val)\ny_pred = np.expm1(y_pred_log)\n\n# Evaluate the model\nmae = mean_absolute_error(y_val, y_pred)\nmse = mean_squared_error(y_val, y_pred)\nrmse = mse ** 0.5\nrsmle_score = rsmle(y_val, y_pred)\n\nprint(f\"Mean Absolute Error (MAE): {mae:.2f}\")\nprint(f\"Mean Squared Error (MSE): {mse:.2f}\")\nprint(f\"Root Mean Squared Error (RMSE): {rmse:.2f}\")\nprint(f\"Root Mean Squared Logarithmic Error (RSMLE): {rsmle_score:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:47:56.323007Z","iopub.execute_input":"2025-01-02T11:47:56.32355Z","iopub.status.idle":"2025-01-02T11:47:58.111727Z","shell.execute_reply.started":"2025-01-02T11:47:56.323526Z","shell.execute_reply":"2025-01-02T11:47:58.110997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import Ridge, ElasticNet\nfrom sklearn.ensemble import StackingRegressor\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.experimental import enable_hist_gradient_boosting  # noqa\nfrom sklearn.ensemble import HistGradientBoostingRegressor\n\n# Create a pipeline for ElasticNet with preprocessing\nelasticnet_pipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy='mean')),  # Handle missing values\n    ('scaler', StandardScaler()),                # Scale the features\n    ('elasticnet', ElasticNet(alpha=0.1, l1_ratio=0.5, random_state=42))  # ElasticNet model\n])\n\n# Add ElasticNet to the stacking ensemble\nstacking_model = StackingRegressor(\n    estimators=[\n        ('lgb', lgb.LGBMRegressor(**best_params, random_state=42)),\n        ('cat', CatBoostRegressor(iterations=500, learning_rate=0.05, depth=6, random_seed=42, loss_function='RMSE', verbose=False)),\n        ('xgb', xgb.XGBRegressor(n_estimators=200, learning_rate=0.05, max_depth=6, random_state=42)),\n        ('hgb', HistGradientBoostingRegressor(max_iter=200, learning_rate=0.05, max_depth=10, random_state=42)),\n        ('elasticnet', elasticnet_pipeline)\n    ],\n    final_estimator=Ridge(alpha=1.0)  # Meta-model\n)\n\n# Train the stacking model\nstacking_model.fit(X_train, y_train_log)\n\n# Predict with the stacking model and inverse log-transform\ny_pred_stack_log = stacking_model.predict(X_val)\ny_pred_stack = np.expm1(y_pred_stack_log)\n\n# Evaluate the stacking model\nmae_stack = mean_absolute_error(y_val, y_pred_stack)\nmse_stack = mean_squared_error(y_val, y_pred_stack)\nrmse_stack = mse_stack ** 0.5\nrsmle_stack = rsmle(y_val, y_pred_stack)\n\n# Print evaluation results\nprint(f\"Stacking Model MAE: {mae_stack:.2f}\")\nprint(f\"Stacking Model MSE: {mse_stack:.2f}\")\nprint(f\"Stacking Model RMSE: {rmse_stack:.2f}\")\nprint(f\"Stacking Model RSMLE: {rsmle_stack:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:47:58.112478Z","iopub.execute_input":"2025-01-02T11:47:58.112696Z","iopub.status.idle":"2025-01-02T11:48:32.037984Z","shell.execute_reply.started":"2025-01-02T11:47:58.112678Z","shell.execute_reply":"2025-01-02T11:48:32.03674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.experimental import enable_hist_gradient_boosting  # noqa\nfrom sklearn.ensemble import HistGradientBoostingRegressor\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error\n\n# Define Histogram Gradient Boosting model\nhgb_model = HistGradientBoostingRegressor(\n    max_iter=200,\n    learning_rate=0.05,\n    max_depth=10,\n    l2_regularization=1.0,\n    random_state=42\n)\n\n# Train the Histogram Gradient Boosting model\nhgb_model.fit(X_train, y_train_log)\n\n# Predict with Histogram Gradient Boosting and inverse log-transform\ny_pred_hgb_log = hgb_model.predict(X_val)\ny_pred_hgb = np.expm1(y_pred_hgb_log)\n\n# Evaluate the model\nmae_hgb = mean_absolute_error(y_val, y_pred_hgb)\nmse_hgb = mean_squared_error(y_val, y_pred_hgb)\nrmse_hgb = mse_hgb ** 0.5\nrsmle_hgb = rsmle(y_val, y_pred_hgb)\n\nprint(f\"Histogram Gradient Boosting MAE: {mae_hgb:.2f}\")\nprint(f\"Histogram Gradient Boosting MSE: {mse_hgb:.2f}\")\nprint(f\"Histogram Gradient Boosting RMSE: {rmse_hgb:.2f}\")\nprint(f\"Histogram Gradient Boosting RSMLE: {rsmle_hgb:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:48:32.039197Z","iopub.execute_input":"2025-01-02T11:48:32.040162Z","iopub.status.idle":"2025-01-02T11:48:32.868669Z","shell.execute_reply.started":"2025-01-02T11:48:32.040124Z","shell.execute_reply":"2025-01-02T11:48:32.868019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import Ridge\nfrom sklearn.ensemble import StackingRegressor\n\nfrom sklearn.linear_model import Ridge\nfrom sklearn.ensemble import StackingRegressor\n\nstacking_model = StackingRegressor(\n    estimators=[\n        ('lgb', lgb.LGBMRegressor(**best_params, random_state=42)),\n        ('cat', CatBoostRegressor(iterations=500, learning_rate=0.05, depth=6, random_seed=42, loss_function='RMSE', verbose=False)),\n        ('xgb', xgb.XGBRegressor(n_estimators=200, learning_rate=0.05, max_depth=6, random_state=42)),\n        ('hgb', HistGradientBoostingRegressor(max_iter=200, learning_rate=0.05, max_depth=10, random_state=42)),\n        ('elasticnet', elasticnet_pipeline)  # Preprocessing pipeline for ElasticNet\n    ],\n    final_estimator=Ridge(alpha=1.0)\n)\n\n\n# Train the stacking modela\nstacking_model.fit(X_train, y_train_log)\n\n# Predict with the stacking model and inverse log-transform\ny_pred_stack_log = stacking_model.predict(X_val)\ny_pred_stack = np.expm1(y_pred_stack_log)\n\n# Evaluate the stacking model\nmae_stack = mean_absolute_error(y_val, y_pred_stack)\nmse_stack = mean_squared_error(y_val, y_pred_stack)\nrmse_stack = mse_stack ** 0.5\nrsmle_stack = rsmle(y_val, y_pred_stack)\n\nprint(f\"Stacking Model MAE: {mae_stack:.2f}\")\nprint(f\"Stacking Model MSE: {mse_stack:.2f}\")\nprint(f\"Stacking Model RMSE: {rmse_stack:.2f}\")\nprint(f\"Stacking Model RSMLE: {rsmle_stack:.2f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T11:48:32.869171Z","iopub.execute_input":"2025-01-02T11:48:32.869374Z","iopub.status.idle":"2025-01-02T11:49:05.882072Z","shell.execute_reply.started":"2025-01-02T11:48:32.869356Z","shell.execute_reply":"2025-01-02T11:49:05.880055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nsubmission_file = pd.read_csv(\"/kaggle/input/playground-series-s4e12/sample_submission.csv\")\nsubmission_file.head()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-02T15:29:33.581622Z","iopub.execute_input":"2025-01-02T15:29:33.581937Z","iopub.status.idle":"2025-01-02T15:29:33.737829Z","shell.execute_reply.started":"2025-01-02T15:29:33.58191Z","shell.execute_reply":"2025-01-02T15:29:33.737064Z"}},"outputs":[],"execution_count":null}]}