{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Regression with an Insurance Dataset\n\nYour Goal: The objectives of this challenge is to predict insurance premiums based on various factors.","metadata":{}},{"cell_type":"markdown","source":"# Preparation","metadata":{}},{"cell_type":"markdown","source":"## Load libraries","metadata":{}},{"cell_type":"code","source":"# !pip install -U lightautoml[all]","metadata":{"trusted":true,"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-12-30T06:46:53.092319Z","iopub.execute_input":"2024-12-30T06:46:53.093165Z","iopub.status.idle":"2024-12-30T06:46:53.115758Z","shell.execute_reply.started":"2024-12-30T06:46:53.093124Z","shell.execute_reply":"2024-12-30T06:46:53.114799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from lightautoml.automl.presets.tabular_presets import TabularAutoML\n# from lightautoml.tasks import Task\n# import torch\n# import os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:46:53.117698Z","iopub.execute_input":"2024-12-30T06:46:53.118056Z","iopub.status.idle":"2024-12-30T06:46:53.123047Z","shell.execute_reply.started":"2024-12-30T06:46:53.118025Z","shell.execute_reply":"2024-12-30T06:46:53.121631Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd              # For data manipulation and analysis\nimport numpy as np               # For numerical computing\nfrom datetime import datetime\nimport scipy.stats as stats      # For statistical analysis\nimport math\nimport matplotlib                # For plotting and visualization\nimport matplotlib.pyplot as plt  \nfrom pandas.plotting import parallel_coordinates\nimport seaborn as sns            # For statistical data visualization\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-12-30T20:25:52.862856Z","iopub.execute_input":"2024-12-30T20:25:52.864145Z","iopub.status.idle":"2024-12-30T20:25:52.873158Z","shell.execute_reply.started":"2024-12-30T20:25:52.864083Z","shell.execute_reply":"2024-12-30T20:25:52.872058Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# For machine learning\nimport xgboost as xgb\nimport lightgbm as lgb\nimport catboost as cb\nfrom xgboost import XGBRegressor\nfrom lightgbm import LGBMRegressor\nfrom catboost import CatBoostRegressor, Pool\n\nfrom lightgbm import early_stopping\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (accuracy_score, precision_score, recall_score, make_scorer, mean_squared_error,\n                             f1_score, confusion_matrix, classification_report, mean_squared_log_error)\nfrom sklearn.model_selection import KFold, StratifiedKFold\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.preprocessing import StandardScaler, OrdinalEncoder, LabelEncoder, FunctionTransformer, OneHotEncoder\nimport category_encoders\nfrom category_encoders import TargetEncoder\nimport optuna","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:25:53.96057Z","iopub.execute_input":"2024-12-30T20:25:53.961524Z","iopub.status.idle":"2024-12-30T20:25:53.968042Z","shell.execute_reply.started":"2024-12-30T20:25:53.961484Z","shell.execute_reply":"2024-12-30T20:25:53.966968Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load datasets","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:55:17.330714Z","iopub.execute_input":"2024-12-30T20:55:17.33147Z","iopub.status.idle":"2024-12-30T20:55:23.537971Z","shell.execute_reply.started":"2024-12-30T20:55:17.331431Z","shell.execute_reply":"2024-12-30T20:55:23.536875Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis","metadata":{}},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:47:08.085929Z","iopub.execute_input":"2024-12-30T06:47:08.086383Z","iopub.status.idle":"2024-12-30T06:47:08.128817Z","shell.execute_reply.started":"2024-12-30T06:47:08.086337Z","shell.execute_reply":"2024-12-30T06:47:08.127575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:47:08.130135Z","iopub.execute_input":"2024-12-30T06:47:08.130559Z","iopub.status.idle":"2024-12-30T06:47:08.796809Z","shell.execute_reply.started":"2024-12-30T06:47:08.130527Z","shell.execute_reply":"2024-12-30T06:47:08.795425Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Univariate Analysis\n\nHow does each explanatory **numerical feature** and target feature distribute?","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (10,6))\n\nax = sns.histplot(data = df_train,\n             x = 'Premium Amount',\n             bins = 20,\n             kde = True,\n             color = '#FFBE98')\nax.set_title('Distribution of Premium Amount ($)')\nax.set_ylabel('')\n\nsns.despine(top = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:47:08.798579Z","iopub.execute_input":"2024-12-30T06:47:08.798908Z","iopub.status.idle":"2024-12-30T06:47:14.090317Z","shell.execute_reply.started":"2024-12-30T06:47:08.798877Z","shell.execute_reply":"2024-12-30T06:47:14.089121Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = [col for col in df_test.columns if df_test[col].dtypes in ['float']]\ncat_cols = [col for col in df_test.columns if df_test[col].dtypes in ['object']]\n\nfig, axes = plt.subplots(4, 2, figsize=(20, 25))\n \nfor i, column in enumerate(num_cols):\n    plt.subplots_adjust(top = 0.85)\n    ax = sns.histplot(data = df_train, \n                x = column, \n                bins = 20,\n                kde = True,\n                ax = axes[i // 2, i % 2])\n    ax.set_title('Distribution of ' + column)\n    \nfig.tight_layout(h_pad = 2)\nplt.subplots_adjust(top = 0.93)\nplt.suptitle('Distribution of Explanatory Features', fontsize = 20)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:47:14.091643Z","iopub.execute_input":"2024-12-30T06:47:14.091942Z","iopub.status.idle":"2024-12-30T06:47:51.900348Z","shell.execute_reply.started":"2024-12-30T06:47:14.091914Z","shell.execute_reply":"2024-12-30T06:47:51.899289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"age_bin = [0, 20, 40, 60, 80]\nincome_bin = [0, 50000, 100000, 150000, 200000]\nhealth_bin = [0, 10, 20, 30, 40, 50, 60]\nvehicle_bin = [0, 5, 10, 15, 20]\ncredit_bin = [0, 200, 400, 600, 800]\n\n# Binning with string labels\ndf_train['age_bin'] = pd.cut(df_train['Age'], bins=age_bin, labels=[f\"{age_bin[i]}-{age_bin[i+1]}\" for i in range(len(age_bin) - 1)])\ndf_train['income_bin'] = pd.cut(df_train['Annual Income'], bins=income_bin, labels=[f\"{income_bin[i]}-{income_bin[i+1]}\" for i in range(len(income_bin) - 1)])\ndf_train['health_bin'] = pd.cut(df_train['Health Score'], bins=health_bin, labels=[f\"{health_bin[i]}-{health_bin[i+1]}\" for i in range(len(health_bin) - 1)])\ndf_train['vehicle_bin'] = pd.cut(df_train['Vehicle Age'], bins=vehicle_bin, labels=[f\"{vehicle_bin[i]}-{vehicle_bin[i+1]}\" for i in range(len(vehicle_bin) - 1)])\ndf_train['credit_bin'] = pd.cut(df_train['Credit Score'], bins=credit_bin, labels=[f\"{credit_bin[i]}-{credit_bin[i+1]}\" for i in range(len(credit_bin) - 1)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:47:51.903996Z","iopub.execute_input":"2024-12-30T06:47:51.904362Z","iopub.status.idle":"2024-12-30T06:47:52.081014Z","shell.execute_reply.started":"2024-12-30T06:47:51.904329Z","shell.execute_reply":"2024-12-30T06:47:52.07981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(5, 1, figsize=(20, 30)) \naxes = axes.flatten()  # Flatten the grid for easy indexing\nbin_cols = ['age_bin', 'income_bin', 'health_bin', 'vehicle_bin', 'credit_bin']\n\nfor i, column in enumerate(bin_cols):\n    # Main chart\n    sns.countplot(data=df_train, x=column, color='#FFBE98', ax=axes[i])\n    axes[i].set_title(f'Distribution of {column}')\n    axes[i].set_ylabel('Count')\n    axes[i].set_xlabel(column)\n    \n    # Secondary axis for average premium\n    ax2 = axes[i].twinx()\n    avg_premium = df_train.groupby(column)['Premium Amount'].mean()\n    ax2.plot(avg_premium.index, avg_premium.values, color='#1873D3', marker='o', label='Avg Premium')\n    ax2.set_ylabel('Avg Premium Amount ($)')\n    \n    # Legends and adjustments\n    ax2.legend(loc='upper right')\n    sns.despine(top=True, right=False, ax=axes[i])\n    sns.despine(top=True, right=False, ax=ax2)\n\n# Remove empty subplots\nfor j in range(len(bin_cols), len(axes)):\n    fig.delaxes(axes[j])\n\nfig.tight_layout(h_pad=3)\nplt.subplots_adjust(top=0.93)\nplt.suptitle('Distribution of Numerical Features and Average Premium Amounts', fontsize=23)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:47:52.082283Z","iopub.execute_input":"2024-12-30T06:47:52.082613Z","iopub.status.idle":"2024-12-30T06:47:54.270156Z","shell.execute_reply.started":"2024-12-30T06:47:52.08258Z","shell.execute_reply":"2024-12-30T06:47:54.269083Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"How does each categorical feature distributed?","metadata":{}},{"cell_type":"code","source":"cat_cols = ['Gender','Marital Status','Education Level','Occupation','Location','Policy Type','Customer Feedback','Smoking Status','Exercise Frequency','Property Type']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:47:54.271834Z","iopub.execute_input":"2024-12-30T06:47:54.272262Z","iopub.status.idle":"2024-12-30T06:47:54.278608Z","shell.execute_reply.started":"2024-12-30T06:47:54.272196Z","shell.execute_reply":"2024-12-30T06:47:54.277287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(5, 2, figsize=(20, 30)) \naxes = axes.flatten()  # Flatten the grid for easy indexing\n\nfor i, column in enumerate(cat_cols):\n    # Main chart\n    sns.countplot(data=df_train, x=column, color='#FFBE98', ax=axes[i])\n    axes[i].set_title(f'Distribution of {column}')\n    axes[i].set_ylabel('Count')\n    axes[i].set_xlabel(column)\n    \n    # Secondary axis for average premium\n    ax2 = axes[i].twinx()\n    avg_premium = df_train.groupby(column)['Premium Amount'].mean()\n    ax2.plot(avg_premium.index, avg_premium.values, color='#1873D3', marker='o', label='Avg Premium')\n    ax2.set_ylabel('Avg Premium Amount ($)')\n    \n    # Legends and adjustments\n    ax2.legend(loc='upper right')\n    sns.despine(top=True, right=False, ax=axes[i])\n    sns.despine(top=True, right=False, ax=ax2)\n\n# Remove empty subplots\nfor j in range(len(cat_cols), len(axes)):\n    fig.delaxes(axes[j])\n\nfig.tight_layout(h_pad=3)\nplt.subplots_adjust(top=0.93)\nplt.suptitle('Distribution of Categorical Features and Average Premium Amounts', fontsize=23)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:47:54.279922Z","iopub.execute_input":"2024-12-30T06:47:54.280334Z","iopub.status.idle":"2024-12-30T06:48:05.256752Z","shell.execute_reply.started":"2024-12-30T06:47:54.280302Z","shell.execute_reply":"2024-12-30T06:48:05.255751Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Multivariate Analysis","metadata":{}},{"cell_type":"code","source":"# Correlation Matrix Heatmap\nfig, ax = plt.subplots(figsize = (25,15))\ndf_features = df_train.select_dtypes(include='float')\ncorr = df_features.corr()\nhm = sns.heatmap(corr,\n                annot = True,\n                ax = ax,\n                cmap = sns.color_palette(\"vlag\", as_cmap = True),\n                fmt = '.5f')\nfig.subplots_adjust(top = 0.95)\nplt.suptitle('Predictors Correlation Heatmap', fontsize = 14)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:48:05.258273Z","iopub.execute_input":"2024-12-30T06:48:05.258684Z","iopub.status.idle":"2024-12-30T06:48:06.229696Z","shell.execute_reply.started":"2024-12-30T06:48:05.258642Z","shell.execute_reply":"2024-12-30T06:48:06.228594Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Statistical Analysis","metadata":{}},{"cell_type":"markdown","source":"## One-Way ANOVA or Kruskal-Wallis Test\nANOVA assumptions\nLike other types of statistical methods, ANOVA compares the means of different groups and shows you if there are any statistical differences between the means. ANOVA is classified as an omnibus test statistic. This means that it can’t tell you which specific groups were statistically significantly different from each other, only that at least two of the groups were.\n\nANOVA relies on three main assumptions that must be met for the test results to be valid.\n1. Normality\nThe first assumption is that the groups each fall into what is called a normal distribution. This means that the groups should have a bell-curve distribution with few or no outliers.\n\n2. Homogeneity of variance\nAlso known as homoscedasticity, this means that the variances between each group are the same.\n\n3. Independence\nThe final assumption is that each value is independent from each other. This means, for example, that unlike a conjoint analysis the same person shouldn’t be measured multiple times.\n\nANOVA Hypotheses\n- Null hypothesis: Groups means are equal (no variation in means of groups)\nH0: $\\mu_1 = \\mu_2 = ... = \\mu_p$\n\n- Alternative hypothesis: At least, one group mean is different from other groups\nH1: All $\\mu$ are not equal\n\nHence, before performing One-Way ANOVA Test, we will test if the assumptions are satisfied. If not, we will perform Kruskal-Wallis Test.","metadata":{}},{"cell_type":"code","source":"# Importing libraries \nimport statsmodels.api as sm \nfrom statsmodels.formula.api import ols ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:48:06.230916Z","iopub.execute_input":"2024-12-30T06:48:06.231246Z","iopub.status.idle":"2024-12-30T06:48:07.381518Z","shell.execute_reply.started":"2024-12-30T06:48:06.231182Z","shell.execute_reply":"2024-12-30T06:48:07.380265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import scipy.stats as stats\nfrom statsmodels.stats.diagnostic import lilliefors\n\n# Function to verify assumptions and perform Kruskal-Wallis test\ndef check_assumptions_and_test(df, feature, target):\n    # Check for normality of target distribution in each group\n    groups = df.groupby(feature)[target].apply(list)\n    normality_results = {}\n    \n    for group, values in groups.items():\n        # Perform Lilliefors test for normality (Kolmogorov-Smirnov for small data)\n        _, p_value = lilliefors(values, dist='norm')\n        normality_results[group] = p_value\n    \n    # Check for homogeneity of variances using Levene's test\n    groups_list = [df[df[feature] == g][target] for g in df[feature].unique()]\n    _, levene_p = stats.levene(*groups_list, center='mean')\n    \n    print(f\"Normality Test (Lilliefors p-values): {normality_results}\")\n    print(f\"Levene's Test for Homogeneity of Variances (p-value): {levene_p}\")\n    \n    # If assumptions are not satisfied, perform Kruskal-Wallis test\n    if any(p < 0.05 for p in normality_results.values()) or levene_p < 0.05:\n        print(\"Assumptions violated. Performing Kruskal-Wallis test.\")\n        _, kruskal_p = stats.kruskal(*groups_list)\n        print(f\"Kruskal-Wallis Test p-value: {kruskal_p}\")\n    else:\n        print(\"Assumptions satisfied. No need for Kruskal-Wallis.\")\n\n# Example usage with your dataset\n# Replace 'feature' and 'target' with the column names from your dataset\nfor feature in ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location',\n                'Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']:\n    print(f\"Testing feature: {feature}\")\n    check_assumptions_and_test(df_train, feature, 'Premium Amount')\n    print(\"\\n\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:48:07.383437Z","iopub.execute_input":"2024-12-30T06:48:07.384024Z","iopub.status.idle":"2024-12-30T06:48:20.149506Z","shell.execute_reply.started":"2024-12-30T06:48:07.38399Z","shell.execute_reply":"2024-12-30T06:48:20.148432Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Interpretation of Results:\n1. Normality Assumption:\n\n- For all features, the Lilliefors test consistently indicates non-normality (p-value < 0.05).\n- Normality assumption is violated for every feature.\n\n2. Homogeneity of Variances:\n\n- Levene's test results show mixed outcomes:\n    - Some features, like Gender, Location, and Smoking Status, satisfy variance homogeneity (p-value > 0.05).\n    - For Property Type, Levene's test indicates variance heterogeneity (p-value < 0.05).\n    - For some features (Marital Status, Occupation, Customer Feedback), nan results suggest issues with the test, possibly due to group size or data inconsistencies.\n\n3. Kruskal-Wallis Test:\n\n- Significance: None of the features tested (including Gender, Education Level, Policy Type) showed significant differences in medians across groups (p-value > 0.05 for all).\n- For features like Marital Status, Occupation, and Customer Feedback, nan p-values suggest further investigation is needed (e.g., checking for missing data or issues in grouping).\n\n\nNext Steps:\n\nSince Kruskal-Wallis results are not significant for any feature, consider:\n- Feature engineering or transformation (e.g., binning, encoding, interaction terms).\n\nUse other feature selection methods (e.g., mutual information, correlation analysis).","metadata":{}},{"cell_type":"markdown","source":"## Cramér's V Test\nMeasures the association between two nominal variables for each categorical pairs as followed.\n1. Compute a contingency table for each pair of categorical columns.\n2. Use the chi-squared test to calculate the chi-squared statistic.\n3. Apply Cramér's V formula","metadata":{}},{"cell_type":"code","source":"from scipy.stats import chi2_contingency\n\ndef cramers_v(x, y):\n    \"\"\"Calculate Cramér's V between two categorical variables.\"\"\"\n    contingency_table = pd.crosstab(x, y)\n    chi2, _, _, _ = chi2_contingency(contingency_table)\n    n = contingency_table.sum().sum()\n    r, k = contingency_table.shape\n    return np.sqrt(chi2 / (n * (min(r, k) - 1)))\n\n# List of categorical columns\ncategorical_cols = ['Gender', 'Marital Status', 'Education Level', 'Occupation', 'Location','Policy Type', 'Customer Feedback', 'Smoking Status', 'Exercise Frequency', 'Property Type']\n\n# Create an empty DataFrame to store results\nresults = pd.DataFrame(index=categorical_cols, columns=categorical_cols)\n\n# Calculate Cramér's V for each pair of categorical columns\nfor col1 in categorical_cols:\n    for col2 in categorical_cols:\n        if col1 == col2:\n            results.loc[col1, col2] = 1.0  # Perfect correlation with itself\n        else:\n            results.loc[col1, col2] = cramers_v(df_train[col1], df_train[col2])\n\n# Convert results to numeric for better display\nresults = results.astype(float)\n\n# Display the correlation matrix\nprint(results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:48:20.15119Z","iopub.execute_input":"2024-12-30T06:48:20.151574Z","iopub.status.idle":"2024-12-30T06:48:41.074473Z","shell.execute_reply.started":"2024-12-30T06:48:20.151538Z","shell.execute_reply":"2024-12-30T06:48:41.073483Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Interpretation of Results:\n- All Cramér’s V values are very close to zero, indicating weak associations between the categorical variables in your dataset.\n- Most of the correlations are near zero, suggesting that the features are largely independent of each other in terms of their categorical distribution.\n\nNext Steps for Feature Engineering:\n- Explore interactions: While individual features show weak correlation, combinations of features (e.g., gender with marital status) may offer more insightful patterns. Consider adding interaction terms.","metadata":{}},{"cell_type":"markdown","source":"# Train-Test Split","metadata":{}},{"cell_type":"code","source":"X = df_train.drop(columns = ['Premium Amount', 'id'], axis = 1)\ny = df_train['Premium Amount']\ny_log = np.log1p(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:55:36.06546Z","iopub.execute_input":"2024-12-30T20:55:36.066636Z","iopub.status.idle":"2024-12-30T20:55:36.238989Z","shell.execute_reply.started":"2024-12-30T20:55:36.066551Z","shell.execute_reply":"2024-12-30T20:55:36.237802Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the dataset into training and validation sets\nX_train, X_valid, y_train, y_valid = train_test_split(X, y_log, test_size=0.2, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:55:38.285724Z","iopub.execute_input":"2024-12-30T20:55:38.286134Z","iopub.status.idle":"2024-12-30T20:55:39.058487Z","shell.execute_reply.started":"2024-12-30T20:55:38.286094Z","shell.execute_reply":"2024-12-30T20:55:39.057369Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"def extract_date_components(df, date_column):\n    # Convert the column to datetime type\n    df[date_column] = pd.to_datetime(df[date_column])\n    # Extract Year, Month, and Day\n    df['Year'] = df[date_column].dt.year\n    df['Month'] = df[date_column].dt.month\n    df['Day'] = df[date_column].dt.day\n    # Drop the original column\n    df.drop(columns=[date_column], inplace=True)\n    return df\n\ndef log_transform(df):\n    df['Log Annual Income'] = np.log(df['Annual Income']+1)\n    # Drop the original column\n    df.drop(columns=['Annual Income'], inplace=True)\n    return df\n\nfor df in [X_train, X_valid, df_test]:\n    extract_date_components(df, 'Policy Start Date')\n    log_transform(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:55:46.904166Z","iopub.execute_input":"2024-12-30T20:55:46.904553Z","iopub.status.idle":"2024-12-30T20:55:48.525291Z","shell.execute_reply.started":"2024-12-30T20:55:46.904509Z","shell.execute_reply":"2024-12-30T20:55:48.524108Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def stat_features(df):\n#     # Numerical Data\n#     df['income_per_dependents'] = df['Annual Income'] * df['Number of Dependents']\n#     df['income_age'] = df['Annual Income'] / df['Age']\n#     df['vehical_insurance'] = df['Vehicle Age'] - df['Insurance Duration']\n#     df['health_credit'] = df['Health Score'] + df['Credit Score']\n#     df['claim_credit'] = df['Previous Claims'] * df['Credit Score']\n#     df['claim_income'] = df['Previous Claims'] * df['Annual Income']\n#     df['health_income'] = df['Health Score'] + df['Annual Income']\n#     df['credit_income'] = df['Credit Score'] + df['Annual Income']\n#     # Categorical Data\n#     # df['gender_education'] = df['Education Level'] + df['Gender']\n#     # df['policy_marital'] = df['Policy Type'] + df['Marital Status']\n#     # df['exercise_marital'] = df['Exercise Frequency'] + df['Marital Status']\n#     # df['location_property'] = df['Location'] + df['Property Type']\n#     # df['smoke_location'] = df['Smoking Status'] + df['Location']\n#     # df['smoke_exercise'] = df['Exercise Frequency'] + df['Smoking Status']\n#     return df\n\n# X_train = stat_features(X_train)\n# X_valid = stat_features(X_valid)\n# df_train = stat_features(df_train)\n# df_test = stat_features(df_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T06:48:44.284775Z","iopub.execute_input":"2024-12-30T06:48:44.285251Z","iopub.status.idle":"2024-12-30T06:48:44.291274Z","shell.execute_reply.started":"2024-12-30T06:48:44.285181Z","shell.execute_reply":"2024-12-30T06:48:44.29011Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"num_cols = [col for col in df_test.columns if df_test[col].dtypes in ['float', 'int32']]\nprint('Numerical Features \\n', num_cols)\ncat_cols = [col for col in df_test.columns if df_test[col].dtypes in ['object']]\nprint('Categorical Features \\n', cat_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:55:52.407125Z","iopub.execute_input":"2024-12-30T20:55:52.407532Z","iopub.status.idle":"2024-12-30T20:55:52.415458Z","shell.execute_reply.started":"2024-12-30T20:55:52.407499Z","shell.execute_reply":"2024-12-30T20:55:52.414203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define preprocessing pipelines\nnumerical_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ('scaler', StandardScaler()),\n    ('convert_to_float32', FunctionTransformer(lambda x: x.astype(np.float32)))\n])\n\ncategorical_pipeline = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='constant', fill_value='missing')),\n    ('target', TargetEncoder())\n])\n\n# Combine the numerical and categorical pipelines\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_pipeline, num_cols),\n        ('cat', categorical_pipeline, cat_cols)\n    ]\n)\n\npreprocessor.set_output(transform=\"pandas\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:55:54.310943Z","iopub.execute_input":"2024-12-30T20:55:54.311313Z","iopub.status.idle":"2024-12-30T20:55:54.350683Z","shell.execute_reply.started":"2024-12-30T20:55:54.311282Z","shell.execute_reply":"2024-12-30T20:55:54.34967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Apply the transformations to the training and validation sets\nX_train_processed = preprocessor.fit_transform(X_train, y_train)\nX_valid_processed = preprocessor.transform(X_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:55:58.555372Z","iopub.execute_input":"2024-12-30T20:55:58.555806Z","iopub.status.idle":"2024-12-30T20:56:08.909411Z","shell.execute_reply.started":"2024-12-30T20:55:58.555761Z","shell.execute_reply":"2024-12-30T20:56:08.908241Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Machine Learning","metadata":{}},{"cell_type":"markdown","source":"## Define Metrics","metadata":{}},{"cell_type":"code","source":"# rmsle with log-transformed y\ndef rmsle(y_true, y_pred):\n    y_pred = np.maximum(y_pred, np.min(y_true))\n    return np.sqrt(mean_squared_error(y_true, y_pred))\n\nRMSLE = make_scorer(rmsle, greater_is_better=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:56:08.911379Z","iopub.execute_input":"2024-12-30T20:56:08.911763Z","iopub.status.idle":"2024-12-30T20:56:08.917344Z","shell.execute_reply.started":"2024-12-30T20:56:08.91173Z","shell.execute_reply":"2024-12-30T20:56:08.916262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def cross_val_score_log(model, X, y, n_splits=5, random_state=42):\n    \n    \"\"\"\"\n    Perform cross-validation and return scores using log-transformed y.\n    \"\"\"\n\n    kf = KFold(n_splits = 5, shuffle=True, random_state=random_state)\n    scores = []\n\n    for train_index, test_index in kf.split(X):\n        X_train_split, X_valid_split = X.iloc[train_idx], X.iloc[val_idx]\n        y_train_split, y_valid_split = y_log.iloc[train_idx], y_log.iloc[val_idx]\n        \n        model.fit(X_train_split, y_train_split)\n        preds = model.predict(X_valid_split)\n        \n        # Calculate the score directly in the logarithmic scale\n        score = np.sqrt(mean_squared_error(y_valid_split, preds))\n        scores.append(score)\n\n    mean_score = np.mean(scores)\n    return mean_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:56:08.918575Z","iopub.execute_input":"2024-12-30T20:56:08.918944Z","iopub.status.idle":"2024-12-30T20:56:08.932293Z","shell.execute_reply.started":"2024-12-30T20:56:08.918906Z","shell.execute_reply":"2024-12-30T20:56:08.931038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# rmsle with original y\ndef rmsle_original(y_true, y_pred):\n    y_pred = np.maximum(y_pred, 20)\n    return np.sqrt(mean_squared_log_error(y_true, y_pred))\n\nRMSLE_ORIGINAl = make_scorer(rmsle_original, greater_is_better=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:56:08.934449Z","iopub.execute_input":"2024-12-30T20:56:08.934896Z","iopub.status.idle":"2024-12-30T20:56:08.942911Z","shell.execute_reply.started":"2024-12-30T20:56:08.934862Z","shell.execute_reply":"2024-12-30T20:56:08.941888Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Selection","metadata":{}},{"cell_type":"code","source":"# Initialize the model\nmodel = XGBRegressor(random_state=42, \n                     tree_method = 'hist', \n                     device = 'cuda', \n                     n_jobs = -1)\n\n# Calculate the CV score with all features\ncv_score = -cross_val_score(model, X_train_processed, np.expm1(y_train), cv=5, scoring = RMSLE_ORIGINAl).mean()\nprint(f'CV score with all features: {cv_score}')\n\n# Store the results in a list\nresults = []\n\n# Loop through each feature and calculate the CV score without that feature\nfor feature in X_train_processed.columns:\n    X_temp = X_train_processed.drop(feature, axis=1)\n    cv_score_temp = -cross_val_score(model, X_temp, np.expm1(y_train), cv=5, scoring = RMSLE_ORIGINAl).mean()\n    print(f'CV score without {feature}: {cv_score_temp}')\n    results.append((feature, cv_score_temp))\n\n# Sort the results in ascending order of CV scores\nresults.sort(key=lambda x: x[1])\n\n# Store features whose removal results in a lower CV score\nlower_cv_features = [feature for feature, score in results if score < cv_score]\n\n# Print the sorted results and the features with lower CV scores\nprint(\"\")\nprint(\"Features whose removal results in a lower CV score:\")\nprint(\"\")\nfor feature, score in results:\n    if score < cv_score:\n        print(f'- {feature}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T20:56:15.146906Z","iopub.execute_input":"2024-12-30T20:56:15.147259Z","iopub.status.idle":"2024-12-30T21:04:09.910653Z","shell.execute_reply.started":"2024-12-30T20:56:15.147229Z","shell.execute_reply":"2024-12-30T21:04:09.909399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train_selected = X_train_processed.drop(columns = lower_cv_features)\nX_valid_selected = X_valid_processed.drop(columns = lower_cv_features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:04:09.912566Z","iopub.execute_input":"2024-12-30T21:04:09.912957Z","iopub.status.idle":"2024-12-30T21:04:09.966881Z","shell.execute_reply.started":"2024-12-30T21:04:09.912922Z","shell.execute_reply":"2024-12-30T21:04:09.965689Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Define Models\n### XGBoost","metadata":{}},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\nxgb_params_1 = {\n    'max_depth': 7, \n    'learning_rate': 0.048600089743471644, \n    'min_child_weight': 1.1685927782882253, \n    'colsample_bytree': 0.5288242720975447, \n    'reg_alpha': 8.062968886652984, \n    'reg_lambda': 0.09005692004244192, \n    'subsample': 0.9618549980615656, \n    'n_estimators': 476,\n    'objective': 'reg:squaredlogerror',\n    'booster': 'gbtree',\n    'eval_metric': 'rmse',\n    'random_state': 0\n}\n\nxgb_1 = XGBRegressor(**xgb_params_1)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    xgb_1.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              verbose=False)\n    y_pred_xgb = xgb_1.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_xgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nxgb_oof_score = np.mean(scores)  \nxgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the XGB model is {xgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the XGB model is {xgb_std}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:10:03.917073Z","iopub.execute_input":"2024-12-30T21:10:03.917523Z","iopub.status.idle":"2024-12-30T21:14:03.698985Z","shell.execute_reply.started":"2024-12-30T21:10:03.917471Z","shell.execute_reply":"2024-12-30T21:14:03.698032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importances = xgb_1.feature_importances_\nprint(feature_importances)\n# Sort feature importances and corresponding feature names\nsorted_indices = feature_importances.argsort()[::-1]\nsorted_features = X_train_selected.columns[sorted_indices]\nsorted_importances = feature_importances[sorted_indices]\nplt.figure(figsize=(10, 5))\nplt.bar(range(len(sorted_importances)), sorted_importances, tick_label=sorted_features)\nplt.title(f\"Feature Importance for XGBoost\")\nplt.xlabel(\"Features\")\nplt.ylabel(\"Importance\")\nplt.xticks(range(len(X_train_selected.columns)), X_train_selected.columns, rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:14:03.701026Z","iopub.execute_input":"2024-12-30T21:14:03.701456Z","iopub.status.idle":"2024-12-30T21:14:04.014939Z","shell.execute_reply.started":"2024-12-30T21:14:03.701409Z","shell.execute_reply":"2024-12-30T21:14:04.013807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\nxgb_params_2 = {\n        'max_depth': 16, \n        'learning_rate': 0.06727751958634316, \n        'min_child_weight': 1.0408910314363977, \n        'colsample_bytree': 0.7262863552374103, \n        'reg_alpha': 3.731490078787427, \n        'reg_lambda': 9.998531024912493, \n        'subsample': 0.6427242018265176, \n        'n_estimators': 689,\n        'objective': 'reg:squaredlogerror',\n        'booster': 'gbtree',\n        'eval_metric': 'rmse',\n        'random_state': 0\n}\n\nxgb_2 = XGBRegressor(**xgb_params_2)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    xgb_2.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              verbose=False)\n    y_pred_xgb = xgb_2.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_xgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nxgb_oof_score = np.mean(scores)  \nxgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the XGB model is {xgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the XGB model is {xgb_std}\") \n\nfeature_importances = xgb_2.feature_importances_\nprint(feature_importances)\n# Sort feature importances and corresponding feature names\nsorted_indices = feature_importances.argsort()[::-1]\nsorted_features = X_train_selected.columns[sorted_indices]\nsorted_importances = feature_importances[sorted_indices]\nplt.figure(figsize=(10, 5))\nplt.bar(range(len(sorted_importances)), sorted_importances, tick_label=sorted_features)\nplt.title(f\"Feature Importance for XGBoost\")\nplt.xlabel(\"Features\")\nplt.ylabel(\"Importance\")\nplt.xticks(range(len(X_train_selected.columns)), X_train_selected.columns, rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:14:04.016865Z","iopub.execute_input":"2024-12-30T21:14:04.017296Z","iopub.status.idle":"2024-12-30T21:17:13.327992Z","shell.execute_reply.started":"2024-12-30T21:14:04.017246Z","shell.execute_reply":"2024-12-30T21:17:13.326899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True)\n\n# Initialize model\nxgb_params_3 = {\n        'booster': 'gbtree',\n        'max_depth': 10, \n        'min_child_weight': 19,\n        'subsample': 0.8882395046974123,\n        'reg_alpha': 0.12361226000167053,\n        'reg_lambda': 0.7568174249367466,\n        'colsample_bytree': 0.83845902004046,\n        'objective': 'reg:squarederror',\n        'n_jobs': -1,\n        'learning_rate': 0.01,\n        'tree_method': 'hist',\n        'n_estimators': 3000,\n        'enable_categorical': True\n}\n\nxgb_3 = XGBRegressor(**xgb_params_3)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    xgb_3.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              verbose=False)\n    y_pred_xgb = xgb_3.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_xgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nxgb_oof_score = np.mean(scores)  \nxgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the XGB model is {xgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the XGB model is {xgb_std}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:17:13.330949Z","iopub.execute_input":"2024-12-30T21:17:13.331396Z","iopub.status.idle":"2024-12-30T21:23:03.303948Z","shell.execute_reply.started":"2024-12-30T21:17:13.331347Z","shell.execute_reply":"2024-12-30T21:23:03.302766Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\nxgb_params_4 = {\n        'booster': 'gbtree',\n        'max_depth': 10,\n        'min_child_weight': 11,\n        'subsample': 0.8709310286076404, \n        'reg_alpha': 0.2429313136646974,\n        'reg_lambda': 0.9443233420006127,\n        'colsample_bytree': 0.8004450303825076,\n        'random_state': 42, \n        'objective': 'reg:squarederror', \n        'n_jobs': -1,\n        'learning_rate': 0.01,\n        'tree_method': 'hist',\n        'n_estimators': 3000,\n        'random_state': 42,\n        'enable_categorical': True,\n        'verbosity': 0,\n        'eval_metric': 'rmse',\n        'device': \"cuda\"\n}\n\nxgb_4 = XGBRegressor(**xgb_params_4)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    xgb_4.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              verbose=False)\n    y_pred_xgb = xgb_4.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_xgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nxgb_oof_score = np.mean(scores)  \nxgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the XGB model is {xgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the XGB model is {xgb_std}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:23:03.305114Z","iopub.execute_input":"2024-12-30T21:23:03.305536Z","iopub.status.idle":"2024-12-30T21:28:55.703641Z","shell.execute_reply.started":"2024-12-30T21:23:03.305489Z","shell.execute_reply":"2024-12-30T21:28:55.702559Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\nxgb_params_5 = {\n        'booster': 'gbtree', \n        'max_depth': 9,\n        'min_child_weight': 13,\n        'subsample': 0.8950190330733485, \n        'reg_alpha': 0.3530275785494063,\n        'reg_lambda': 0.9564526039891854, \n        'colsample_bytree': 0.8488129245455193,\n        'random_state': 42,\n        'objective': 'reg:squarederror',\n        'n_jobs': -1,\n        'learning_rate': 0.01,\n        'tree_method': 'hist',\n        'n_estimators': 3000,\n        'objective': 'reg:squarederror',\n        'random_state': 42,\n        'enable_categorical': True,\n        'verbosity': 0,\n        'eval_metric': 'rmse'\n}\n\nxgb_5 = XGBRegressor(**xgb_params_5)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    xgb_5.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              verbose=False)\n    y_pred_xgb = xgb_5.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_xgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nxgb_oof_score = np.mean(scores)  \nxgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the XGB model is {xgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the XGB model is {xgb_std}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:28:55.705164Z","iopub.execute_input":"2024-12-30T21:28:55.705591Z","iopub.status.idle":"2024-12-30T21:35:29.903398Z","shell.execute_reply.started":"2024-12-30T21:28:55.705538Z","shell.execute_reply":"2024-12-30T21:35:29.902409Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### LightGBM","metadata":{}},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\nlgb_params_1 = {\n    'max_depth': 11, \n    'num_leaves': 28, \n    'learning_rate': 0.07647055187712827, \n    'feature_fraction': 0.9845340232901648, \n    'bagging_fraction': 0.9751777653062266, \n    'bagging_freq': 4, \n    'reg_alpha': 1.9794504987659611, \n    'reg_lambda': 1.5074268152177714, \n    'n_estimators': 904,\n    'verbosity':-1,\n    'boosting_type': 'gbdt',\n    'objective': 'regression',\n    'metric': 'rmse'}\n\nlgb_1 = LGBMRegressor(**lgb_params_1)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    lgb_1.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)])\n    y_pred_lgb = lgb_1.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_lgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nlgb_oof_score = np.mean(scores)  \nlgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the LightGBM model is {lgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the LightGBM model is {lgb_std}\") ","metadata":{"trusted":true,"scrolled":true,"execution":{"iopub.status.busy":"2024-12-30T21:35:29.904823Z","iopub.execute_input":"2024-12-30T21:35:29.905242Z","iopub.status.idle":"2024-12-30T21:41:24.177877Z","shell.execute_reply.started":"2024-12-30T21:35:29.905195Z","shell.execute_reply":"2024-12-30T21:41:24.176661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importances = lgb_1.feature_importances_\nprint(feature_importances)\n# Sort feature importances and corresponding feature names\nsorted_indices = feature_importances.argsort()[::-1]\nsorted_features = X_train_selected.columns[sorted_indices]\nsorted_importances = feature_importances[sorted_indices]\nplt.figure(figsize=(10, 5))\nplt.bar(range(len(sorted_importances)), sorted_importances, tick_label=sorted_features)\nplt.title(f\"Feature Importance for LightGBM\")\nplt.xlabel(\"Features\")\nplt.ylabel(\"Importance\")\nplt.xticks(range(len(X_train_selected.columns)), X_train_selected.columns, rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:41:24.179375Z","iopub.execute_input":"2024-12-30T21:41:24.179817Z","iopub.status.idle":"2024-12-30T21:41:24.54882Z","shell.execute_reply.started":"2024-12-30T21:41:24.179782Z","shell.execute_reply":"2024-12-30T21:41:24.547685Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\nlgb_params_2 = {\n        'max_depth': 10, \n        'num_leaves': 58, \n        'learning_rate': 0.08579016254024871, \n        'feature_fraction': 0.734968274884088, \n        'bagging_fraction': 0.8428557428919629, \n        'bagging_freq': 3, \n        'reg_alpha': 3.1802796944775924, \n        'reg_lambda': 2.0434363964173574, \n        'n_estimators': 797,\n        'verbosity':-1,\n        'boosting_type': 'gbdt',\n        'objective': 'regression',\n        'metric': 'rmse'\n}\n\nlgb_2 = LGBMRegressor(**lgb_params_2)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    lgb_2.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)])\n    y_pred_lgb = lgb_2.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_lgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nlgb_oof_score = np.mean(scores)  \nlgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the LightGBM model is {lgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the LightGBM model is {lgb_std}\") \n\nfeature_importances = lgb_2.feature_importances_\nprint(feature_importances)\n# Sort feature importances and corresponding feature names\nsorted_indices = feature_importances.argsort()[::-1]\nsorted_features = X_train_selected.columns[sorted_indices]\nsorted_importances = feature_importances[sorted_indices]\nplt.figure(figsize=(10, 5))\nplt.bar(range(len(sorted_importances)), sorted_importances, tick_label=sorted_features)\nplt.title(f\"Feature Importance for LightGBM\")\nplt.xlabel(\"Features\")\nplt.ylabel(\"Importance\")\nplt.xticks(range(len(X_train_selected.columns)), X_train_selected.columns, rotation=90)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:41:24.550253Z","iopub.execute_input":"2024-12-30T21:41:24.550658Z","iopub.status.idle":"2024-12-30T21:47:29.950511Z","shell.execute_reply.started":"2024-12-30T21:41:24.550596Z","shell.execute_reply":"2024-12-30T21:47:29.949647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\nlgb_params_3 = {\n        'objective': 'regression_l2',\n        'metric': 'rmse',\n        'max_depth': 11,\n        'num_leaves': 169, \n        'min_child_samples': 27,\n        'min_child_weight': 13,\n        'colsample_bytree': 0.4844468439076937,\n        'reg_alpha': 0.08086875774692211, \n        'reg_lambda': 0.9676995833820066,\n        'random_state': 42,\n        'verbose': -1,\n        'n_estimators': 5000,\n        'learning_rate': 0.01\n}\n\nlgb_3 = LGBMRegressor(**lgb_params_3)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    lgb_3.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)])\n    y_pred_lgb = lgb_3.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_lgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nlgb_oof_score = np.mean(scores)  \nlgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the LightGBM model is {lgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the LightGBM model is {lgb_std}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T21:47:29.953628Z","iopub.execute_input":"2024-12-30T21:47:29.953932Z","iopub.status.idle":"2024-12-30T22:24:51.750509Z","shell.execute_reply.started":"2024-12-30T21:47:29.953903Z","shell.execute_reply":"2024-12-30T22:24:51.749054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\nlgb_params_4 = {\n        'objective': 'regression_l2',\n        'metric': 'rmse',\n        'max_depth': 12,\n        'num_leaves': 246,\n        'min_child_samples': 10,\n        'min_child_weight': 24,\n        'colsample_bytree': 0.49704729723832064,\n        'reg_alpha': 0.2791624804454722,\n        'reg_lambda': 0.7306747260034717,\n        'random_state': 42,\n        'verbose': -1,\n        'n_estimators': 5000,\n        'learning_rate': 0.01\n}\n\nlgb_4 = LGBMRegressor(**lgb_params_4)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    lgb_4.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)])\n    y_pred_lgb = lgb_4.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_lgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nlgb_oof_score = np.mean(scores)  \nlgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the LightGBM model is {lgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the LightGBM model is {lgb_std}\") ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-30T22:24:51.752069Z","iopub.execute_input":"2024-12-30T22:24:51.752429Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\nlgb_params_5 = {\n        'random_state': 42,\n        'verbose': -1,\n        'boosting_type': 'gbdt',\n        'n_estimators': 3000,\n        'eval_metric': 'rmse',\n        'objective': 'regression_l2',\n        'learning_rate': 0.01,\n        'max_bin': 8000,\n        'max_depth': 8, \n        'num_leaves': 796,\n        'min_child_samples': 21,\n        'min_child_weight': 11,\n        'colsample_bytree': 0.48551477424501904, \n        'reg_alpha': 0.44307695050069845, \n        'reg_lambda': 0.9310703189038081\n}\n\nlgb_5 = LGBMRegressor(**lgb_params_5)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    lgb_5.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)])\n    y_pred_lgb = lgb_5.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_lgb)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\nlgb_oof_score = np.mean(scores)  \nlgb_std = np.std(scores)\nprint(f\"The 5-fold average oof RMSLE score of the LightGBM model is {lgb_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the LightGBM model is {lgb_std}\") ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### CatBoost","metadata":{}},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\ncat_params_1 = {\n    'iterations': 971, \n    'depth': 9, \n    'learning_rate': 0.209173094311765, \n    'random_strength': 3.7025175269170347, \n    'bagging_temperature': 0.07181853264492258, \n    'border_count': 209, \n    'l2_leaf_reg': 31.926710123760373,         \n    'eval_metric': 'RMSE',\n    'random_state': 0,\n    'verbose': 0}\n\ncat_1 = CatBoostRegressor(**cat_params_1)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    cat_1.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              cat_features=None,\n              verbose=False)\n    y_pred_cat = cat_1.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_cat)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\ncat_oof_score = np.mean(scores)  \ncat_std = np.std(scores)\n\nprint(f\"The 5-fold average oof RMSLE score of the CatBoost model is {cat_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the CatBoost model is {cat_std}\") ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_importances = cat_1.get_feature_importance()\nprint(feature_importances)\n# Sort feature importances and corresponding feature names\nsorted_indices = feature_importances.argsort()[::-1]\nsorted_features = X_train_selected.columns[sorted_indices]\nsorted_importances = feature_importances[sorted_indices]\nplt.figure(figsize=(10, 5))\nplt.bar(range(len(sorted_importances)), sorted_importances, tick_label=sorted_features)\nplt.title(f\"Feature Importance for CatBoost\")\nplt.xlabel(\"Features\")\nplt.ylabel(\"Importance\")\nplt.xticks(range(len(X_train_selected.columns)), X_train_selected.columns, rotation=90)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\ncat_params_2 = {\n        'verbose': 0,\n        'random_state': 42,\n        'eval_metric': \"RMSE\",\n        'objective': 'RMSE', \n        'depth': 13, \n        'subsample': 0.9998449533801151,\n        'min_data_in_leaf': 63,\n        'l2_leaf_reg': 0.3242106539167982,\n        \"iterations\" : 1000,\n        'learning_rate': 0.1\n}\n\ncat_2 = CatBoostRegressor(**cat_params_2)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    cat_2.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              cat_features=None,\n              verbose=False)\n    y_pred_cat = cat_2.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_cat)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\ncat_oof_score = np.mean(scores)  \ncat_std = np.std(scores)\n\nprint(f\"The 5-fold average oof RMSLE score of the CatBoost model is {cat_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the CatBoost model is {cat_std}\") ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\ncat_params_3 = {\n        'objective': 'RMSE',\n        'depth': 11,\n        'subsample': 0.9581878706329088,\n        'min_data_in_leaf': 43,\n        'l2_leaf_reg': 0.08249502036269224,\n        'verbose': 0,\n        'random_state': 42,\n        'eval_metric': \"RMSE\",\n        \"iterations\" : 1000,\n        'learning_rate': 0.05,\n        'bootstrap_type': 'Bernoulli'\n}\n\ncat_3 = CatBoostRegressor(**cat_params_3)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    cat_3.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              cat_features=None,\n              verbose=False)\n    y_pred_cat = cat_3.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_cat)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\ncat_oof_score = np.mean(scores)  \ncat_std = np.std(scores)\n\nprint(f\"The 5-fold average oof RMSLE score of the CatBoost model is {cat_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the CatBoost model is {cat_std}\") ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True)\n\n# Initialize model\ncat_params_4 = {\n        'objective': 'RMSE',\n        'depth': 11, \n        'min_data_in_leaf': 51, \n        'l2_leaf_reg': 7.51216069323258,\n        'random_state': 42,\n        'early_stopping_rounds': 200,\n        'eval_metric': \"RMSE\",\n        \"iterations\" : 1000,\n        'learning_rate': 0.01,\n        'bootstrap_type': 'Bernoulli'\n}\n\ncat_4 = CatBoostRegressor(**cat_params_4)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    cat_4.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              cat_features=None,\n              verbose=False)\n    y_pred_cat = cat_4.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_cat)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\ncat_oof_score = np.mean(scores)  \ncat_std = np.std(scores)\n\nprint(f\"The 5-fold average oof RMSLE score of the CatBoost model is {cat_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the CatBoost model is {cat_std}\") ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize KFold\nkf = KFold(n_splits=5, shuffle=True,random_state=42)\n\n# Initialize model\ncat_params_5 = {\n        'objective': 'RMSE',\n        'logging_level': 'Silent',\n        'depth': 11,\n        'min_data_in_leaf': 62,\n        'l2_leaf_reg': 7.887761285633826,\n        'random_state': 42,\n        'eval_metric': \"RMSE\",\n        \"iterations\" : 1000,\n        'learning_rate': 0.01\n}\n\ncat_5 = CatBoostRegressor(**cat_params_5)\nscores = []\n\n# Loop through each fold\nfor i, (train_idx, val_idx) in enumerate(kf.split(X_train_selected)):\n    print(f\"------------ Working on Fold {i} ------------\")\n    \n    X_train_split, X_valid_split = X_train_selected.iloc[train_idx], X_train_selected.iloc[val_idx]\n    y_train_split, y_valid_split = y_train.iloc[train_idx], y_train.iloc[val_idx]\n    \n    # Fit model\n    cat_5.fit(X_train_split,y_train_split,\n              eval_set=[(X_valid_split, y_valid_split)],\n              early_stopping_rounds=50,\n              cat_features=None,\n              verbose=False)\n    y_pred_cat = cat_5.predict(X_valid_split)\n    score = rmsle(y_valid_split, y_pred_cat)\n    print(f\"The oof RMSLE score is {score}\")\n    scores.append(score)\n\ncat_oof_score = np.mean(scores)  \ncat_std = np.std(scores)\n\nprint(f\"The 5-fold average oof RMSLE score of the CatBoost model is {cat_oof_score}\")\nprint(f\"The 5-fold std oof RMSLE score of the CatBoost model is {cat_std}\") ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Define Regressor class\n# class Regressor:\n#     def __init__(self, eval_metric):\n#         self.eval_metric = eval_metric\n#         self.models = self._initialize_models()\n\n#     def _initialize_models(self):\n#         # Define model parameters\n#         xgb_params_1 = {\n#             'max_depth': 7, \n#             'learning_rate': 0.08892164153650667, \n#             'min_child_weight': 1.000812173349267, \n#             'colsample_bytree': 0.6037524966913294, \n#             'reg_alpha': 1.587471515505295, \n#             'reg_lambda': 5.340290532777649, \n#             'subsample': 0.583899963777622, \n#             'n_estimators': 494,\n#             'objective': 'reg:squaredlogerror',\n#             'booster': 'gbtree',\n#             'eval_metric': 'rmse',\n#             'random_state': 0}\n#         xgb_params_2 = {\n#             'max_depth': 7, \n#             'learning_rate': 0.048600089743471644, \n#             'min_child_weight': 1.1685927782882253, \n#             'colsample_bytree': 0.5288242720975447, \n#             'reg_alpha': 8.062968886652984, \n#             'reg_lambda': 0.09005692004244192, \n#             'subsample': 0.9618549980615656, \n#             'n_estimators': 476,\n#             'objective': 'reg:squaredlogerror',\n#             'booster': 'gbtree',\n#             'eval_metric': 'rmse',\n#             'random_state': 0\n#         }\n#         lgb_params_1 = {\n#             'max_depth': 10, \n#             'num_leaves': 58, \n#             'learning_rate': 0.08579016254024871, \n#             'feature_fraction': 0.734968274884088, \n#             'bagging_fraction': 0.8428557428919629, \n#             'bagging_freq': 3, \n#             'reg_alpha': 3.1802796944775924, \n#             'reg_lambda': 2.0434363964173574, \n#             'n_estimators': 797,\n#             'boosting_type': 'gbdt',\n#             'objective': 'regression',\n#             'metric': 'rmse'\n#         }\n#         lgb_params_2 = {\n#             'max_depth': 11, \n#             'num_leaves': 28, \n#             'learning_rate': 0.07647055187712827, \n#             'feature_fraction': 0.9845340232901648, \n#             'bagging_fraction': 0.9751777653062266, \n#             'bagging_freq': 4, \n#             'reg_alpha': 1.9794504987659611, \n#             'reg_lambda': 1.5074268152177714, \n#             'n_estimators': 904,\n#             'boosting_type': 'gbdt',\n#             'objective': 'regression',\n#             'metric': 'rmse'}\n#         cat_params_1 = {\n#             'iterations': 971, \n#             'depth': 9, \n#             'learning_rate': 0.209173094311765, \n#             'random_strength': 3.7025175269170347, \n#             'bagging_temperature': 0.07181853264492258, \n#             'border_count': 209, \n#             'l2_leaf_reg': 31.926710123760373,         \n#             'eval_metric': 'RMSE',\n#             'random_state': 0,\n#             'verbose': 0}\n#         # Define models\n#         return {\n#             'xgb_1': XGBRegressor(**xgb_params_1),\n#             'xgb_2': XGBRegressor(**xgb_params_2),\n#             'lgb_1': LGBMRegressor(**lgb_params_1),\n#             'lgb_2': LGBMRegressor(**lgb_params_2),\n#             'cat_1': CatBoostRegressor(**cat_params_1)\n#         }\n#     # Fit model with cross-validation\n#     def fit(self, X, y):\n#         results = []\n#         # Split the data\n#         X_train_split, X_valid_split, y_train_split, y_valid_split = train_test_split(X, y, test_size=0.2, random_state=42)\n        \n#         for name, model in self.models.items():\n#             if ('xgb' in name) or ('lgb' in name) or ('cat' in name):\n#                 if 'lgb' in name:\n#                     model.fit(X_train_split, y_train_split, eval_set=[(X_valid_split, y_valid_split)])\n#                 elif 'cat' in name:\n#                     model.fit(\n#                         X_train_split, y_train_split, \n#                         eval_set=[(X_valid_split, y_valid_split)],\n#                         early_stopping_rounds=50,\n#                         cat_features=None,\n#                         verbose=False\n#                     )  \n#                 else:\n#                     model.fit(\n#                         X_train_split, y_train_split, \n#                         eval_set=[(X_valid_split, y_valid_split)],\n#                         early_stopping_rounds=50,\n#                         verbose=False\n#                     )\n            \n#             # Predict and calculate metric\n#             y_pred = model.predict(X_valid_split)\n#             metric_value = self.eval_metric(y_valid_split, y_pred)\n            \n#             # Save the result\n#             results.append({\n#                 \"Model\": name,\n#                 \"Metric\": metric_value,\n#                 \"Predictions\": y_pred\n#             })\n            \n#             # Plot feature importance if available\n#             self._plot_feature_importance(name, model, X)\n#         return results\n        \n#     def _plot_feature_importance(self, name, model, X):\n#         # Check if model supports feature importance\n#         if hasattr(model, 'feature_importances_'):\n#             feature_importances = model.feature_importances_\n#             print(feature_importances)\n#             # Sort feature importances and corresponding feature names\n#             sorted_indices = feature_importances.argsort()[::-1]\n#             sorted_features = X.columns[sorted_indices]\n#             sorted_importances = feature_importances[sorted_indices]\n#             plt.figure(figsize=(10, 5))\n#             plt.bar(range(len(sorted_importances)), sorted_importances, tick_label=sorted_features)\n#             plt.title(f\"Feature Importance for {name}\")\n#             plt.xlabel(\"Features\")\n#             plt.ylabel(\"Importance\")\n#             plt.xticks(range(len(X.columns)), X.columns, rotation=90)\n#             plt.show()\n#         elif hasattr(model, 'get_feature_importance'):  # For CatBoost\n#             feature_importances = model.get_feature_importance()\n#             print(feature_importances)\n#             # Sort feature importances and corresponding feature names\n#             sorted_indices = feature_importances.argsort()[::-1]\n#             sorted_features = X.columns[sorted_indices]\n#             sorted_importances = feature_importances[sorted_indices]\n#             plt.figure(figsize=(10, 5))\n#             plt.bar(range(len(sorted_importances)), sorted_importances, tick_label=sorted_features)\n#             plt.title(f\"Feature Importance for {name}\")\n#             plt.xlabel(\"Features\")\n#             plt.ylabel(\"Importance\")\n#             plt.xticks(range(len(X.columns)), X.columns, rotation=90)\n#             plt.show()\n#         else:\n#             print(f\"{name} does not support feature importance.\")\n    \n#     # Predict and evaluate on valid split\n#     def validate(self, X, y):\n#         validations = []\n#         for name, model in self.models.items():\n#             # Predict and calculate metric\n#             y_pred = model.predict(X)\n#             metric_value = self.eval_metric(y, y_pred)\n\n#             # Save the result\n#             validations.append({\n#                 \"Model\": name,\n#                 \"Metric\": metric_value,\n#                 \"Predictions\": y_pred\n#             })\n#             # feature importance\n#             print(model.feature_importances_)\n#             # plot\n#             pyplot.bar(range(len(model.feature_importances_)), model.feature_importances_)\n#             pyplot.show()\n#         return validations","metadata":{"trusted":true,"_kg_hide-output":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Initialize evaluator\n# model = Regressor(eval_metric=rmsle)\n\n# # Fit and evaluate models\n# results = model.fit(X_train_processed, y_train)\n\n# # Display results\n# for result in results:\n#     print(f\"Model: {result['Model']}, RMSLE: {result['Metric']}\")","metadata":{"trusted":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LightAutoML","metadata":{}},{"cell_type":"code","source":"# def map_class(x, task, reader):\n#     if task.name == 'multiclass':\n#         return reader[x]\n#     else:\n#         return x\n\n# mapped = np.vectorize(map_class)\n\n# def score(task, y_true, y_pred):\n#     if task.name == 'binary':\n#         return roc_auc_score(y_true, y_pred)\n#     elif task.name == 'multiclass':\n#         return accuracy_score(y_true, np.argmax(y_pred, 1))\n#     elif task.name == 'reg' or task.name == 'multi:reg':\n#         return median_absolute_error(y_true, y_pred)\n#     else:\n#         raise 'Task is not correct.'\n        \n# def take_pred_from_task(pred, task):\n#     if task.name == 'binary' or task.name == 'reg':\n#         return pred[:, 0]\n#     elif task.name == 'multiclass' or task.name == 'multi:reg':\n#         return pred\n#     else:\n#         raise 'Task is not correct.'\n        \n# def use_plr(USE_PLR):\n#     if USE_PLR:\n#         return \"plr\"\n#     else:\n#         return \"cont\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# RANDOM_STATE = 42\n# N_THREADS = os.cpu_count()\n# TIMEOUT = 9 * 3600\n# N_FOLDS = 10\n# np.random.seed(RANDOM_STATE)\n# torch.set_num_threads(N_THREADS)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# task = Task('reg') \n# automl = TabularAutoML(\n#     task = task, \n#     timeout = 9 * 3600,\n#     cpu_limit = os.cpu_count(),\n#     nn_params = {\n#     'stop_by_metric': True,\n#     'verbose_bar': True},\n#     nn_pipeline_params = {\"use_qnt\": False, \"use_te\": False},\n#     reader_params = {'n_jobs': N_THREADS, 'cv': N_FOLDS, 'random_state': RANDOM_STATE, 'advanced_roles': True}\n# )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# task = Task('reg') \n# automl = TabularAutoML(\n#     task = task,\n#     timeout = TIMEOUT,\n#     cpu_limit = N_THREADS,\n#     general_params = {\"use_algos\": [[\"nn\"]]}, # ['nn', 'mlp', 'dense', 'denselight', 'resnet', 'snn', 'node', 'autoint', 'fttransformer'] or custom torch model\n#     nn_params = {\"n_epochs\": 10, \"bs\": 512, \"num_workers\": 0, \"path_to_save\": None, \"freeze_defaults\": True},\n#     nn_pipeline_params = {\"use_qnt\": True, \"use_te\": False},\n#     reader_params = {'n_jobs': N_THREADS, 'cv': N_FOLDS, 'random_state': RANDOM_STATE}\n# )","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# out_of_fold_predictions = automl.fit_predict(\n#     df_train_automl,\n#     roles = {\n#         'target': 'Premium Amount',\n#         'drop': 'id'\n#     }, \n#     verbose = 3\n# )","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluate","metadata":{}},{"cell_type":"code","source":"# # Validate LightAutoML model\n# rmsle(out_of_fold_predictions.data[:, 0], df_train_automl['Premium Amount'])","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_preds = {}\n\ny_preds['xgb_1'] = xgb_1.predict(X_valid_selected)\ny_preds['xgb_2'] = xgb_2.predict(X_valid_selected)\ny_preds['xgb_3'] = xgb_3.predict(X_valid_selected)\ny_preds['xgb_4'] = xgb_4.predict(X_valid_selected)\ny_preds['xgb_5'] = xgb_5.predict(X_valid_selected)\n\ny_preds['lgb_1'] = lgb_1.predict(X_valid_selected)\ny_preds['lgb_2'] = lgb_2.predict(X_valid_selected)\ny_preds['lgb_3'] = lgb_3.predict(X_valid_selected)\ny_preds['lgb_4'] = lgb_4.predict(X_valid_selected)\ny_preds['lgb_5'] = lgb_5.predict(X_valid_selected)\n\ny_preds['cat_1'] = cat_1.predict(X_valid_selected)\ny_preds['cat_2'] = cat_2.predict(X_valid_selected)\ny_preds['cat_3'] = cat_3.predict(X_valid_selected)\ny_preds['cat_4'] = cat_4.predict(X_valid_selected)\ny_preds['cat_5'] = cat_5.predict(X_valid_selected)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# rmsle = {}\n\n# rmsle['xgb_1'] = mean_squared_log_error(y_valid, y_preds['xgb_1'])\n# rmsle['xgb_2'] = mean_squared_log_error(y_valid, y_preds['xgb_2'])\n# rmsle['xgb_3'] = mean_squared_log_error(y_valid, y_preds['xgb_3'])\n# rmsle['xgb_4'] = mean_squared_log_error(y_valid, y_preds['xgb_4'])\n# rmsle['xgb_5'] = mean_squared_log_error(y_valid, y_preds['xgb_5'])\n\n# rmsle['lgb_1'] = mean_squared_log_error(y_valid, y_preds['lgb_1'])\n# rmsle['lgb_2'] = mean_squared_log_error(y_valid, y_preds['lgb_2'])\n# rmsle['lgb_3'] = mean_squared_log_error(y_valid, y_preds['lgb_3'])\n# rmsle['lgb_4'] = mean_squared_log_error(y_valid, y_preds['lgb_4'])\n# rmsle['lgb_5'] = mean_squared_log_error(y_valid, y_preds['lgb_5'])\n\n# rmsle['cat_1'] = mean_squared_log_error(y_valid, y_preds['cat_1'])\n# rmsle['cat_2'] = mean_squared_log_error(y_valid, y_preds['cat_2'])\n# rmsle['cat_3'] = mean_squared_log_error(y_valid, y_preds['cat_3'])\n# rmsle['cat_4'] = mean_squared_log_error(y_valid, y_preds['cat_4'])\n# rmsle['cat_5'] = mean_squared_log_error(y_valid, y_preds['cat_5'])\n\n# for model, score in rmsle.items():\n#     print(f\"The RMSLE score of the {model} is {score}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Sort the dictionary by keys\n# sorted_rmsle = dict(sorted(rmsle.items()))\n\n# # Create lists of keys and values for plotting\n# models = list(sorted_rmsle.keys())\n# scores = list(sorted_rmsle.values())\n\n# # Create a bar chart\n# plt.figure(figsize=(10,6))\n# hbars = plt.bar(models, scores, width = 0.7)\n# plt.bar_label(hbars, fmt='%.4f')\n# plt.xlabel('Models')\n# plt.ylabel('RMSLE')\n# plt.title('Model Evaluation')          \n# plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Ensemble","metadata":{}},{"cell_type":"markdown","source":"## Stacking ","metadata":{}},{"cell_type":"code","source":"# from sklearn.ensemble import StackingRegressor\n# from sklearn.ensemble import HistGradientBoostingRegressor\n# meta_model = HistGradientBoostingRegressor(max_depth = 3)\n\n# base_models = [\n#     ('xgb_1', xgb_1),\n#     ('lgb_1', lgb_1),\n#     ('cat_1', cat_1)\n# ]\n\n# stacking_regressor = StackingRegressor(\n#     estimators=base_models,\n#     final_estimator=meta_model,\n#     cv=5 \n# )\n# stacking_regressor.fit(X_train_selected, y_train)","metadata":{"trusted":true,"scrolled":true,"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# scores = cross_val_score(stacking_regressor, X_train_selected, y_train, cv=5, scoring=RMSLE)\n\n# print(\"Cross-Validation RMSLE:\", -scores.mean())\n\n# y_pred_stack = stacking_regressor.predict(X_valid_selected)\n\n# # Evaluate final ensemble\n# stack_rmsle = rmsle(y_valid, y_pred_stack)\n# print(f\"Stacking Ensemble RMSLE: {stack_rmsle}\")","metadata":{"trusted":true,"_kg_hide-output":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Blending","metadata":{}},{"cell_type":"code","source":"w_xgb_1 = 0.1\nw_xgb_2 = 0.1\nw_xgb_3 = 0.1\nw_xgb_4 = 0.1\nw_xgb_5 = 0.1\n\nw_lgb_1 = 0.06\nw_lgb_2 = 0.06\nw_lgb_3 = 0.06\nw_lgb_4 = 0.06\nw_lgb_5 = 0.06\n\nw_cat_1 = 0.04\nw_cat_2 = 0.04\nw_cat_3 = 0.04\nw_cat_4 = 0.04\nw_cat_5 = 0.04\n\ny_pred_ensemble = (\n    w_xgb_1 * y_preds['xgb_1'] +\n    w_xgb_2 * y_preds['xgb_2'] +\n    w_xgb_3 * y_preds['xgb_3'] +\n    w_xgb_4 * y_preds['xgb_4'] +\n    w_xgb_5 * y_preds['xgb_5'] +\n    w_lgb_1 * y_preds['lgb_1'] +\n    w_lgb_2 * y_preds['lgb_2'] +\n    w_lgb_3 * y_preds['lgb_3'] +\n    w_lgb_4 * y_preds['lgb_4'] +\n    w_lgb_5 * y_preds['lgb_5'] +\n    w_cat_1 * y_preds['cat_1'] +\n    w_cat_2 * y_preds['cat_2'] +\n    w_cat_3 * y_preds['cat_3'] +\n    w_cat_4 * y_preds['cat_4'] +\n    w_cat_5 * y_preds['cat_5']\n)\n\n# Calculate RMSLE for the ensemble\nblend_rmsle = rmsle(y_valid, y_pred_ensemble)\n# blend_rmsle = rmsle(y_valid, np.expm1(y_pred_ensemble))\nprint(f\"Blending Ensemble RMSLE: {blend_rmsle}\")","metadata":{"trusted":true,"scrolled":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"X = extract_date_components(X, 'Policy Start Date')\nX = log_transform(X)\nX_processed = preprocessor.fit_transform(X, y_log)\nX_selected = X_processed.drop(columns = lower_cv_features)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_processed = preprocessor.transform(df_test)\ndf_test_selected = df_test_processed.drop(columns = lower_cv_features)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# if stack_rmsle < blend_rmsle:\n#     # Stacking\n#     y_pred_test_ensemble = stacking_regressor.predict(df_test_selected)\n# else:\n#     # Blending\ntest_preds = {}\ntest_preds['xgb_1'] = xgb_1.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['xgb_2'] = xgb_2.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['xgb_3'] = xgb_3.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['xgb_4'] = xgb_4.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['xgb_5'] = xgb_5.fit(X_selected, y_log).predict(df_test_selected)\n\ntest_preds['lgb_1'] = lgb_1.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['lgb_2'] = lgb_2.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['lgb_3'] = lgb_3.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['lgb_4'] = lgb_4.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['lgb_5'] = lgb_5.fit(X_selected, y_log).predict(df_test_selected)\n\ntest_preds['cat_1'] = cat_1.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['cat_2'] = cat_2.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['cat_3'] = cat_3.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['cat_4'] = cat_4.fit(X_selected, y_log).predict(df_test_selected)\ntest_preds['cat_5'] = cat_5.fit(X_selected, y_log).predict(df_test_selected)\n\n# Combine predictions with weights\ny_log_pred_test_ensemble = (\n    w_xgb_1 * test_preds['xgb_1'] +\n    w_xgb_2 * test_preds['xgb_2'] +\n    w_xgb_3 * test_preds['xgb_3'] +\n    w_xgb_4 * test_preds['xgb_4'] +\n    w_xgb_5 * test_preds['xgb_5'] +\n    w_lgb_1 * test_preds['lgb_1'] +\n    w_lgb_2 * test_preds['lgb_2'] +\n    w_lgb_3 * test_preds['lgb_3'] +\n    w_lgb_4 * test_preds['lgb_4'] +\n    w_lgb_5 * test_preds['lgb_5'] +\n    w_cat_1 * test_preds['cat_1'] +\n    w_cat_2 * test_preds['cat_2'] +\n    w_cat_3 * test_preds['cat_3'] +\n    w_cat_4 * test_preds['cat_4'] +\n    w_cat_5 * test_preds['cat_5']\n)\n\n# Transform back to the original scale\ny_pred_test_ensemble = np.expm1(y_log_pred_test_ensemble)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create submission file\ndf_sub = pd.read_csv('/kaggle/input/playground-series-s4e12/sample_submission.csv')\ndf_sub['Premium Amount'] = y_pred_test_ensemble\ndf_sub.to_csv('submission_ensemble.csv', index=False)\ndf_sub.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Appendix","metadata":{}},{"cell_type":"markdown","source":"## XGB Optuna","metadata":{}},{"cell_type":"code","source":"# def objective(trial):\n#     # Define hyperparameter search space\n#     params = {\n#         'booster': 'gbtree',\n#         'objective': 'reg:squaredlogerror',  # For RMSLE\n#         'eval_metric': 'rmse',              # Metric for evaluation\n#         'max_depth': trial.suggest_int('max_depth', 3, 20),\n#         'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.1),\n#         'min_child_weight': trial.suggest_float('min_child_weight', 1, 15),\n#         'colsample_bytree': trial.suggest_float('colsample_bytree', 0.5, 1.0),\n#         'reg_alpha': trial.suggest_float('reg_alpha', 0.01, 10.0),\n#         'reg_lambda': trial.suggest_float('reg_lambda', 0.01, 10.0),\n#         'subsample': trial.suggest_float('subsample', 0.5, 1.0),\n#         'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n#         'device': 'cuda',  # Use GPU if available\n#         'random_state': 0\n#     }\n\n#     # Split data for validation\n#     X_train_split, X_valid_split, y_train_split, y_valid_split = train_test_split(\n#         X_train_processed, y_train, test_size=0.2, random_state=42)\n\n#     # Train XGBoost model\n#     model = XGBRegressor(**params)\n#     model.fit(\n#         X_train_split, y_train_split,\n#         eval_set=[(X_valid_split, y_valid_split)],\n#         early_stopping_rounds=50,\n#         verbose=False\n#     )\n\n#     # Predict on validation data\n#     y_pred = model.predict(X_valid_split)\n\n#     # Calculate RMSLE\n#     rmsle_score = rmsle(y_valid_split, y_pred)\n#     return rmsle_score","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# study = optuna.create_study(direction='minimize')  # Minimize RMSLE\n# study.optimize(objective, n_trials=50)\n\n# # Get the best parameters\n# xgb_best_params = study.best_params\n# print(\"Best Parameters:\", xgb_best_params)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Split the training data to include a validation set for early stopping\n# X_train_split, X_valid_split, y_train_split, y_valid_split = train_test_split(X_train, y_train, test_size=0.2, random_state=42)\n\n# # Create the XGBoost model using Optuna model\n# xgb_best_params_1 = {'max_depth': 4, \n#                     'learning_rate': 0.028949993238818562, \n#                     'min_child_weight': 7.472178342152908, \n#                     'colsample_bytree': 0.5048225534122298, \n#                     'reg_alpha': 8.993064644792046, \n#                     'reg_lambda': 3.4598638747943316, \n#                     'subsample': 0.5687531348526244, \n#                     'n_estimators': 750,\n#                     'objective': 'reg:squaredlogerror',\n#                     'booster': 'gbtree',\n#                     'eval_metric': 'rmse',\n#                     'device': 'cuda',\n#                     'random_state': 0}\n\n# xgb_1 = XGBRegressor(**xgb_best_params)\n\n# # Fit the model with early stopping\n# xgb_1.fit(X_train_split, y_train_split,\n#           eval_set=[(X_valid_split, y_valid_split)],\n#           early_stopping_rounds=50,\n#           verbose=500)\n\n# # Predict probabilities on validation data\n# y_pred_xgb_1 = xgb_1.predict(X_valid)\n\n# # Calculate Accuracy Score on validation data\n# xgb_rmsle = rmsle(y_valid, y_pred_xgb_1)\n# print(\"RMSLE on Validation Data:\", xgb_rmsle)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## LGBM Optuna","metadata":{}},{"cell_type":"code","source":"# # Define RMSLE evaluation metric\n# def rmsle(y_true, y_pred):\n#     return np.sqrt(np.mean((np.log1p(y_true) - np.log1p(y_pred)) ** 2))\n\n# # Define the objective function\n# def objective(trial):\n#     # Define the hyperparameter search space\n#     params = {\n#         'boosting_type': 'gbdt',\n#         'objective': 'regression',\n#         'metric': 'rmse',\n#         'max_depth': trial.suggest_int('max_depth', 3, 12),\n#         'num_leaves': trial.suggest_int('num_leaves', 20, 150),\n#         'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n#         'feature_fraction': trial.suggest_float('feature_fraction', 0.5, 1.0),\n#         'bagging_fraction': trial.suggest_float('bagging_fraction', 0.5, 1.0),\n#         'bagging_freq': trial.suggest_int('bagging_freq', 1, 10),\n#         'reg_alpha': trial.suggest_float('reg_alpha', 0.0, 10.0),\n#         'reg_lambda': trial.suggest_float('reg_lambda', 0.0, 10.0),\n#         'n_estimators': trial.suggest_int('n_estimators', 100, 1000),\n#         'random_state': 0,\n#         'verbose': -1\n#     }\n\n#     # Split data into training and validation sets\n#     X_train_split, X_valid_split, y_train_split, y_valid_split = train_test_split(\n#         X_train_processed, y_train, test_size=0.2, random_state=42\n#     )\n\n#     # Train LightGBM model\n#     model = LGBMRegressor(**params)\n#     model.fit(X_train_split, y_train_split)\n\n#     # Predict on validation data\n#     y_pred = model.predict(X_valid_split)\n\n#     # Calculate RMSLE\n#     return rmsle(y_valid_split, y_pred)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Create an Optuna study to minimize RMSLE\n# study = optuna.create_study(direction='minimize')\n# study.optimize(objective, n_trials=50)\n\n# # Get and print the best parameters\n# lgb_best_params = study.best_params\n# print(\"Best Parameters:\", lgb_best_params)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CatBoost Optuna","metadata":{}},{"cell_type":"code","source":"# def objective(trial):\n#     # Define hyperparameter search space\n#     params = {\n#         'iterations': trial.suggest_int('iterations', 100, 1000),\n#         'depth': trial.suggest_int('depth', 4, 10),\n#         'learning_rate': trial.suggest_float('learning_rate', 0.01, 0.3),\n#         'random_strength': trial.suggest_float('random_strength', 0.1, 10.0),\n#         'bagging_temperature': trial.suggest_float('bagging_temperature', 0.0, 1.0),\n#         'border_count': trial.suggest_int('border_count', 1, 255),\n#         'l2_leaf_reg': trial.suggest_float('l2_leaf_reg', 1e-5, 100),\n#         'eval_metric': 'RMSE',\n#         'random_state': 0,\n#         'verbose': 0\n#     }\n\n#     # Split the training data into training and validation sets\n#     X_train_split, X_valid_split, y_train_split, y_valid_split = train_test_split(X_train_processed, y_train, test_size=0.2, random_state=0)\n\n#     # Train CatBoost model with current hyperparameters\n#     model = CatBoostRegressor(**params)\n#     model.fit(X_train_split, y_train_split, eval_set=(X_valid_split, y_valid_split), early_stopping_rounds=50)\n\n#     # Predict on validation set\n#     y_pred = model.predict(X_valid_split)\n\n#     # Calculate RMSLE\n#     rmsle_score = rmsle(y_valid_split, y_pred)\n#     return rmsle_score","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Optimize hyperparameters using Optuna\n# study = optuna.create_study(direction='minimize')  # Minimize RMSLE\n# study.optimize(objective, n_trials=50)\n\n# # Get best hyperparameters\n# cat_best_params = study.best_params\n# print(\"Best Hyperparameters:\", cat_best_params)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}