{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84896,"databundleVersionId":10305135,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Initial Settings","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:05:50.943488Z","iopub.execute_input":"2024-12-26T12:05:50.943837Z","iopub.status.idle":"2024-12-26T12:05:51.249503Z","shell.execute_reply.started":"2024-12-26T12:05:50.943811Z","shell.execute_reply":"2024-12-26T12:05:51.248653Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install scikit-learn==1.5.2","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:05:51.250618Z","iopub.execute_input":"2024-12-26T12:05:51.251115Z","iopub.status.idle":"2024-12-26T12:05:59.944773Z","shell.execute_reply.started":"2024-12-26T12:05:51.251082Z","shell.execute_reply":"2024-12-26T12:05:59.943801Z"},"jupyter":{"source_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%capture\n!pip install itables\n\nfrom itables import init_notebook_mode, show\ninit_notebook_mode(all_interactive=False,connected=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:05:59.946388Z","iopub.execute_input":"2024-12-26T12:05:59.94672Z","iopub.status.idle":"2024-12-26T12:06:04.160832Z","shell.execute_reply.started":"2024-12-26T12:05:59.946682Z","shell.execute_reply":"2024-12-26T12:06:04.159785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set seed for reproducibility \nseed = 42\nnp.random.seed(seed) \n\ntrain_df = pd.read_csv('/kaggle/input/playground-series-s4e12/train.csv',index_col=[0])\ntest_df = pd.read_csv('/kaggle/input/playground-series-s4e12/test.csv',index_col=[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:06:04.162353Z","iopub.execute_input":"2024-12-26T12:06:04.1628Z","iopub.status.idle":"2024-12-26T12:06:12.525719Z","shell.execute_reply.started":"2024-12-26T12:06:04.16277Z","shell.execute_reply":"2024-12-26T12:06:12.525065Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Log-Transforming the Target Variable","metadata":{}},{"cell_type":"markdown","source":"**Why Log-Transforming the Target Variable is Crucial**\n\nBefore delving into Exploratory Data Analysis (EDA), transforming the target variable, particularly using logarithmic transformation, can offer substantial benefits, especially when dealing with:\n\n* **Skewed Distributions:** Numerous real-world datasets, such as income, stock prices, or click counts, exhibit a skewed distribution. This implies a significant portion of the data is concentrated towards the lower end, while a smaller portion extends to very high values. Logarithmic transformation can effectively compress this range, resulting in a more symmetrical and manageable distribution for modeling purposes.\n\n* **Values Close to Zero:** When the target variable contains values close to zero, minor prediction errors can lead to substantial relative errors. Logarithmic transformation can mitigate this issue. By applying `log(y+1)`, we ensure the logarithm is always defined, even for values approaching zero.\n\n* **Enhancing Model Performance:** Many machine learning algorithms assume the data is normally distributed. By transforming the target variable to a more normal-like distribution, we can often enhance the performance of these models.\n\n* **Improved Interpretability:** In certain cases, log-transformed relationships might be more interpretable. For instance, in economics, growth rates are often modeled using logarithmic differences, as they represent proportional changes.\n\n**Note:**\n\n* Always remember to back-transform the predictions made on the log-transformed data to the original scale using `np.expm1()`.\n* Logarithmic transformation is just one of several possible transformations. Other transformations like square root or Box-Cox can also be effective depending on the specific characteristics of the data.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Define y\ntarget =  'Premium Amount'\n\n# Create a new column with the log-transformed values\ntrain_df['y_log'] = np.log1p(train_df[target]) \n\n# Create subplots\nfig, axes = plt.subplots(1, 2, figsize=(12, 4))\n\n# Plot the original distribution\nsns.histplot(train_df[target], kde=True, ax=axes[0], color='red', bins=40)\naxes[0].set_title('Original Distribution')\naxes[0].set_xlabel(str(target))\n\n# Plot the log-transformed distribution\nsns.histplot(train_df['y_log'], kde=True, ax=axes[1], color='green',bins=40)\naxes[1].set_title('Log-Transformed Distribution')\naxes[1].set_xlabel('log(y+1)')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:06:12.526557Z","iopub.execute_input":"2024-12-26T12:06:12.52677Z","iopub.status.idle":"2024-12-26T12:06:23.380281Z","shell.execute_reply.started":"2024-12-26T12:06:12.526751Z","shell.execute_reply":"2024-12-26T12:06:23.37928Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Data Analysis (EDA)\n\nExploratory Data Analysis (EDA) is a crucial step in the data science process. It involves summarizing the main characteristics of a dataset, often using visual methods. EDA helps us understand the data's structure, detect anomalies, identify patterns, and form hypotheses for further analysis.","metadata":{}},{"cell_type":"markdown","source":"## Data Overview","metadata":{}},{"cell_type":"code","source":"show(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:06:23.381137Z","iopub.execute_input":"2024-12-26T12:06:23.381718Z","iopub.status.idle":"2024-12-26T12:06:23.504714Z","shell.execute_reply.started":"2024-12-26T12:06:23.381676Z","shell.execute_reply":"2024-12-26T12:06:23.503909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:06:23.505778Z","iopub.execute_input":"2024-12-26T12:06:23.50615Z","iopub.status.idle":"2024-12-26T12:06:24.074563Z","shell.execute_reply.started":"2024-12-26T12:06:23.506116Z","shell.execute_reply":"2024-12-26T12:06:24.073678Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Missing Values","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\ndfs = [train_df, test_df]\nfor df in dfs:\n    # Create the heatmap\n    plt.figure(figsize=(18, 6))  # Adjust figure size for better readability\n    sns.heatmap(df.isnull(), cbar=True, cbar_kws={'label': 'Missing Values'}, yticklabels=False)\n    \n    # Customize the plot\n    plt.title('Missing Value Heatmap')\n    plt.xlabel('Features')\n    plt.ylabel('Observations')\n    \n    # Rotate x-axis labels for better visibility\n    plt.xticks(rotation=45, ha='right')\n    \n    # Show the plot\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:06:24.076906Z","iopub.execute_input":"2024-12-26T12:06:24.077179Z","iopub.status.idle":"2024-12-26T12:06:58.706187Z","shell.execute_reply.started":"2024-12-26T12:06:24.077157Z","shell.execute_reply":"2024-12-26T12:06:58.705204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate missing value counts and proportions\npd.concat([\n        train_df.isnull().sum(), \n        (train_df.isnull().sum() / train_df.shape[0]).round(2)], \n    axis=1, \n    keys=['Raw Count', 'Proportion']\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:06:58.707811Z","iopub.execute_input":"2024-12-26T12:06:58.708081Z","iopub.status.idle":"2024-12-26T12:06:59.816544Z","shell.execute_reply.started":"2024-12-26T12:06:58.708059Z","shell.execute_reply":"2024-12-26T12:06:59.815529Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Descriptive Statistics","metadata":{}},{"cell_type":"code","source":"# Select categorical and numerical columns (initial)\ncategorical_columns = train_df.select_dtypes(include=['object']).columns\nnumerical_columns = train_df.select_dtypes(exclude=['object']).columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:06:59.817568Z","iopub.execute_input":"2024-12-26T12:06:59.817909Z","iopub.status.idle":"2024-12-26T12:07:00.073671Z","shell.execute_reply.started":"2024-12-26T12:06:59.817872Z","shell.execute_reply":"2024-12-26T12:07:00.072902Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(train_df.describe(include='object'))\ndisplay(train_df.describe(exclude='object').round(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:00.074615Z","iopub.execute_input":"2024-12-26T12:07:00.074978Z","iopub.status.idle":"2024-12-26T12:07:02.531423Z","shell.execute_reply.started":"2024-12-26T12:07:00.074939Z","shell.execute_reply":"2024-12-26T12:07:02.53031Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Correlation","metadata":{}},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorrelation_matrix = train_df[numerical_columns].corr()\n\n# Plot the heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_matrix, annot=True, cmap=\"coolwarm\", cbar=True)\nplt.title(\"Correlation Heatmap\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:02.532531Z","iopub.execute_input":"2024-12-26T12:07:02.532932Z","iopub.status.idle":"2024-12-26T12:07:03.625311Z","shell.execute_reply.started":"2024-12-26T12:07:02.532877Z","shell.execute_reply":"2024-12-26T12:07:03.624392Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preliminary Fix","metadata":{}},{"cell_type":"code","source":"date_column = 'Policy Start Date'\n\n# Convert 'Policy Start Date' to date format and make year_only col\nfor df in dfs:\n    df[date_column] = pd.to_datetime(df[date_column]).dt.date\n    df[date_column+\"_year_only\"] = pd.to_datetime(df[date_column]).dt.year","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:03.626388Z","iopub.execute_input":"2024-12-26T12:07:03.626694Z","iopub.status.idle":"2024-12-26T12:07:04.989512Z","shell.execute_reply.started":"2024-12-26T12:07:03.626662Z","shell.execute_reply":"2024-12-26T12:07:04.988455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show(train_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:04.990512Z","iopub.execute_input":"2024-12-26T12:07:04.990842Z","iopub.status.idle":"2024-12-26T12:07:05.102667Z","shell.execute_reply.started":"2024-12-26T12:07:04.990809Z","shell.execute_reply":"2024-12-26T12:07:05.101889Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualize Distribution\n\nOne effective way to visualize the distribution of each feature in a dataset is by using histograms.\n\n**Why Use Histograms?**\n\n* **Distribution Insight:** Histograms provide a visual representation of the distribution of data points within a feature. This helps us understand the spread, central tendency, and variability of the data.\n* **Detecting Outliers:** By visualizing the data, we can easily spot outliers or unusual values that might need further investigation or preprocessing.\n* **Feature Engineering:** Understanding the distribution of features can guide feature engineering decisions, such as normalization, transformation, or binning.\n* **Comparing Features:** Multi-histogram plots allow us to compare the distributions of multiple features side by side, making it easier to identify relationships and differences between them.","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\nfrom IPython.display import clear_output\n\n# Hide deprecation warnings \nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)\n\n# Create a copy of the DataFrame to avoid modifying the original data\ndf = train_df.copy()\n\n# Handle NaN values by filling them with a placeholder\ndf[numerical_columns] = df[numerical_columns].fillna(df[numerical_columns].median())\ndf[categorical_columns] = df[categorical_columns].fillna('nan')\n\nfig, axes = plt.subplots(nrows=5, ncols=5, figsize=(15, 10))\ncolumns_to_plot = [col for col in df.columns if col != \"Policy Start Date\"] # Too many unique values\nfor i, column in enumerate(columns_to_plot):\n    clear_output(wait=False)\n    print(f'Processing {column}')\n    ax = axes.flatten()[i]\n    sns.histplot(df[column], ax=ax, bins=15, kde=False)\n    ax.set_title(column, fontsize=9)\n    ax.tick_params(axis='both', which='major', labelsize=6)\nplt.suptitle('Dataset Feature Distributions (train)', fontsize=11)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:05.103765Z","iopub.execute_input":"2024-12-26T12:07:05.104129Z","iopub.status.idle":"2024-12-26T12:07:32.625182Z","shell.execute_reply.started":"2024-12-26T12:07:05.104096Z","shell.execute_reply":"2024-12-26T12:07:32.62436Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Selection","metadata":{}},{"cell_type":"code","source":"# Split train data into features and target\nX = train_df.copy().drop('Policy Start Date',axis=1)\ny = X.pop('Premium Amount')\ny_log = X.pop('y_log')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:32.626236Z","iopub.execute_input":"2024-12-26T12:07:32.626593Z","iopub.status.idle":"2024-12-26T12:07:33.372989Z","shell.execute_reply.started":"2024-12-26T12:07:32.626557Z","shell.execute_reply":"2024-12-26T12:07:33.372215Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Leveraging MI and Tree-Based Feature Importance for Enhanced Feature Selection**\n\nThis approach combines two powerful techniques for a more robust and informative feature selection process:\n\n* **Mutual Information (MI):** \n    * Measures the **general dependency** between features and the target variable. \n    * Captures **non-linear relationships** effectively.\n    * Provides a **global perspective** on feature importance, independent of any specific model.\n\n* **Tree-Based Model Feature Importance:**\n    * Tree-based models (like Random Forest, Gradient Boosting) excel at capturing **complex interactions** within the data.\n    * Provide **model-specific feature importance scores**, reflecting how each feature contributes to the model's predictive power.\n\n**Synergistic Benefits:**\n\n* **Complementary Strengths:** By combining these methods, we leverage their complementary strengths: MI captures general dependencies, while tree-based models capture model-specific importance.\n* **Improved Ranking:** Combining scores (e.g., weighted averaging) provides a more robust and informative feature ranking.\n* **Enhanced Interpretability:**\n    * Comparing MI and tree-based scores can highlight:\n        * **Features with high MI but low model importance:** Might indicate redundant information or interactions the model doesn't utilize effectively.\n        * **Features with low MI but high model importance:** Might reveal complex, non-linear relationships that MI might not fully capture. \n* **Robustness:** This combined approach can be more robust to various data characteristics, such as non-linearity, high dimensionality, and the presence of interactions.\n","metadata":{}},{"cell_type":"markdown","source":"## Mutual Information","metadata":{}},{"cell_type":"markdown","source":"Mutual information (MI) is a measure of the dependency between two variables. It quantifies the amount of information obtained about one variable through the other variable. Unlike correlation, which measures linear relationships, mutual information can capture both linear and non-linear associations.\n\n**Key Points:**\n\n* **Mutual Information:** Measures the dependency between variables.\n* **Range:** MI values range from 0 (no dependency) to higher positive values (greater dependency).\n* **Non-linear Relationships:** MI can capture complex, non-linear relationships that correlation might miss.","metadata":{}},{"cell_type":"code","source":"X_mi = X.copy()\n\n# Select categorical and numerical columns (initial)\ncategorical_columns = X_mi.select_dtypes(include=['object']).columns\nnumerical_columns = X_mi.select_dtypes(exclude=['object']).columns\n\nX_mi[categorical_columns] = X_mi[categorical_columns].fillna('missing')\nX_mi[numerical_columns] = X_mi[numerical_columns].fillna(X_mi[numerical_columns].median())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:33.373825Z","iopub.execute_input":"2024-12-26T12:07:33.374096Z","iopub.status.idle":"2024-12-26T12:07:35.545069Z","shell.execute_reply.started":"2024-12-26T12:07:33.374073Z","shell.execute_reply":"2024-12-26T12:07:35.54434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Label encoding for categoricals\nfor colname in X_mi[categorical_columns]:\n    X_mi[colname], _ = X_mi[colname].factorize()\n\n# All discrete features should now have integer dtypes\ndiscrete_features = X_mi.dtypes == int","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:35.545893Z","iopub.execute_input":"2024-12-26T12:07:35.546223Z","iopub.status.idle":"2024-12-26T12:07:36.318008Z","shell.execute_reply.started":"2024-12-26T12:07:35.546189Z","shell.execute_reply":"2024-12-26T12:07:36.317272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show(X_mi)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:36.318709Z","iopub.execute_input":"2024-12-26T12:07:36.318961Z","iopub.status.idle":"2024-12-26T12:07:36.411725Z","shell.execute_reply.started":"2024-12-26T12:07:36.318928Z","shell.execute_reply":"2024-12-26T12:07:36.410993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.feature_selection import mutual_info_regression\n\n# Sample a subset of the data\nsampled_X = X_mi.sample(frac=0.5, random_state=seed)\n\n# Get the corresponding y values for the sampled X\nsampled_y = y[sampled_X.index] \n\ndef make_mi_scores(X, y, discrete_features, log_y=False):\n    \"\"\"\n    Calculates Mutual Information scores between features and the target variable.\n\n    Args:\n        X: DataFrame of features.\n        y: Series of target variable.\n        discrete_features: List of indices of discrete features.\n        log_y: Boolean, whether to use log-transformed target variable. Defaults to False.\n\n    Returns:\n        Series of Mutual Information scores, sorted in descending order.\n    \"\"\"\n    if log_y:\n        y = np.log1p(y) \n    mi_scores = mutual_info_regression(X, y, discrete_features=discrete_features, random_state=seed)\n    mi_scores = pd.Series(mi_scores, name=\"MI Scores\", index=X.columns)\n    mi_scores = mi_scores.sort_values(ascending=False)\n    return mi_scores\n\nmi_scores = make_mi_scores(sampled_X, sampled_y, discrete_features, log_y=False) \nselected_features = mi_scores[:10].keys()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:07:36.412583Z","iopub.execute_input":"2024-12-26T12:07:36.412775Z","iopub.status.idle":"2024-12-26T12:09:06.669049Z","shell.execute_reply.started":"2024-12-26T12:07:36.412758Z","shell.execute_reply":"2024-12-26T12:09:06.667943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mi_scores[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:09:06.670069Z","iopub.execute_input":"2024-12-26T12:09:06.670628Z","iopub.status.idle":"2024-12-26T12:09:06.677036Z","shell.execute_reply.started":"2024-12-26T12:09:06.670603Z","shell.execute_reply":"2024-12-26T12:09:06.67621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_mi_scores(X, y, scores, log_y=False):\n    \"\"\"\n    Plots Mutual Information scores.\n\n    Args:\n        scores: Series of Mutual Information scores.\n        log_y: Boolean, whether to include a plot for log-transformed target. \n               Defaults to False.\n    \"\"\"\n    scores = scores.sort_values(ascending=True)\n    width = np.arange(len(scores))\n    ticks = list(scores.index)\n\n    if log_y:\n        fig, axes = plt.subplots(1, 2, figsize=(12, 5))\n        axes[0].barh(width, scores)\n        axes[0].set_yticks(width)\n        axes[0].set_yticklabels(ticks)\n        axes[0].set_title(\"Mutual Information Scores (Original y)\")\n\n        # Calculate MI scores for log(y+1)\n        log_y_scores = make_mi_scores(X, np.log1p(y), discrete_features, log_y=False) \n        log_y_scores = log_y_scores.sort_values(ascending=True)\n        axes[1].barh(width, log_y_scores)\n        axes[1].set_yticks(width)\n        axes[1].set_yticklabels(ticks)\n        axes[1].set_title(\"Mutual Information Scores (log(y+1))\")\n    else:\n        plt.figure(dpi=100, figsize=(8, 5))\n        plt.barh(width, scores)\n        plt.yticks(width, ticks)\n        plt.title(\"Mutual Information Scores\")\n\n    plt.show()\n\nplot_mi_scores(sampled_X, sampled_y, mi_scores, log_y=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:09:06.677821Z","iopub.execute_input":"2024-12-26T12:09:06.678106Z","iopub.status.idle":"2024-12-26T12:10:39.228738Z","shell.execute_reply.started":"2024-12-26T12:09:06.678085Z","shell.execute_reply":"2024-12-26T12:10:39.227798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selected_features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:11:13.39605Z","iopub.execute_input":"2024-12-26T12:11:13.396414Z","iopub.status.idle":"2024-12-26T12:11:13.401721Z","shell.execute_reply.started":"2024-12-26T12:11:13.396385Z","shell.execute_reply":"2024-12-26T12:11:13.400829Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# DNN","metadata":{}},{"cell_type":"code","source":"import optuna\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import root_mean_squared_log_error\n\nimport tensorflow\nfrom tensorflow import keras\nfrom tensorflow.keras.layers import Dense, BatchNormalization, Dropout\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.losses import MeanSquaredLogarithmicError\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.callbacks import EarlyStopping","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:11:19.287338Z","iopub.execute_input":"2024-12-26T12:11:19.287665Z","iopub.status.idle":"2024-12-26T12:11:19.292377Z","shell.execute_reply.started":"2024-12-26T12:11:19.28764Z","shell.execute_reply":"2024-12-26T12:11:19.291443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Split the data into training and testing sets\nfiltered_X = X[selected_features]\n\nX_train, X_valid, y_train, y_valid = train_test_split(filtered_X[selected_features], y_log, test_size=0.3, random_state=seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:11:22.073378Z","iopub.execute_input":"2024-12-26T12:11:22.073688Z","iopub.status.idle":"2024-12-26T12:11:22.562193Z","shell.execute_reply.started":"2024-12-26T12:11:22.073665Z","shell.execute_reply":"2024-12-26T12:11:22.561424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show(X_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:11:26.40859Z","iopub.execute_input":"2024-12-26T12:11:26.408948Z","iopub.status.idle":"2024-12-26T12:11:26.541985Z","shell.execute_reply.started":"2024-12-26T12:11:26.408896Z","shell.execute_reply":"2024-12-26T12:11:26.540991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"display(X_train.describe(exclude='object'))\ndisplay(X_train.describe(include='object'))\n\nimport pandas as pd\n\n# Display unique values for each object-type column\nfor col in X_train.select_dtypes(include='object').columns:\n    unique_vals = X_train[col].unique()\n    print(f\"Unique values in '{col}': {unique_vals}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:11:29.714584Z","iopub.execute_input":"2024-12-26T12:11:29.71502Z","iopub.status.idle":"2024-12-26T12:11:30.454595Z","shell.execute_reply.started":"2024-12-26T12:11:29.714973Z","shell.execute_reply":"2024-12-26T12:11:30.453784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.impute import KNNImputer, SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder, OneHotEncoder, StandardScaler, MinMaxScaler \nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.ensemble import RandomForestRegressor\n\n# Define the categorical columns \nordinal_features = ['Education Level','Customer Feedback'] \nnominal_features = ['Marital Status']\nnumeric_featues = [col for col in X_train.columns if col not in ordinal_features + nominal_features]\n\n\n# Define the order for ordinal features \neducation_level_order = [\"High School\", \"Bachelor's\", \"Master's\", \"PhD\"] \ncustomer_feedback_order = ['Poor', 'Average', 'Good']\n\n# Define the preprocessing for numerical and categorical features\nnumerical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    # ('imputer', KNNImputer()),\n    ('scaler', MinMaxScaler())\n])\n\nordinal_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('ordinal', OrdinalEncoder(categories=[education_level_order,customer_feedback_order]))\n])\n\nnominal_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Combine preprocessing steps\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numeric_featues),\n        ('ord', ordinal_transformer, ordinal_features),\n        ('nom', nominal_transformer, nominal_features)\n    ]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:15:14.258166Z","iopub.execute_input":"2024-12-26T12:15:14.258514Z","iopub.status.idle":"2024-12-26T12:15:14.298131Z","shell.execute_reply.started":"2024-12-26T12:15:14.258491Z","shell.execute_reply":"2024-12-26T12:15:14.297225Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# https://stackoverflow.com/questions/43855162/rmse-rmsle-loss-function-in-keras\n# RMSLE for TF\ndef rmsle(y_true, y_pred):\n    msle = MeanSquaredLogarithmicError()\n    return K.sqrt(msle(y_true, y_pred)) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:15:15.977284Z","iopub.execute_input":"2024-12-26T12:15:15.977964Z","iopub.status.idle":"2024-12-26T12:15:15.981894Z","shell.execute_reply.started":"2024-12-26T12:15:15.977932Z","shell.execute_reply":"2024-12-26T12:15:15.981154Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Define Objective Function for Optuna (DNN)","metadata":{}},{"cell_type":"code","source":"# Set seed for TensorFlow \ntensorflow.random.set_seed(seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:15:18.919182Z","iopub.execute_input":"2024-12-26T12:15:18.919489Z","iopub.status.idle":"2024-12-26T12:15:18.923294Z","shell.execute_reply.started":"2024-12-26T12:15:18.919464Z","shell.execute_reply":"2024-12-26T12:15:18.922431Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_model(trial):\n    \"\"\"\n    Creates a DNN model with hyperparameters tuned by Optuna.\n\n    Args:\n        trial: An Optuna trial object.\n\n    Returns:\n        A compiled Keras model.\n    \"\"\"\n    # Define hyperparameters\n    num_layers = trial.suggest_int('num_layers', 1, 6)\n    units = trial.suggest_int('units', 32, 128, step=32) \n    learning_rate = trial.suggest_float('learning_rate', 1e-4, 1e-2, log=True)\n    dropout_rate = trial.suggest_float('dropout_rate', 0.0, 0.5) \n\n    # Create the model\n    model = Sequential()\n    for _ in range(num_layers):\n        model.add(Dense(units, activation='relu'))\n        model.add(BatchNormalization())  # Add Batch Normalization\n        model.add(Dropout(dropout_rate))  # Add Dropout\n    model.add(Dense(1))  # Output layer\n\n    # Compile the model with RMSLE loss\n    model.compile(loss=rmsle, metrics=[rmsle], optimizer=keras.optimizers.Adam(learning_rate=learning_rate)) \n\n    return model\n\ndef objective(trial):\n    \"\"\"\n    Objective function for Optuna optimization.\n\n    Args:\n        trial: An Optuna trial object.\n\n    Returns:\n        The validation loss.\n    \"\"\"\n    # Create the model\n    model = create_model(trial)\n    early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n    # Preprocess the data \n    X_train_preprocessed = preprocessor.fit_transform(X_train) \n    X_valid_preprocessed = preprocessor.transform(X_valid)\n    \n    # # Create a pipeline with preprocessing and model\n    # pipeline = Pipeline(steps=[\n    #     ('preprocessor', preprocessor),\n    #     ('classifier', model)\n    # ])\n    \n    # Train the model with early stopping\n    # pipeline.fit(X_train_preprocessed, y_train, classifier__epochs=50, classifier__batch_size=2048, classifier__verbose=0, classifier__callbacks=[early_stopping], classifier__validation_data=(X_valid_preprocessed, y_valid))\n    model.fit(X_train_preprocessed, y_train, epochs=50, batch_size=2048, verbose=0, callbacks=[early_stopping], validation_data=(X_valid_preprocessed, y_valid))\n    preds = model.predict(X_valid_preprocessed)\n\n    rmsle = root_mean_squared_log_error(y_valid, preds)\n    return rmsle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T11:58:27.479625Z","iopub.execute_input":"2024-12-26T11:58:27.479906Z","iopub.status.idle":"2024-12-26T11:58:27.487533Z","shell.execute_reply.started":"2024-12-26T11:58:27.479885Z","shell.execute_reply":"2024-12-26T11:58:27.486344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess the data \nX_train_preprocessed = preprocessor.fit_transform(X_train) \nX_valid_preprocessed = preprocessor.transform(X_valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:15:24.680459Z","iopub.execute_input":"2024-12-26T12:15:24.680786Z","iopub.status.idle":"2024-12-26T12:15:26.483551Z","shell.execute_reply.started":"2024-12-26T12:15:24.680762Z","shell.execute_reply":"2024-12-26T12:15:26.482767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_model(trial):\n    \"\"\"\n    Creates a DNN model with hyperparameters tuned by Optuna.\n\n    Args:\n        trial: An Optuna trial object.\n\n    Returns:\n        A compiled Keras model.\n    \"\"\"\n    # Define hyperparameters\n    num_layers = trial.suggest_int('num_layers', 1, 6)\n    units = trial.suggest_int('units', 32, 128, step=32) \n    learning_rate = trial.suggest_float('learning_rate', 1e-4, 1e-2, log=True)\n    dropout_rate = trial.suggest_float('dropout_rate', 0.0, 0.5) \n\n    # Create the model\n    model = Sequential()\n    for _ in range(num_layers):\n        model.add(Dense(units, activation='relu'))\n        model.add(BatchNormalization())  # Add Batch Normalization\n        model.add(Dropout(dropout_rate))  # Add Dropout\n    model.add(Dense(1))  # Output layer\n\n    # Compile the model with RMSLE loss\n    model.compile(loss=rmsle, metrics=[rmsle], optimizer=keras.optimizers.Adam(learning_rate=learning_rate)) \n\n    return model\n\ndef objective(trial):\n    \"\"\"\n    Objective function for Optuna optimization.\n\n    Args:\n        trial: An Optuna trial object.\n\n    Returns:\n        The validation loss.\n    \"\"\"\n    # Create the model\n    model = create_model(trial)\n    early_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n    \n    # Train the model with early stopping\n    # pipeline.fit(X_train_preprocessed, y_train, classifier__epochs=50, classifier__batch_size=2048, classifier__verbose=0, classifier__callbacks=[early_stopping], classifier__validation_data=(X_valid_preprocessed, y_valid))\n    model.fit(X_train_preprocessed, y_train, epochs=50, batch_size=2048, verbose=0, callbacks=[early_stopping], validation_data=(X_valid_preprocessed, y_valid))\n    preds = model.predict(X_valid_preprocessed)\n\n    rmsle = root_mean_squared_log_error(y_valid, np.maximum(preds, 0))\n    return rmsle","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T12:15:31.798865Z","iopub.execute_input":"2024-12-26T12:15:31.799198Z","iopub.status.idle":"2024-12-26T12:15:31.806084Z","shell.execute_reply.started":"2024-12-26T12:15:31.799176Z","shell.execute_reply":"2024-12-26T12:15:31.805158Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Hyperparameter Tuning","metadata":{}},{"cell_type":"code","source":"# # Create a study object\n# study = optuna.create_study(direction='minimize')\n\n# # Optimize hyperparameters\n# study.optimize(objective, n_trials=100)  # Adjust n_trials as needed\n\n# # Get the best hyperparameters\n# best_params = study.best_params\n# print(f'Best Params: {best_params}')\n# print(\"Best trial:\")\n# trial = study.best_trial \n# print(\"  Value: {}\".format(trial.value))\n# print(\"  Params: \")\n# for key, value in trial.params.items():\n#     print(\"    {}: {}\".format(key, value))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T14:36:25.847512Z","iopub.execute_input":"2024-12-26T14:36:25.847847Z","iopub.status.idle":"2024-12-26T14:36:25.852053Z","shell.execute_reply.started":"2024-12-26T14:36:25.847818Z","shell.execute_reply":"2024-12-26T14:36:25.850852Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Trial 53 finished with value: 0.1603104416567761 and parameters: {'num_layers': 6, 'units': 128, 'learning_rate': 0.00192388666506763, 'dropout_rate': 0.01837882905165548}","metadata":{}},{"cell_type":"code","source":"# optuna.visualization.plot_param_importances(study)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T14:36:20.668791Z","iopub.status.idle":"2024-12-26T14:36:20.669102Z","shell.execute_reply":"2024-12-26T14:36:20.668983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# study.best_params\nbest_params =  {'num_layers': 6, 'units': 128, 'learning_rate': 0.00192388666506763, 'dropout_rate': 0.01837882905165548}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T14:36:36.302737Z","iopub.execute_input":"2024-12-26T14:36:36.303116Z","iopub.status.idle":"2024-12-26T14:36:36.307475Z","shell.execute_reply.started":"2024-12-26T14:36:36.303086Z","shell.execute_reply":"2024-12-26T14:36:36.306244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the best_model\nbest_model = Sequential()\nfor _ in range(best_params['num_layers']):\n    best_model.add(Dense(best_params['units'], activation='relu'))\n    best_model.add(BatchNormalization())  # Add Batch Normalization\n    best_model.add(Dropout(best_params['dropout_rate']))  # Add Dropout\nbest_model.add(Dense(1))  # Output layer\n\n# Compile the best_model with RMSLE loss\nbest_model.compile(loss=rmsle, metrics=[rmsle], optimizer=keras.optimizers.Adam(learning_rate=best_params['learning_rate'])) \n\n# Define early stopping callback\nearly_stopping = EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True)\n\n# Create a pipeline with preprocessing and model\nfinal_pipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('classifier', best_model)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T14:36:38.338641Z","iopub.execute_input":"2024-12-26T14:36:38.338972Z","iopub.status.idle":"2024-12-26T14:36:38.372483Z","shell.execute_reply.started":"2024-12-26T14:36:38.338909Z","shell.execute_reply":"2024-12-26T14:36:38.371507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare Data\nfull_filtered_X = train_df[selected_features]\ny = train_df['Premium Amount']\ny_log = train_df['y_log']\n\nfiltered_test = test_df[selected_features]\n\n# Fit the pipeline to the full filtered training data (using y_log for consistency)\nfinal_pipeline.fit(full_filtered_X, y_log, classifier__epochs=50, classifier__batch_size=2048, classifier__verbose=0, classifier__callbacks=[early_stopping], classifier__validation_split=0.2)\n\n# Make predictions on the filtered test data\npreds_log = final_pipeline.predict(filtered_test)\n\n# Back-transform predictions to the original scale\npreds = np.expm1(preds_log).flatten()  # Back-transform using np.expm1()\n\n# Create the submission DataFrame using the correct index\nsubmission_df = pd.DataFrame({'id': test_df.index, 'Premium Amount': preds})\n\n# Save the submission DataFrame to a CSV file\n# submission_df.to_csv('regression_insurance_dataset_DNN_optuna.csv', index=False)  # Avoid including the index in the CSV\nsubmission_df.to_csv('submission.csv', index=False)  # Avoid including the index in the CSV\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-26T14:40:15.581197Z","iopub.execute_input":"2024-12-26T14:40:15.581486Z","iopub.status.idle":"2024-12-26T14:40:16.575837Z","shell.execute_reply.started":"2024-12-26T14:40:15.581465Z","shell.execute_reply":"2024-12-26T14:40:16.57502Z"}},"outputs":[],"execution_count":null}]}